{"meta":{"query_hash":"7de042e048ae","filters":{"venue":"ACM Transactions on Software Engineering and Methodology"},"cohort_total":200,"direct_labels_cover":1,"predictions_cover":200,"exported":200,"export_cap":100000,"truncated":false,"label_status":"direct model label, unvalidated","prediction_status":"machine_predicted_unvalidated (Codex and Gemma teacher distillation)","score_status":"score_only:v0-immature-baseline","snapshot":{"source":"OpenAlex, pinned release, all 482 partitions","release":"2026-06-24","frame_built":"2026-07-12"},"permalink":"https://metacan.xera.ac/q/7de042e048ae","api":"https://metacan.xera.ac/api/v1/cohort?venue=ACM+Transactions+on+Software+Engineering+and+Methodology"},"results":[{"id":"W1794343297","doi":"10.1145/2699697","title":"aToucan","year":2015,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":98,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Sequence diagram; Use Case Diagram; Class diagram; Traceability; Programming language; Unified Modeling Language; Software engineering; Completeness (order theory); Activity diagram; Class (philosophy); Requirements traceability; Consistency (knowledge bases); Requirements analysis; Artificial intelligence; Software; Requirement","score_opus":0.14914112278278302,"score_gpt":0.34295324255414494,"score_spread":0.19381211977136192,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1794343297","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007277597,0.0029493726,0.06000167,0.0053345053,0.0033235564,0.0006968992,0.013328424,0.060694784,0.84639317],"genre_scores_gemma":[0.051556498,0.0028765588,0.055575524,0.003289152,0.00067224365,0.0010976692,0.03215849,0.017586358,0.83518755],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972174,0.0004171599,0.00014443736,0.00077047804,0.0011493491,0.000301141],"domain_scores_gemma":[0.99575496,0.00082780694,0.00023498244,0.0008967222,0.001357749,0.00092776626],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020506058,0.0014906581,0.001106242,0.0026961353,0.002429506,0.008143694,0.003230916,0.003007208,0.472743],"category_scores_gemma":[0.0065023564,0.00077793625,0.0010124525,0.0020490629,0.0010032314,0.0047964845,0.005487242,0.0026513769,0.2882946],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010109887,0.0002515735,0.002122747,0.0008856338,0.00005238814,0.0007845222,0.00063026225,0.001369114,0.0067453873,0.042520974,0.5563378,0.38728866],"study_design_scores_gemma":[0.0000796728,0.00007191285,0.0006198965,0.00012577121,0.000022006938,0.00036203253,0.00014112925,0.0017254468,0.002057321,0.0064101894,0.9883504,0.000034367287],"about_ca_topic_score_codex":0.0044008377,"about_ca_topic_score_gemma":0.0051028556,"teacher_disagreement_score":0.472743,"about_ca_system_score_codex":0.0017164267,"about_ca_system_score_gemma":0.0042969105,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W1975286358","doi":"10.1145/1101815.1101819","title":"Reasoning about static and dynamic properties in alloy","year":2005,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Formal Methods in Verification","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"","keywords":"Iterated function; Computer science; Relational calculus; Formalism (music); Programming language; Calculus (dental); Theoretical computer science; Relational model; Mathematics; Relational database; Data mining","score_opus":0.06189680358928729,"score_gpt":0.3087775904416149,"score_spread":0.24688078685232762,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1975286358","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023836017,0.0003557074,0.9679267,0.00036121527,0.000047215097,0.00007211213,0.00018985233,0.0010541821,0.006157085],"genre_scores_gemma":[0.41803682,0.0010867079,0.57243896,0.00030343444,0.00019383316,0.00016157371,0.0010461077,0.00037778533,0.0063547418],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9967981,0.0007120604,0.00031573843,0.00044367832,0.0013716375,0.00035885838],"domain_scores_gemma":[0.99646676,0.0023070248,0.00030151726,0.000416023,0.0004359736,0.00007276393],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0032282271,0.00095533184,0.0007480868,0.0023052262,0.0012287537,0.00402129,0.0019314885,0.0010252154,0.0019523603],"category_scores_gemma":[0.0064898925,0.00092455465,0.0030710497,0.0013162388,0.002862093,0.006222242,0.0024533782,0.0016874501,0.0006423805],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000056164878,0.000052117197,0.0011009262,0.00015353826,0.000058209065,0.00068046025,0.0012758843,0.049413897,0.005374251,0.9133048,0.0009383619,0.027591381],"study_design_scores_gemma":[0.000036256773,0.000081584905,0.0004937605,0.00009282734,0.00018789948,0.00035901088,0.00047012986,0.29524064,0.017487016,0.6583613,0.027120482,0.00006902343],"about_ca_topic_score_codex":0.010280525,"about_ca_topic_score_gemma":0.010423162,"teacher_disagreement_score":0.010280525,"about_ca_system_score_codex":0.0019997442,"about_ca_system_score_gemma":0.0018586027,"threshold_uncertainty_score":0.020441353},"labels":[],"label_agreement":null},{"id":"W1975318342","doi":"10.1145/2512207","title":"Degree-of-knowledge","year":2014,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":61,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; International Business Machines Corporation","keywords":"Codebase; Computer science; Robustness (evolution); Source code; Code (set theory); Code review; Point (geometry); Software; Software engineering; Software development; Static program analysis; Programming language; Set (abstract data type)","score_opus":0.11510190202852359,"score_gpt":0.3348329567299539,"score_spread":0.21973105470143028,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1975318342","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.056185424,0.0005019475,0.9034237,0.0021681276,0.00008193305,0.0002288064,0.0009312434,0.0004868903,0.03599189],"genre_scores_gemma":[0.8683514,0.0004428699,0.12172078,0.00023448032,0.00013034153,0.00035541487,0.0007136973,0.00012605479,0.007924987],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9890665,0.0030448872,0.0007625224,0.0032458145,0.0030716117,0.00080874766],"domain_scores_gemma":[0.9524715,0.02929797,0.0035716859,0.008569116,0.0044339527,0.0016556348],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006494386,0.0010347078,0.001257253,0.0044274637,0.0014351876,0.0054447385,0.0031945945,0.0026362124,0.008682928],"category_scores_gemma":[0.05491863,0.0007869877,0.0017768416,0.0037241227,0.0043905103,0.014377599,0.003775098,0.0023293323,0.001555151],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020481006,0.00019895373,0.021840807,0.00041765615,0.00022879655,0.00034748908,0.002368189,0.07776704,0.0016529579,0.7810506,0.0049548205,0.10896787],"study_design_scores_gemma":[0.00003744187,0.00010592884,0.0058447276,0.000100478675,0.00010887673,0.000627529,0.00048046547,0.19349031,0.0011053799,0.7793666,0.018662585,0.00006974406],"about_ca_topic_score_codex":0.004637455,"about_ca_topic_score_gemma":0.0033858228,"teacher_disagreement_score":0.008682928,"about_ca_system_score_codex":0.0031620504,"about_ca_system_score_gemma":0.0016820568,"threshold_uncertainty_score":0.034346044},"labels":[],"label_agreement":null},{"id":"W1977244105","doi":"10.1145/2559936","title":"Automated cookie collection testing","year":2014,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta; The King's University","funders":"","keywords":"Computer science; Web testing; Web application; Web application security; Security testing; Software engineering; World Wide Web; Web development; Web service; Operating system; Cloud computing","score_opus":0.08726401518129814,"score_gpt":0.3104312761742673,"score_spread":0.22316726099296916,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1977244105","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4839038,0.0009765662,0.46978024,0.00035099464,0.00014084086,0.00090808864,0.0012172062,0.026184622,0.016537528],"genre_scores_gemma":[0.8801357,0.00022849235,0.11233967,0.00018106736,0.000026391554,0.00031217726,0.0013023212,0.00058814837,0.0048859594],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9953499,0.00096770644,0.00021365225,0.0006838789,0.0024059846,0.00037883534],"domain_scores_gemma":[0.9845718,0.008094545,0.0013963344,0.0029258835,0.002731422,0.000279973],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013070145,0.0010756274,0.0006867674,0.0020648637,0.0004979591,0.0010974501,0.0014814029,0.0007557955,0.004114399],"category_scores_gemma":[0.010545782,0.00038622908,0.0004897864,0.0011142652,0.00079659215,0.0013664634,0.0012050867,0.00070267613,0.0011269667],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00094921776,0.00094733125,0.024419079,0.0005958052,0.00012352901,0.0010809035,0.00080059026,0.056714185,0.16883667,0.008604217,0.010272431,0.726656],"study_design_scores_gemma":[0.00023110172,0.0022208903,0.035064336,0.00022423961,0.00015008533,0.0027224845,0.0004994121,0.6159855,0.28911078,0.021313023,0.032221515,0.00025659302],"about_ca_topic_score_codex":0.0043498846,"about_ca_topic_score_gemma":0.0046205134,"teacher_disagreement_score":0.0043498846,"about_ca_system_score_codex":0.0007255694,"about_ca_system_score_gemma":0.0017783832,"threshold_uncertainty_score":0.013764024},"labels":[],"label_agreement":null},{"id":"W1982335236","doi":"10.1145/2594458","title":"Peer Review on Open-Source Software Projects","year":2014,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":108,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria; Concordia University","funders":"","keywords":"Computer science; Software technical review; Technical peer review; Peer review; Quality assurance; Software peer review; Quality (philosophy); Empirical research; Software; Software quality assurance; Data science; Knowledge management; Software development; Software quality; Operations management; Engineering; Software construction; Political science","score_opus":0.1125589028151714,"score_gpt":0.34279438878182694,"score_spread":0.23023548596665555,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1982335236","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9018621,0.0045725284,0.044406664,0.003036129,0.00042241268,0.0028268262,0.0014744342,0.0008778266,0.040521033],"genre_scores_gemma":[0.97627807,0.0014796562,0.014740908,0.00023075209,0.000311242,0.0009435561,0.00074128865,0.00014472136,0.0051297126],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.82056373,0.07552339,0.014820913,0.010321692,0.075677715,0.003092597],"domain_scores_gemma":[0.20097888,0.5094696,0.12035548,0.045984138,0.114658594,0.008553343],"candidate_categories":["metaresearch","open_science"],"consensus_categories":[],"category_scores_codex":[0.08174722,0.00046088258,0.0013104235,0.012102654,0.0037756055,0.0050780587,0.0024620255,0.0014096395,0.004320083],"category_scores_gemma":[0.5293169,0.00068687997,0.00057044864,0.008061309,0.0038146519,0.0059214397,0.005668247,0.0014682966,0.0015680336],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008520453,0.0003926612,0.36960888,0.0031960597,0.00036902726,0.0014797794,0.041083388,0.004671469,0.005527455,0.017911658,0.018559702,0.5363479],"study_design_scores_gemma":[0.00018578391,0.0018181319,0.78594035,0.0027182177,0.0002697374,0.0019046091,0.026149392,0.01949151,0.009230405,0.024347508,0.12749505,0.00044936175],"about_ca_topic_score_codex":0.0027367754,"about_ca_topic_score_gemma":0.0034577653,"teacher_disagreement_score":0.997538,"about_ca_system_score_codex":0.0028567852,"about_ca_system_score_gemma":0.0074911523,"threshold_uncertainty_score":0.43232578},"labels":[],"label_agreement":null},{"id":"W1987035533","doi":"10.1145/1391984.1391987","title":"Evaluating the benefits of context-sensitive points-to analysis using a BDD-based implementation","year":2008,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":117,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University; University of Waterloo","funders":"","keywords":"Computer science; Heap (data structure); Pointer analysis; Call graph; Java; Pointer (user interface); Programming language; Paddle; Theoretical computer science; Static analysis; Artificial intelligence; Operating system","score_opus":0.22449726015584578,"score_gpt":0.4221343935073883,"score_spread":0.1976371333515425,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1987035533","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.256029,0.0010401562,0.71650904,0.00043468407,0.00013841807,0.00029151127,0.0005001547,0.018459378,0.0065975673],"genre_scores_gemma":[0.60130996,0.0003024575,0.39598802,0.0001462303,0.000021241236,0.0001238183,0.0003793913,0.0007469842,0.000981873],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9937617,0.0019180984,0.00044434378,0.0005512314,0.002903679,0.0004208953],"domain_scores_gemma":[0.98484576,0.009851998,0.00060625497,0.0026822079,0.001825813,0.00018790684],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044389395,0.0013347734,0.0009597012,0.0018409297,0.0006144313,0.0021188206,0.0023762668,0.0012029146,0.0025557068],"category_scores_gemma":[0.021809414,0.001051126,0.0012164488,0.001522847,0.0013705532,0.003356234,0.0016787898,0.0016183348,0.0005650828],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0035745967,0.000685513,0.015997842,0.001441418,0.0004715285,0.00039671746,0.00056577916,0.55773777,0.06500242,0.06154757,0.004244809,0.28833407],"study_design_scores_gemma":[0.00023928605,0.00040106862,0.0011857579,0.00006190225,0.00017264245,0.0001004424,0.000073029085,0.9388827,0.04317279,0.012575103,0.003073983,0.00006131453],"about_ca_topic_score_codex":0.006658626,"about_ca_topic_score_gemma":0.0057504675,"teacher_disagreement_score":0.006658626,"about_ca_system_score_codex":0.001415097,"about_ca_system_score_gemma":0.0025845536,"threshold_uncertainty_score":0.023475647},"labels":[],"label_agreement":null},{"id":"W1988506401","doi":"10.1145/2089116.2089119","title":"Weak Alphabet Merging of Partial Behavior Models","year":2012,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Formal Methods in Verification","field":"Computer Science","cited_by":32,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Viewpoints; Alphabet; Property (philosophy); Process (computing); Algebraic properties; Theoretical computer science; Component (thermodynamics); Extension (predicate logic); Programming language; Mathematics","score_opus":0.13249928710046624,"score_gpt":0.34050713105717073,"score_spread":0.2080078439567045,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W1988506401","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.036602706,0.00005314432,0.9605004,0.00012003173,0.000015407728,0.00011531257,0.00009877506,0.00097013236,0.0015240354],"genre_scores_gemma":[0.39471945,0.0001322053,0.60127467,0.00011092108,0.000017524573,0.00028726403,0.00057784235,0.00029939867,0.0025807822],"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99439573,0.0018794325,0.00057118444,0.0008441873,0.0018462606,0.00046323825],"domain_scores_gemma":[0.9872285,0.005640464,0.00090724364,0.004245615,0.0016653015,0.0003128934],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004931708,0.00078713585,0.000840826,0.0015940989,0.0011328431,0.00214246,0.002451977,0.0011165102,0.0024831034],"category_scores_gemma":[0.016458848,0.00094432884,0.002671444,0.0010041604,0.002195061,0.0048691593,0.0049718013,0.0019408208,0.0005151603],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005860697,0.00025274535,0.008190339,0.00059727853,0.00024970787,0.0012610072,0.004370173,0.428552,0.04776376,0.36652547,0.0013627193,0.14028876],"study_design_scores_gemma":[0.00003626349,0.00021041559,0.00064061757,0.000088274,0.00016238613,0.00020941005,0.0005015284,0.7514014,0.035468712,0.20261613,0.008603143,0.00006171552],"about_ca_topic_score_codex":0.0043023797,"about_ca_topic_score_gemma":0.006044613,"teacher_disagreement_score":0.004931708,"about_ca_system_score_codex":0.0015173359,"about_ca_system_score_gemma":0.002760039,"threshold_uncertainty_score":0.026081622},"labels":[],"label_agreement":null},{"id":"W2004901564","doi":"10.1145/1243987.1243989","title":"Metamodel-based model conformance and multiview consistency checking","year":2007,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Model-Driven Software Engineering Techniques","field":"Computer Science","cited_by":125,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Engineering and Physical Sciences Research Council; European Commission","keywords":"Metamodeling; Computer science; Sequence diagram; Unified Modeling Language; Conformance checking; Consistency (knowledge bases); Model checking; Programming language; Software engineering; Completeness (order theory); Applications of UML; Consistency model; Automation; Artificial intelligence; Work in process","score_opus":0.09095907436759029,"score_gpt":0.31688830244321353,"score_spread":0.22592922807562324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2004901564","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0029158625,0.000097523814,0.9950937,0.00014821166,0.000023352917,0.000052163712,0.00003604595,0.00073018495,0.0009029458],"genre_scores_gemma":[0.14191103,0.00030836216,0.8551917,0.0002328539,0.000062173385,0.00032806126,0.00032510774,0.00048879284,0.0011519692],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9777223,0.009087015,0.0015654955,0.0028921757,0.0076988847,0.0010341035],"domain_scores_gemma":[0.96282464,0.018736972,0.0029234642,0.010843972,0.004294383,0.00037643904],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.018057382,0.0013029439,0.001824969,0.003159142,0.0012991292,0.0038429527,0.004879058,0.0023922215,0.0036210231],"category_scores_gemma":[0.05480262,0.0014373583,0.0039839065,0.0021916754,0.0045817886,0.007047303,0.0057754368,0.0040081586,0.0007465453],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003748919,0.0001871424,0.0035554357,0.0006905149,0.0003670196,0.00055943016,0.0011683702,0.29949397,0.010625099,0.50755715,0.0036447195,0.17177625],"study_design_scores_gemma":[0.00011431464,0.0001171144,0.0004273902,0.00026710474,0.00016837427,0.00034381123,0.00016244425,0.62576634,0.018375864,0.34231952,0.011846015,0.000091679445],"about_ca_topic_score_codex":0.0034974674,"about_ca_topic_score_gemma":0.0034577518,"teacher_disagreement_score":0.018057382,"about_ca_system_score_codex":0.0027436074,"about_ca_system_score_gemma":0.0038080146,"threshold_uncertainty_score":0.09549767},"labels":[],"label_agreement":null},{"id":"W2014744822","doi":"10.1145/2491509.2491521","title":"Use case and task models","year":2013,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Software engineering; Task (project management); Software development; Set (abstract data type); Software; Human–computer interaction; Systems engineering; Programming language","score_opus":0.1401807566114291,"score_gpt":0.3052749112876002,"score_spread":0.1650941546761711,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2014744822","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015703363,0.0003314895,0.9255687,0.0015242664,0.00010907214,0.0010928836,0.0012973469,0.0010616104,0.05331125],"genre_scores_gemma":[0.27138722,0.0012141115,0.68927324,0.00039614836,0.00017924426,0.0036511591,0.0038358304,0.00051088585,0.02955226],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.98724085,0.0046759844,0.0014498184,0.0011352174,0.004512462,0.0009858079],"domain_scores_gemma":[0.98435956,0.008846676,0.0011370938,0.0028938884,0.002271581,0.00049126573],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007165967,0.0015763575,0.000557622,0.005729703,0.0016154129,0.0054027936,0.0031119315,0.003154555,0.01093099],"category_scores_gemma":[0.02260936,0.0010625351,0.0020955678,0.0028946016,0.0022397914,0.0069275354,0.0032637925,0.0019958369,0.0024261083],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000107025015,0.00026745198,0.0028970197,0.00033170098,0.000041700372,0.0011541211,0.0020260836,0.040686753,0.0026183813,0.8910915,0.0075435815,0.051234625],"study_design_scores_gemma":[0.000120920886,0.00023185952,0.0015724477,0.00048569028,0.00013728834,0.0014805966,0.000898924,0.38249558,0.0064552827,0.40511116,0.20089394,0.000116271505],"about_ca_topic_score_codex":0.009396277,"about_ca_topic_score_gemma":0.007379666,"teacher_disagreement_score":0.01093099,"about_ca_system_score_codex":0.0020785637,"about_ca_system_score_gemma":0.004093134,"threshold_uncertainty_score":0.037897706},"labels":[],"label_agreement":null},{"id":"W2018414565","doi":"10.1145/2491509.2491515","title":"The value of design rationale information","year":2013,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Norges Forskningsråd","keywords":"Documentation; Computer science; Personalization; Context (archaeology); Value (mathematics); World Wide Web; Programming language","score_opus":0.07125196779048361,"score_gpt":0.29658960558982694,"score_spread":0.22533763779934335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2018414565","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9223481,0.0015379203,0.03804011,0.002433552,0.00006721656,0.00068337447,0.00020419086,0.00044045644,0.034245078],"genre_scores_gemma":[0.9682442,0.00028367015,0.02976554,0.00024249089,0.000027795524,0.00015585298,0.00016439674,0.000053847434,0.0010622034],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9235763,0.04606177,0.00542973,0.0028649922,0.020977318,0.0010897886],"domain_scores_gemma":[0.4092968,0.4888514,0.032584462,0.051220287,0.0149747655,0.0030723219],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.037581634,0.00077129214,0.0005875738,0.0026445729,0.00075759285,0.0065659573,0.0018731083,0.001932398,0.0018744299],"category_scores_gemma":[0.27770883,0.0008141117,0.00062476646,0.00185092,0.0018044723,0.0077359886,0.0027571688,0.0024581475,0.0004810723],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017737579,0.0019317608,0.09612615,0.002201723,0.0003846243,0.0005765003,0.02058841,0.0071123824,0.034550276,0.017607007,0.0019906084,0.8151569],"study_design_scores_gemma":[0.0013483647,0.017070435,0.5074738,0.006483981,0.0026105302,0.0060781417,0.02708307,0.08021771,0.0811794,0.13137959,0.13786675,0.0012081778],"about_ca_topic_score_codex":0.00058074313,"about_ca_topic_score_gemma":0.00078059395,"teacher_disagreement_score":0.037581634,"about_ca_system_score_codex":0.0022782334,"about_ca_system_score_gemma":0.0024384428,"threshold_uncertainty_score":0.19875306},"labels":[],"label_agreement":null},{"id":"W2021538299","doi":"10.1145/2377656.2377657","title":"Systematizing pragmatic software reuse","year":2012,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":88,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; University of Waterloo","funders":"","keywords":"Computer science; Reuse; Variety (cybernetics); Software engineering; Task (project management); Plan (archaeology); Software development; Process (computing); Source code; Software; Metaphor; Human–computer interaction; Systems engineering; Programming language; Artificial intelligence; Engineering","score_opus":0.08412914294878511,"score_gpt":0.3257056353884473,"score_spread":0.24157649243966223,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2021538299","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.048673127,0.0002140773,0.916039,0.0017796209,0.0000596262,0.0003479256,0.000070869886,0.0016899658,0.031125855],"genre_scores_gemma":[0.40849477,0.00026152856,0.5819048,0.00037916395,0.000030915457,0.00048192302,0.00023709638,0.00039579632,0.007813944],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.991389,0.005046333,0.0005200965,0.0011389261,0.0014474333,0.000458278],"domain_scores_gemma":[0.98701966,0.006278205,0.00085805764,0.004549233,0.0010385633,0.00025624002],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0073005245,0.0010854407,0.00041287267,0.0016389132,0.0018834046,0.0038660457,0.0018199492,0.0017105897,0.005763175],"category_scores_gemma":[0.018234458,0.0009876089,0.0014627534,0.00085321354,0.011554603,0.007571033,0.007905814,0.0027621423,0.0008484352],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000048351758,0.00009994249,0.0024374404,0.0003479735,0.000040245704,0.00026768987,0.011081992,0.013429065,0.0075278096,0.88389677,0.0023643079,0.078458324],"study_design_scores_gemma":[0.00011583921,0.0002474734,0.0015770944,0.00021131962,0.00009070793,0.00083305786,0.003397471,0.10894268,0.011024316,0.760729,0.11273045,0.0001006466],"about_ca_topic_score_codex":0.0045564556,"about_ca_topic_score_gemma":0.0052221986,"teacher_disagreement_score":0.0073005245,"about_ca_system_score_codex":0.0026245895,"about_ca_system_score_gemma":0.0042872247,"threshold_uncertainty_score":0.038609326},"labels":[],"label_agreement":null},{"id":"W2034102118","doi":"10.1145/990010.990011","title":"Multi-valued symbolic model-checking","year":2003,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Formal Methods in Verification","field":"Computer Science","cited_by":188,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto","funders":"","keywords":"Model checking; Computer science; CTL*; Computation tree logic; Theoretical computer science; Generalization; Kripke structure; Symbolic trajectory evaluation; Abstraction model checking; Class (philosophy); Temporal logic; Extension (predicate logic); Algorithm; Programming language; Artificial intelligence; Mathematics","score_opus":0.153068929282084,"score_gpt":0.34476235356119034,"score_spread":0.19169342427910635,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2034102118","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011204792,0.00017100899,0.98203063,0.00039028513,0.000093968825,0.00010829422,0.00018076641,0.0019931134,0.0038271043],"genre_scores_gemma":[0.41593492,0.0003185491,0.5790212,0.0003638101,0.00006672918,0.00033929368,0.00053350115,0.00039523936,0.0030267588],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9928832,0.0022897401,0.00053395936,0.0010932577,0.0026345374,0.0005652841],"domain_scores_gemma":[0.98680335,0.008081825,0.00086336,0.0026616333,0.0013417086,0.00024814118],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005600933,0.0011323073,0.0014799203,0.0015129037,0.0012647357,0.003920918,0.0034765087,0.0015577578,0.0047924533],"category_scores_gemma":[0.016556012,0.0007470801,0.0028413206,0.0013493457,0.0039000968,0.006764889,0.004209601,0.002886376,0.00057005516],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022396397,0.00010968015,0.0015438135,0.00035665487,0.00013913972,0.00045985793,0.00045127817,0.2755729,0.0053887162,0.6705075,0.001877678,0.043368816],"study_design_scores_gemma":[0.0000582361,0.00003594123,0.00007916423,0.00006606608,0.000050943843,0.00009422375,0.00004556232,0.74629456,0.008944369,0.23891218,0.0053893523,0.000029487212],"about_ca_topic_score_codex":0.003913706,"about_ca_topic_score_gemma":0.004841553,"teacher_disagreement_score":0.005600933,"about_ca_system_score_codex":0.0031851092,"about_ca_system_score_gemma":0.0038675552,"threshold_uncertainty_score":0.029620886},"labels":[],"label_agreement":null},{"id":"W2039822794","doi":"10.1145/1767751.1767754","title":"Clone region descriptors","year":2010,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":70,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"clone (Java method); Computer science; Cloning (programming); Source code; Software maintenance; Software; Code refactoring; Code (set theory); Software system; Programming language; Biology; Genetics; Gene","score_opus":0.08537287773891615,"score_gpt":0.3123188283960462,"score_spread":0.22694595065713002,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2039822794","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06564231,0.002512052,0.80405194,0.00059327245,0.00066492934,0.0017319659,0.063005015,0.033005066,0.028793523],"genre_scores_gemma":[0.2774572,0.0019082067,0.5981359,0.00073448924,0.00026968218,0.002337646,0.092723146,0.0052692196,0.02116448],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9975586,0.0002459388,0.00052287686,0.00053157797,0.0009818113,0.0001592272],"domain_scores_gemma":[0.9899761,0.0032101432,0.0013744816,0.002448106,0.002688449,0.00030266235],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018182928,0.0008652869,0.0011191554,0.006439158,0.0008188445,0.0028055962,0.0019412681,0.001610651,0.0097695235],"category_scores_gemma":[0.012638841,0.0005281951,0.0010165089,0.006371964,0.00093455426,0.004146131,0.002059548,0.0012154175,0.005075682],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012214324,0.00021448573,0.0327211,0.0017976689,0.00011991272,0.0012932075,0.0027115464,0.010363257,0.041991208,0.15762453,0.09656881,0.6533728],"study_design_scores_gemma":[0.00020451631,0.00044220372,0.014872849,0.0004110702,0.00015445771,0.0031500582,0.00097061065,0.048803426,0.051117826,0.071919695,0.8076757,0.00027751183],"about_ca_topic_score_codex":0.003828177,"about_ca_topic_score_gemma":0.0031539686,"teacher_disagreement_score":0.0097695235,"about_ca_system_score_codex":0.0013544243,"about_ca_system_score_gemma":0.0017454588,"threshold_uncertainty_score":0.03268236},"labels":[],"label_agreement":null},{"id":"W2052401040","doi":"10.1145/2685613","title":"Conditional Commitments","year":2014,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Multi-Agent Systems and Negotiation","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Engineering and Physical Sciences Research Council; Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Model checking; Computation tree logic; Scalability; Theoretical computer science; Software engineering; Programming language","score_opus":0.08299718159182957,"score_gpt":0.3041286519701256,"score_spread":0.22113147037829606,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2052401040","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.011286165,0.00016002658,0.9514679,0.0012394356,0.00021912272,0.00045304367,0.00077152095,0.0017126941,0.032690093],"genre_scores_gemma":[0.47422218,0.00056676636,0.48028925,0.0012095472,0.00026030012,0.00142325,0.0021493016,0.0009845382,0.03889501],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99324834,0.0022413302,0.0006140723,0.0013344836,0.0018507885,0.0007110651],"domain_scores_gemma":[0.9887605,0.0051468327,0.0010910006,0.0028823577,0.0016867624,0.0004326882],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050284136,0.0013358976,0.00064317696,0.0013142426,0.0023130036,0.004590657,0.0032057196,0.0020016837,0.019073306],"category_scores_gemma":[0.016299764,0.00076213776,0.0020686383,0.0011767447,0.0040514027,0.009237354,0.0045468975,0.0034657368,0.0028849782],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00009577165,0.000042064705,0.00065866375,0.00015524532,0.00003466572,0.00027865544,0.0006768868,0.011153029,0.0018221985,0.95908684,0.0042635202,0.021732442],"study_design_scores_gemma":[0.000059588037,0.0001030593,0.00033635166,0.00014915911,0.00009674612,0.00040552125,0.00050811423,0.11508401,0.008684806,0.7516748,0.122815266,0.00008254271],"about_ca_topic_score_codex":0.0054092444,"about_ca_topic_score_gemma":0.005247678,"teacher_disagreement_score":0.019073306,"about_ca_system_score_codex":0.0025611122,"about_ca_system_score_gemma":0.0041513504,"threshold_uncertainty_score":0.06380659},"labels":[],"label_agreement":null},{"id":"W2055098550","doi":"10.1145/2430536.2430538","title":"Views","year":2013,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Parallel Computing and Optimization Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Division of Computing and Communication Foundations","keywords":"Computer science; Concurrency; Programming language; Compiler; Debugging; Distributed computing","score_opus":0.08723794299725363,"score_gpt":0.3124842067430095,"score_spread":0.22524626374575585,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2055098550","genre_codex":"methods","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0053164735,0.0025025918,0.34934774,0.002791385,0.002429239,0.00087400916,0.045099497,0.25463906,0.33699995],"genre_scores_gemma":[0.10952952,0.0045740963,0.27045447,0.004983263,0.0011903733,0.0012966436,0.1679648,0.08873404,0.35127273],"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","domain_scores_codex":[0.9977404,0.00033758747,0.00019546434,0.0004346967,0.0010126354,0.00027931336],"domain_scores_gemma":[0.9963341,0.0006033793,0.00013541445,0.0015665783,0.0011011345,0.00025945858],"candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0019145136,0.001513434,0.00094582816,0.001504339,0.0011312659,0.0056802193,0.003255399,0.0019715952,0.17928901],"category_scores_gemma":[0.007773486,0.0010680296,0.0015930558,0.0013629106,0.0007296768,0.007213444,0.0053295684,0.0027763837,0.10664125],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006610765,0.00012403373,0.0019764907,0.00078505446,0.00008851315,0.0003817594,0.0005760589,0.0021674333,0.006147926,0.10373718,0.6186079,0.2647466],"study_design_scores_gemma":[0.00004251348,0.000033668217,0.00031512554,0.00009242902,0.00002127559,0.00021029176,0.000102280144,0.002588085,0.003364065,0.020057065,0.9731406,0.0000325512],"about_ca_topic_score_codex":0.004058715,"about_ca_topic_score_gemma":0.0046827802,"teacher_disagreement_score":0.820711,"about_ca_system_score_codex":0.0009873487,"about_ca_system_score_gemma":0.0019614664,"threshold_uncertainty_score":0},"labels":[],"label_agreement":null},{"id":"W2056488032","doi":"10.1145/1391984.1391985","title":"Unit-level test adequacy criteria for visual dataflow languages and a testing methodology","year":2008,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":92,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Dataflow; Context (archaeology); Unit testing; Visual programming language; Empirical research; Software engineering; Programming language; Software","score_opus":0.3232977272659157,"score_gpt":0.41639520446411027,"score_spread":0.09309747719819456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2056488032","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0924143,0.00022208976,0.9037453,0.0001838755,0.000012396602,0.00025516242,0.00010529999,0.00076938054,0.002292308],"genre_scores_gemma":[0.6517582,0.000057913192,0.34695372,0.00008101921,0.000023303352,0.00041347998,0.00024574326,0.000115078954,0.00035150707],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9824097,0.008039705,0.0019729317,0.00097650156,0.0060489066,0.00055231777],"domain_scores_gemma":[0.89618325,0.08041725,0.0067043253,0.004858478,0.010725857,0.0011107944],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010728257,0.0009186876,0.0007564116,0.0066180634,0.0006563935,0.0019363379,0.0015020302,0.0015145217,0.0017493347],"category_scores_gemma":[0.08693962,0.00039971937,0.0010738028,0.0018175408,0.0021377504,0.0024964712,0.0015668418,0.00093394046,0.00022096738],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011819414,0.0009115228,0.070972666,0.0011489677,0.00026975645,0.001300904,0.0018657002,0.22412844,0.07255206,0.14304198,0.0028473097,0.47977868],"study_design_scores_gemma":[0.00011847433,0.00088182854,0.0071166647,0.00019748445,0.000080428545,0.0010196358,0.00040821446,0.90512645,0.038119003,0.04430988,0.002561794,0.000060204307],"about_ca_topic_score_codex":0.0018150571,"about_ca_topic_score_gemma":0.0018517158,"teacher_disagreement_score":0.010728257,"about_ca_system_score_codex":0.0012778394,"about_ca_system_score_gemma":0.0014907888,"threshold_uncertainty_score":0.056737125},"labels":[],"label_agreement":null},{"id":"W2056899516","doi":"10.1145/2491509.2491520","title":"Using a functional size measurement procedure to evaluate the quality of models in MDD environments","year":2013,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Model-Driven Software Engineering Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec","funders":"","keywords":"Computer science; Correctness; Consistency (knowledge bases); Quality (philosophy); Quality assurance; Reliability engineering; Software quality; Software quality assurance; Software; Artificial intelligence; Software development; Algorithm; Programming language","score_opus":0.2607569112903205,"score_gpt":0.3488981002180973,"score_spread":0.08814118892777678,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2056899516","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6160746,0.00016139666,0.37767884,0.00007953884,0.00002629063,0.00038238984,0.00053964194,0.0031820287,0.0018752917],"genre_scores_gemma":[0.7724638,0.00006766395,0.22576053,0.000025311916,0.000007354381,0.0002790277,0.000761767,0.00019009862,0.000444438],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.995387,0.0013128441,0.00035898143,0.0005075042,0.0022777754,0.00015597897],"domain_scores_gemma":[0.9620295,0.023505162,0.004405693,0.0040220916,0.005692084,0.0003454434],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041661453,0.000750043,0.0004299878,0.0034214843,0.00031953547,0.00072564685,0.00080655346,0.0006015231,0.001007441],"category_scores_gemma":[0.035489712,0.00023375126,0.0006201202,0.0012182287,0.0005652566,0.0012174028,0.0007689749,0.00051784224,0.00022362566],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001371103,0.0010233638,0.15594912,0.00082468207,0.00043775298,0.00059183256,0.0023950802,0.06623969,0.2765025,0.004940028,0.0016817106,0.48804316],"study_design_scores_gemma":[0.0001318841,0.004738317,0.18731822,0.000117126205,0.0003638333,0.0012391212,0.0010122677,0.4721684,0.32506463,0.0034805266,0.004129981,0.00023586249],"about_ca_topic_score_codex":0.0020278003,"about_ca_topic_score_gemma":0.002262656,"teacher_disagreement_score":0.0041661453,"about_ca_system_score_codex":0.0005817916,"about_ca_system_score_gemma":0.00062691554,"threshold_uncertainty_score":0.022032917},"labels":[],"label_agreement":null},{"id":"W2068726498","doi":"10.1145/839268.839270","title":"Feature specification and automated conflict detection","year":2003,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Formal Methods in Verification","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Computer science; Feature (linguistics); Model checking; Formal specification; Formal methods; Specification language; Extractor; Field (mathematics); Software engineering; Programming language; Data mining","score_opus":0.0813120162940838,"score_gpt":0.3161018302563561,"score_spread":0.23478981396227228,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2068726498","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.020014632,0.00008185664,0.9735869,0.00009555811,0.000013508478,0.00008522848,0.0002255251,0.0049530338,0.0009438202],"genre_scores_gemma":[0.23589376,0.00013043058,0.7597038,0.00011525386,0.000018833021,0.00029024694,0.0017806143,0.00074367545,0.0013234111],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98782617,0.0038622816,0.0011709292,0.0012804728,0.0052341004,0.0006259946],"domain_scores_gemma":[0.9756245,0.014362265,0.002800654,0.0043883636,0.0026035884,0.00022053652],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0056231446,0.0012457222,0.001009199,0.0035280774,0.0009204544,0.001968911,0.0023948231,0.0013597348,0.0032453972],"category_scores_gemma":[0.029067198,0.0012669421,0.0015820987,0.002751593,0.0015912486,0.0039362204,0.002495034,0.0015914917,0.00085900986],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00078685867,0.00040648217,0.034115594,0.0012412699,0.0003058965,0.0020742344,0.0025501286,0.14582483,0.07926623,0.11207541,0.009626948,0.61172605],"study_design_scores_gemma":[0.00017270393,0.00027588248,0.0038308136,0.00014452747,0.00012757764,0.0014292074,0.00043964622,0.78632516,0.10515119,0.08068908,0.021258436,0.00015575546],"about_ca_topic_score_codex":0.0037956382,"about_ca_topic_score_gemma":0.003432441,"teacher_disagreement_score":0.0056231446,"about_ca_system_score_codex":0.0011108826,"about_ca_system_score_gemma":0.0021267007,"threshold_uncertainty_score":0.029738426},"labels":[],"label_agreement":null},{"id":"W2090908516","doi":"10.1145/2430536.2430539","title":"Facilitating the transition from use case models to analysis models","year":2013,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":114,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University","funders":"Pearl Therapeutics; Fonds National de la Recherche Luxembourg","keywords":"Use Case Diagram; Computer science; Ambiguity; Set (abstract data type); Unified Modeling Language; Sequence diagram; Class diagram; Class (philosophy); Quality (philosophy); Data mining; Natural language processing; Artificial intelligence; Programming language; Software","score_opus":0.15956612434369744,"score_gpt":0.31830746037286617,"score_spread":0.15874133602916873,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2090908516","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01792267,0.0001074642,0.9701198,0.000608583,0.00004599776,0.0016226632,0.00025636828,0.005306606,0.0040098275],"genre_scores_gemma":[0.07200021,0.00017642182,0.9220719,0.0002585861,0.000030271516,0.001715,0.00083598925,0.001229864,0.001681772],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9399031,0.03390531,0.005365497,0.0056413217,0.013567452,0.0016173404],"domain_scores_gemma":[0.8098975,0.12667574,0.006813629,0.04305445,0.011898023,0.0016607273],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.036810204,0.0020029242,0.0013586127,0.0039366577,0.0013706106,0.010157862,0.005424087,0.0030195853,0.006089839],"category_scores_gemma":[0.14628083,0.003400991,0.0025945222,0.0022431528,0.0022597301,0.013599474,0.011685369,0.0070995544,0.004046622],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008878723,0.0022230248,0.009445328,0.0021565855,0.00030909045,0.0035932434,0.046303824,0.05535394,0.062059544,0.20462935,0.013731982,0.5993062],"study_design_scores_gemma":[0.00037055844,0.0008276296,0.0044268463,0.0017873782,0.0003050089,0.0015647135,0.006243342,0.51316196,0.077855214,0.1715712,0.22136275,0.0005233063],"about_ca_topic_score_codex":0.0024021138,"about_ca_topic_score_gemma":0.0025722855,"teacher_disagreement_score":0.036810204,"about_ca_system_score_codex":0.0025842108,"about_ca_system_score_gemma":0.0054863286,"threshold_uncertainty_score":0.1946733},"labels":[],"label_agreement":null},{"id":"W2095443555","doi":"10.1145/363516.363523","title":"Proof linking","year":2000,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Security and Verification in Computing","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Correctness; Bytecode; Programming language; Code (set theory); Implementation; Modular design; Distributed computing; Java; Set (abstract data type)","score_opus":0.0799470620929626,"score_gpt":0.3050163120891167,"score_spread":0.22506924999615413,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2095443555","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0034519283,0.00018046526,0.97692263,0.00051035243,0.00020676675,0.000397682,0.00018613579,0.004308522,0.013835585],"genre_scores_gemma":[0.13110524,0.00061541697,0.84370416,0.00059574243,0.00022435757,0.0006763092,0.00092820363,0.001955933,0.020194624],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9914151,0.002655142,0.0007284482,0.0020320804,0.0025350477,0.0006341253],"domain_scores_gemma":[0.96304786,0.01527025,0.002193484,0.014214074,0.0046516987,0.00062264566],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008000394,0.0014224871,0.0009974112,0.0025466261,0.0025469565,0.0060180486,0.0040198504,0.003222593,0.037457157],"category_scores_gemma":[0.038741503,0.0012812568,0.002609892,0.0017437339,0.0036473537,0.012425341,0.0092294235,0.0037552603,0.0126773575],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016457195,0.0001550916,0.0009607801,0.0005504677,0.000059247424,0.00031645113,0.0007883611,0.0080931615,0.0060776356,0.74927217,0.0106897345,0.2228723],"study_design_scores_gemma":[0.00014316985,0.0001955566,0.00034100472,0.0004929268,0.0001281578,0.00090100436,0.00027602463,0.071161844,0.039792255,0.67061555,0.21583432,0.00011830774],"about_ca_topic_score_codex":0.00086931855,"about_ca_topic_score_gemma":0.0007328133,"teacher_disagreement_score":0.037457157,"about_ca_system_score_codex":0.0017489068,"about_ca_system_score_gemma":0.0037966534,"threshold_uncertainty_score":0.12530673},"labels":[],"label_agreement":null},{"id":"W2113351233","doi":"10.1145/2000791.2000794","title":"Reducing the effort of bug report triage","year":2011,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":281,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Process (computing); Variety (cybernetics); Key (lock); Recommender system; Software engineering; Software; Software development; Triage; Data science; World Wide Web; Artificial intelligence; Computer security","score_opus":0.13501545347678479,"score_gpt":0.32812057111168713,"score_spread":0.19310511763490235,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2113351233","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.31704542,0.0022642368,0.5776415,0.002847219,0.00061548594,0.0019101911,0.0009615791,0.08520993,0.011504425],"genre_scores_gemma":[0.47875762,0.00052836287,0.5046675,0.000497588,0.00025787434,0.00064379716,0.002144687,0.0030150781,0.009487489],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.97731394,0.008962637,0.0019120545,0.0035357364,0.007493264,0.0007823518],"domain_scores_gemma":[0.8559619,0.065880224,0.013762273,0.0408127,0.020237021,0.0033459803],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015081359,0.002487269,0.002079073,0.005286417,0.0014714202,0.0032664011,0.004399469,0.0023507446,0.004200506],"category_scores_gemma":[0.12860788,0.0015993713,0.0013488401,0.0026194432,0.00075777806,0.005054185,0.0040662717,0.0027902394,0.0055732327],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075109396,0.0012075446,0.03174909,0.0004650582,0.00016539785,0.00033806096,0.0025370428,0.015125246,0.023525584,0.001578991,0.020393774,0.902163],"study_design_scores_gemma":[0.0006759297,0.003977825,0.07669143,0.00049313763,0.0007387581,0.0024754978,0.0037484618,0.7444115,0.06900931,0.0067415196,0.09050238,0.00053431396],"about_ca_topic_score_codex":0.009170181,"about_ca_topic_score_gemma":0.010913923,"teacher_disagreement_score":0.015081359,"about_ca_system_score_codex":0.0013726144,"about_ca_system_score_gemma":0.00374247,"threshold_uncertainty_score":0.07975882},"labels":[],"label_agreement":null},{"id":"W2116357421","doi":"10.1145/13487689.13487691","title":"Topology analysis of software dependencies","year":2008,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":106,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Intuition; Source code; Task (project management); Static program analysis; Set (abstract data type); Dependency (UML); Software; Program analysis; Fuzzy logic; Programming language; Software engineering; Data mining; Theoretical computer science; Software development; Artificial intelligence; Systems engineering","score_opus":0.09963517125290641,"score_gpt":0.32526599820654656,"score_spread":0.22563082695364015,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2116357421","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.355107,0.00054016546,0.63205165,0.00019798729,0.000027593123,0.0002038154,0.0011590231,0.0016435302,0.009069142],"genre_scores_gemma":[0.8431698,0.00031646524,0.15314306,0.000022361808,0.000028477945,0.00015220813,0.0013061303,0.00018501827,0.0016764305],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9988888,0.00023367487,0.00007411212,0.00017309077,0.0005214809,0.00010899149],"domain_scores_gemma":[0.9924523,0.004070837,0.00097363273,0.0005325613,0.0016987245,0.0002718887],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00080035697,0.000475609,0.0004778739,0.009035432,0.000898425,0.0012290154,0.0005961668,0.0006912605,0.0031873777],"category_scores_gemma":[0.0101045575,0.00043057784,0.00074992725,0.0029745277,0.0006364359,0.0018552567,0.0008162188,0.00056848244,0.00050692505],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006002327,0.00020008784,0.06587483,0.00070384855,0.00019844755,0.0010577643,0.0022054375,0.3435285,0.059789963,0.08554745,0.0057346174,0.43455884],"study_design_scores_gemma":[0.00001784388,0.00013716359,0.027332252,0.000054355078,0.00007540002,0.00066380535,0.000515606,0.89809406,0.013262604,0.052761763,0.0070227482,0.00006246495],"about_ca_topic_score_codex":0.003563928,"about_ca_topic_score_gemma":0.0037618023,"teacher_disagreement_score":0.009035432,"about_ca_system_score_codex":0.0008121468,"about_ca_system_score_gemma":0.00073932175,"threshold_uncertainty_score":0.010662854},"labels":[],"label_agreement":null},{"id":"W2133363731","doi":"10.1145/2000799.2000805","title":"Recommending Adaptive Changes for Framework Evolution","year":2011,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":98,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Code refactoring; Computer science; Software evolution; Software engineering; Open source; Simple (philosophy); Software development; Programming language; Software","score_opus":0.20430693453139503,"score_gpt":0.3413837973744098,"score_spread":0.1370768628430148,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2133363731","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46915826,0.003677941,0.46731436,0.0022713763,0.00042967388,0.0016061414,0.0036905827,0.04315315,0.008698518],"genre_scores_gemma":[0.5414737,0.0006407325,0.44720843,0.0004599423,0.00007802441,0.00042538642,0.00589248,0.0014015039,0.0024198098],"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99493223,0.0011786463,0.00046117313,0.0016550571,0.0015113349,0.00026158028],"domain_scores_gemma":[0.978534,0.009020343,0.0019846573,0.0045066397,0.005311026,0.00064340804],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0048439866,0.0017156376,0.0011220205,0.0053145466,0.0010379109,0.0019834077,0.0025727027,0.0022299476,0.0018632421],"category_scores_gemma":[0.038507614,0.0008873825,0.001154219,0.002558654,0.00057140907,0.003448948,0.0013391535,0.002028638,0.00090220204],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00056734215,0.00068149855,0.12856078,0.00074021204,0.00033865662,0.000601795,0.0017832118,0.023066405,0.01804238,0.0020023908,0.020609446,0.8030059],"study_design_scores_gemma":[0.00029939387,0.00091199134,0.06358908,0.00044461028,0.0008050622,0.0009759191,0.0021515058,0.82200897,0.035806987,0.0065128156,0.06613196,0.00036173215],"about_ca_topic_score_codex":0.013587486,"about_ca_topic_score_gemma":0.0347345,"teacher_disagreement_score":0.013587486,"about_ca_system_score_codex":0.0012622398,"about_ca_system_score_gemma":0.0023500144,"threshold_uncertainty_score":0.027016759},"labels":[],"label_agreement":null},{"id":"W2153546999","doi":"10.1145/1189748.1189751","title":"Representing concerns in source code","year":2007,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":215,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; McGill University","funders":"","keywords":"Computer science; Software engineering; Source code; Task (project management); Software; Separation of concerns; Software development; Software system; Code (set theory); Programming language; Systems engineering","score_opus":0.10964718519581704,"score_gpt":0.3609016914329558,"score_spread":0.25125450623713874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2153546999","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008804573,0.00032031053,0.9702351,0.00047754883,0.00008934473,0.00020829869,0.0014819748,0.010506837,0.007875996],"genre_scores_gemma":[0.11992464,0.00077325944,0.86055046,0.00022102708,0.000068039044,0.00036950386,0.0058035823,0.004059004,0.008230489],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.996808,0.0010572473,0.00034718262,0.00047552818,0.001110387,0.00020168551],"domain_scores_gemma":[0.9896856,0.004874286,0.00096520985,0.002587159,0.0017113094,0.00017642848],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0034129526,0.001178576,0.0004223899,0.0036961983,0.0011753131,0.005305973,0.002091182,0.0024522287,0.0067003174],"category_scores_gemma":[0.016814869,0.001208891,0.00142574,0.0030115915,0.001545534,0.0055382815,0.0033715987,0.0020657168,0.002726634],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002943833,0.0002505311,0.007780003,0.0013211223,0.00013880509,0.002141246,0.011407663,0.073358044,0.019070683,0.5377777,0.02912435,0.3173354],"study_design_scores_gemma":[0.00009328977,0.00011811751,0.0018669942,0.0005797774,0.00015650773,0.0011956012,0.0009828404,0.21352082,0.024541702,0.3052825,0.45151332,0.00014847574],"about_ca_topic_score_codex":0.005438718,"about_ca_topic_score_gemma":0.0044760844,"teacher_disagreement_score":0.0067003174,"about_ca_system_score_codex":0.001470471,"about_ca_system_score_gemma":0.0022767792,"threshold_uncertainty_score":0.022414744},"labels":[],"label_agreement":null},{"id":"W2165688098","doi":"10.1145/941566.941569","title":"Static analysis to support the evolution of exception structure in object-oriented systems","year":2003,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":170,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Exception handling; Java; Control flow; Programming language; Software engineering; Control flow analysis; Object-oriented programming; Robustness (evolution); Static analysis; Static program analysis; Source code; Program code; Information flow; Software; Software development; Programming paradigm; Procedural programming; Inductive programming","score_opus":0.04314179430167696,"score_gpt":0.31228285955602275,"score_spread":0.2691410652543458,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2165688098","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.027318181,0.00015744899,0.95927864,0.00020088335,0.000037750768,0.000083309074,0.00010649866,0.010904479,0.0019129142],"genre_scores_gemma":[0.3413603,0.00031530915,0.6530353,0.00020625655,0.000099042,0.00022336794,0.0006915707,0.001872323,0.002196506],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997719,0.0007013709,0.0002057338,0.00021318998,0.0009819233,0.0001787707],"domain_scores_gemma":[0.9905697,0.005771393,0.0011663233,0.0011946203,0.0011364796,0.00016143029],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003328451,0.00076416455,0.00078890874,0.003572659,0.0010987195,0.001657991,0.0013370706,0.0009290273,0.0017504118],"category_scores_gemma":[0.015866864,0.00089979876,0.0009369204,0.0018570016,0.0014186592,0.0036033103,0.0015075252,0.0016157221,0.0004043793],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00038336116,0.0003612521,0.02077246,0.0006440838,0.00013252538,0.001562361,0.004110036,0.23977277,0.03475048,0.26977688,0.009671343,0.4180625],"study_design_scores_gemma":[0.000050889965,0.00008051828,0.0021770776,0.00014102776,0.00008478268,0.00025314838,0.00016434895,0.86456543,0.020205315,0.09576637,0.016437301,0.000073773874],"about_ca_topic_score_codex":0.0043627447,"about_ca_topic_score_gemma":0.0046367706,"teacher_disagreement_score":0.0043627447,"about_ca_system_score_codex":0.0012610976,"about_ca_system_score_gemma":0.0021773654,"threshold_uncertainty_score":0.017602801},"labels":[],"label_agreement":null},{"id":"W2271850540","doi":"10.1145/2789209","title":"Test Case Prioritization Using Extended Digraphs","year":2015,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Regression testing; Test case; Test suite; Digraph; Model-based testing; Machine learning; Prioritization; Data mining; Test (biology); Hidden Markov model; Fault detection and isolation; Artificial intelligence; Reliability engineering; Software; Regression analysis; Software development","score_opus":0.15832191192600087,"score_gpt":0.3445707515609084,"score_spread":0.18624883963490751,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2271850540","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.024109783,0.00019473242,0.9716684,0.0001505265,0.000026514499,0.00016350627,0.00014649019,0.002418541,0.0011213769],"genre_scores_gemma":[0.56642234,0.00024519206,0.4296098,0.00021810373,0.000030268966,0.0002893061,0.00058838516,0.00025007993,0.0023465068],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9981041,0.0006027333,0.0001551574,0.0004425025,0.000504749,0.00019075916],"domain_scores_gemma":[0.9944049,0.003574082,0.00046366637,0.00070250366,0.00060257205,0.00025227596],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001682438,0.0011896515,0.0009545083,0.0024335368,0.00040315322,0.001089879,0.0019617595,0.00074451126,0.0025961848],"category_scores_gemma":[0.008681951,0.0008039783,0.000906834,0.0011627383,0.0006861479,0.0019298356,0.0011544852,0.0012379215,0.00043215023],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026858205,0.00026205316,0.00421592,0.00025260128,0.00012118399,0.00041268955,0.00024088769,0.60398823,0.016096942,0.018997489,0.0021849908,0.35295844],"study_design_scores_gemma":[0.000024362475,0.00006844501,0.00036579158,0.000015638447,0.000021042746,0.000078842495,0.000015540381,0.9839827,0.00350517,0.010808996,0.0011003459,0.00001306053],"about_ca_topic_score_codex":0.011422841,"about_ca_topic_score_gemma":0.01477972,"teacher_disagreement_score":0.011422841,"about_ca_system_score_codex":0.0016072661,"about_ca_system_score_gemma":0.0020955577,"threshold_uncertainty_score":0.022712708},"labels":[],"label_agreement":null},{"id":"W2291690114","doi":"10.1145/2824234","title":"Type-Based Call Graph Construction Algorithms for Scala","year":2015,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Logic, programming, and type systems","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Scala; Computer science; Bytecode; Programming language; Compiler; Call graph; Java bytecode; Graph; Algorithm; Java; Theoretical computer science; Java applet; Java annotation","score_opus":0.12323723367273838,"score_gpt":0.3188342743765967,"score_spread":0.19559704070385836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2291690114","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0047670947,0.00008916911,0.97602785,0.0001381451,0.000038388625,0.00009503941,0.0002764751,0.016993824,0.0015740388],"genre_scores_gemma":[0.05982255,0.00014957105,0.93024766,0.00013896117,0.000041569372,0.00025449833,0.0014727064,0.0051109213,0.0027615868],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.997996,0.0003191729,0.00018201783,0.00042910903,0.00086260843,0.00021106446],"domain_scores_gemma":[0.9933652,0.0027171543,0.0005910013,0.0019325387,0.0012377958,0.00015623891],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013374806,0.0013231768,0.0008501776,0.003420665,0.0017205914,0.0021081415,0.002231494,0.0013166888,0.010669621],"category_scores_gemma":[0.009884927,0.0010806996,0.0027425548,0.0031251442,0.0016693741,0.0028040265,0.002866195,0.0025962577,0.0035089296],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003591282,0.00022110851,0.005530916,0.0005556621,0.00016362444,0.0002966629,0.0006265455,0.102265395,0.03536883,0.11809745,0.028249398,0.7082653],"study_design_scores_gemma":[0.00011879005,0.00007164041,0.0016825114,0.00012077999,0.00011701844,0.0003770599,0.00017847623,0.74217355,0.03744832,0.17865364,0.038932927,0.00012538787],"about_ca_topic_score_codex":0.0074727363,"about_ca_topic_score_gemma":0.010311911,"teacher_disagreement_score":0.010669621,"about_ca_system_score_codex":0.0024709252,"about_ca_system_score_gemma":0.0031050264,"threshold_uncertainty_score":0.035693407},"labels":[],"label_agreement":null},{"id":"W2395760792","doi":"10.1145/2876441","title":"Understanding JavaScript Event-Based Interactions with Clematis","year":2016,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Intel Corporation","keywords":"Computer science; JavaScript; Program comprehension; Event (particle physics); Software engineering; Web application; Visualization; Asynchronous communication; Programming language; Human–computer interaction; Software; Software system; Artificial intelligence; World Wide Web","score_opus":0.1857356358688734,"score_gpt":0.3321283856873676,"score_spread":0.14639274981849418,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2395760792","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4847833,0.00027082395,0.48570284,0.00049012434,0.000047230387,0.0002491143,0.0011714393,0.017321322,0.009963888],"genre_scores_gemma":[0.8314207,0.00015513942,0.16276968,0.00013630114,0.000026496287,0.00010627987,0.0015636642,0.00095897686,0.002862778],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99940825,0.00014250558,0.00003309064,0.00017619558,0.00019129716,0.000048590297],"domain_scores_gemma":[0.99667346,0.0021796736,0.00043018095,0.0003034464,0.00031812012,0.00009506097],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00051570765,0.0006735823,0.0003200843,0.0009090802,0.0004241507,0.0017005209,0.0007442747,0.000936832,0.0022249678],"category_scores_gemma":[0.004944259,0.00035416498,0.0004208995,0.0003881984,0.00051728555,0.0018217535,0.00069891947,0.0009973927,0.00062573183],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016735003,0.0012107008,0.07766018,0.0012643597,0.00017008538,0.004091684,0.011615396,0.13093407,0.33316782,0.020269929,0.011042559,0.4068998],"study_design_scores_gemma":[0.000038760438,0.00019805081,0.04795338,0.00007758341,0.00004641796,0.00067126297,0.0005924123,0.87245613,0.055986352,0.008432076,0.013479848,0.000067835834],"about_ca_topic_score_codex":0.005272829,"about_ca_topic_score_gemma":0.006575718,"teacher_disagreement_score":0.005272829,"about_ca_system_score_codex":0.00063431315,"about_ca_system_score_gemma":0.0005535263,"threshold_uncertainty_score":0.010484278},"labels":[],"label_agreement":null},{"id":"W2409613233","doi":"10.1145/2904904","title":"Multi-Step Learning and Adaptive Search for Learning Complex Model Transformations from Examples","year":2016,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Model-Driven Software Engineering Techniques","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Transformation (genetics); Model transformation; Context (archaeology); Process (computing); Consistency (knowledge bases); Artificial intelligence; Theoretical computer science; Machine learning; Programming language","score_opus":0.17202754829192649,"score_gpt":0.3368951754105206,"score_spread":0.16486762711859412,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2409613233","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06155665,0.0003263594,0.9341817,0.00025194965,0.000017174027,0.00016745718,0.00008112819,0.0020872378,0.0013303114],"genre_scores_gemma":[0.45183063,0.00015405529,0.5453499,0.0002132046,0.000023698256,0.00032785442,0.0005956207,0.00019238125,0.0013126194],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984053,0.00068943907,0.00012374384,0.00039556038,0.00028676717,0.00009918653],"domain_scores_gemma":[0.9928566,0.0057505188,0.00022224183,0.00069293415,0.00037265525,0.00010496399],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0023027698,0.0012113486,0.0010684829,0.0012615012,0.0004183344,0.00083065155,0.0025452138,0.0016552504,0.002659611],"category_scores_gemma":[0.011236956,0.0006706371,0.00111495,0.0012473791,0.0008957829,0.0027393054,0.0021666382,0.002076371,0.0005262478],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00036135368,0.00056116347,0.0035823437,0.0003184497,0.00012777674,0.00018775609,0.00025922895,0.5161954,0.0048825014,0.009964633,0.0023921214,0.46116728],"study_design_scores_gemma":[0.000020964051,0.0000570155,0.00013233864,0.0000069991383,0.000011696861,0.000022079108,0.00002053157,0.99384576,0.001003205,0.0045930175,0.00028249508,0.000003858176],"about_ca_topic_score_codex":0.0023904035,"about_ca_topic_score_gemma":0.005683546,"teacher_disagreement_score":0.002659611,"about_ca_system_score_codex":0.00080334116,"about_ca_system_score_gemma":0.0012625061,"threshold_uncertainty_score":0.012178361},"labels":[],"label_agreement":null},{"id":"W2475137645","doi":"10.1145/2932631","title":"Multi-Criteria Code Refactoring Using Search-Based Software Engineering","year":2016,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":137,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Japan Society for the Promotion of Science","keywords":"Code refactoring; Computer science; Consistency (knowledge bases); Software quality; Benchmark (surveying); Software engineering; Search-based software engineering; Code smell; Software; Class (philosophy); Source code; Code (set theory); Software development; Software design; Programming language; Artificial intelligence","score_opus":0.17718848535030324,"score_gpt":0.36867093108150445,"score_spread":0.1914824457312012,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2475137645","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09506798,0.0011389497,0.89731044,0.00032815168,0.00005044449,0.00063978916,0.00013948468,0.0024636977,0.0028609827],"genre_scores_gemma":[0.3452206,0.00023940137,0.65166944,0.0001928225,0.000029421792,0.00040380534,0.0004458659,0.00024745724,0.0015511501],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99552625,0.0015426842,0.0003428651,0.00075693603,0.0015991774,0.00023208608],"domain_scores_gemma":[0.9913339,0.0053983815,0.0009318027,0.00057962426,0.0015298134,0.00022644295],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0046374486,0.002499873,0.0021837344,0.005374388,0.0008700876,0.001124849,0.0026742623,0.0017328786,0.0017606147],"category_scores_gemma":[0.013034788,0.00088885834,0.0017300138,0.0028081175,0.0009062238,0.0012987575,0.0016350774,0.0010096933,0.00048453722],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025128791,0.00069857173,0.0045741033,0.0005219888,0.00028253975,0.0002414539,0.00040974142,0.6370179,0.009899888,0.0037841168,0.0018699066,0.34044853],"study_design_scores_gemma":[0.000060697173,0.00019132493,0.00063988566,0.00003318688,0.00005430854,0.000059035756,0.00005339431,0.99476516,0.0018127164,0.0017244141,0.000587496,0.000018389479],"about_ca_topic_score_codex":0.010444199,"about_ca_topic_score_gemma":0.012172607,"teacher_disagreement_score":0.010444199,"about_ca_system_score_codex":0.0016255851,"about_ca_system_score_gemma":0.002543405,"threshold_uncertainty_score":0.024525464},"labels":[],"label_agreement":null},{"id":"W2570857834","doi":"10.1145/2990497","title":"Generating API Call Rules from Version History and Stack Overflow Posts","year":2017,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Android (operating system); Application programming interface; Cluster analysis; World Wide Web; Precision and recall; Set (abstract data type); Baseline (sea); Information retrieval; Operating system; Programming language; Machine learning","score_opus":0.08205151378663754,"score_gpt":0.30898722927561467,"score_spread":0.22693571548897712,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2570857834","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7576191,0.001642416,0.15724087,0.0008394619,0.00034930577,0.0014563078,0.039772183,0.0292039,0.011876453],"genre_scores_gemma":[0.7136769,0.00043713918,0.2050214,0.00030226167,0.00021290102,0.00077053724,0.07370201,0.00082483084,0.005052022],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9949752,0.0006962207,0.0005335193,0.001818013,0.0016350652,0.00034204405],"domain_scores_gemma":[0.9783272,0.011727023,0.0024274092,0.0025618605,0.004424856,0.0005315783],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037035209,0.0018536547,0.0010368619,0.010216343,0.0007798289,0.0019530877,0.0017545385,0.001707218,0.0012178429],"category_scores_gemma":[0.022524655,0.0006275139,0.0017708762,0.0038448845,0.000442361,0.0025217144,0.0011641434,0.0015111659,0.0021558683],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00080456736,0.0013354184,0.37255377,0.0009625536,0.0006627453,0.0015676107,0.0014962233,0.027404282,0.012120432,0.002515914,0.0379222,0.5406543],"study_design_scores_gemma":[0.00013625043,0.000442441,0.16075072,0.00028575622,0.0005766766,0.0011099975,0.00073235534,0.77907217,0.02622111,0.0062276954,0.024246844,0.00019804499],"about_ca_topic_score_codex":0.014528703,"about_ca_topic_score_gemma":0.027987193,"teacher_disagreement_score":0.014528703,"about_ca_system_score_codex":0.0008861358,"about_ca_system_score_gemma":0.0021420554,"threshold_uncertainty_score":0.028888226},"labels":[],"label_agreement":null},{"id":"W2805382256","doi":"10.1145/3196831","title":"An Empirical Study of Meta- and Hyper-Heuristic Search for Multi-Objective Release Planning","year":2018,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":50,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Engineering and Physical Sciences Research Council; Dalian University of Technology; Core Research for Evolutional Science and Technology; Universiteit Utrecht; University College London","keywords":"Computer science; Heuristics; Heuristic; Meta heuristic; Variety (cybernetics); Genetic algorithm; Machine learning; Empirical research; Search algorithm; Beam search; Artificial intelligence; Mathematical optimization; Data mining; Algorithm; Mathematics; Statistics","score_opus":0.2711195274073803,"score_gpt":0.4358261630705214,"score_spread":0.1647066356631411,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2805382256","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95105547,0.0046621608,0.035870552,0.00064538914,0.000050814986,0.0002758244,0.00074297236,0.00023262925,0.006464182],"genre_scores_gemma":[0.9621511,0.0006909062,0.035640642,0.00007679192,0.000018985413,0.00018311916,0.0007844206,0.000046278186,0.00040778145],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99240375,0.0056334836,0.00036740914,0.0005031058,0.000912673,0.00017956161],"domain_scores_gemma":[0.86168456,0.12571962,0.0038731783,0.004842163,0.0032499,0.0006306001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012543626,0.00070341886,0.00057407084,0.001985681,0.0005259116,0.0012080122,0.0015280828,0.001098335,0.0013099618],"category_scores_gemma":[0.062257312,0.00040056085,0.0006340345,0.0034265595,0.00095560215,0.002743319,0.0006684618,0.0014636012,0.00018438257],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014895735,0.002280018,0.06607106,0.0014259561,0.0007526112,0.00018761338,0.0005051528,0.7375212,0.0014580645,0.008577608,0.0036579925,0.1760732],"study_design_scores_gemma":[0.00030313787,0.001971972,0.03452019,0.00022132172,0.00016537451,0.00028815048,0.00061661453,0.9518177,0.0020074588,0.004024442,0.0040091956,0.000054548298],"about_ca_topic_score_codex":0.0031093748,"about_ca_topic_score_gemma":0.004559703,"teacher_disagreement_score":0.012543626,"about_ca_system_score_codex":0.0013842066,"about_ca_system_score_gemma":0.0010689899,"threshold_uncertainty_score":0.06633788},"labels":[],"label_agreement":null},{"id":"W2807405309","doi":"10.1145/3196883","title":"Inferring Extended Probabilistic Finite-State Automaton Models from Software Executions","year":2018,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Probabilistic automaton; Probabilistic logic; Automaton; Finite-state machine; Reinforcement learning; Decidability; Theoretical computer science; Software; Inference; Deterministic automaton; Flexibility (engineering); Büchi automaton; Deterministic finite automaton; Artificial intelligence; Programming language; Machine learning; Mathematics","score_opus":0.08968027278245541,"score_gpt":0.31906479794928544,"score_spread":0.22938452516683003,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W2807405309","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031240245,0.00009004868,0.9652371,0.00013374393,0.000018364228,0.00009664472,0.0004263254,0.002219214,0.0005384465],"genre_scores_gemma":[0.5752295,0.0002798764,0.4205123,0.00010051342,0.000028984521,0.000353472,0.0021513149,0.00031771685,0.0010262796],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99762124,0.0008826413,0.00017098116,0.00053101976,0.00066611695,0.00012798498],"domain_scores_gemma":[0.9843507,0.012313972,0.00088488485,0.0016153094,0.00069169747,0.00014338916],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002069041,0.0010192159,0.00060565706,0.0012984923,0.00038871018,0.0009670246,0.0020755434,0.0011448261,0.0015099411],"category_scores_gemma":[0.018179648,0.00076337124,0.001767405,0.0008255128,0.0010290806,0.0023919358,0.0013765214,0.0021466508,0.00043383843],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016723588,0.00017236985,0.008232334,0.00027939273,0.00013281302,0.00044991725,0.0004095875,0.877772,0.006483487,0.026711084,0.0009765938,0.07821313],"study_design_scores_gemma":[0.000007146339,0.000015471169,0.0002444067,0.0000138749865,0.000014241037,0.000028955126,0.000017170682,0.97711277,0.0019387917,0.02022257,0.00037622327,0.000008353268],"about_ca_topic_score_codex":0.005044392,"about_ca_topic_score_gemma":0.0112063745,"teacher_disagreement_score":0.005044392,"about_ca_system_score_codex":0.000972234,"about_ca_system_score_gemma":0.002041923,"threshold_uncertainty_score":0.01094228},"labels":[],"label_agreement":null},{"id":"W3010215199","doi":"10.1145/3375633","title":"Visualizing Distributed System Executions","year":2020,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":41,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada; U.S. Air Force; Canadian Network for Research and Innovation in Machining Technology, Natural Sciences and Engineering Research Council of Canada; Universities Space Research Association; U.S. Air Force Academy; National Science Foundation","keywords":"Computer science; Timestamp; Process (computing); Distributed computing; Software; Theoretical computer science; Programming language; Real-time computing","score_opus":0.0773188187625623,"score_gpt":0.2982571631356249,"score_spread":0.2209383443730626,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3010215199","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20873518,0.001185257,0.7457102,0.0009438465,0.00015283884,0.0003820379,0.0049435794,0.02082503,0.01712207],"genre_scores_gemma":[0.6974926,0.0009052538,0.2935075,0.00007626793,0.000053800464,0.0003109321,0.0029143607,0.0010033633,0.0037359816],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999348,0.00026822704,0.000042085085,0.000110477915,0.00018483776,0.00004640189],"domain_scores_gemma":[0.99657416,0.0024226694,0.00023707045,0.00031277188,0.00031660634,0.00013668515],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013099506,0.0010548206,0.00036099248,0.0026478842,0.00047064043,0.0020486105,0.0006246357,0.0005670994,0.0066195936],"category_scores_gemma":[0.0059293485,0.00027107165,0.00043755438,0.0014281226,0.00041385883,0.002231319,0.0015909479,0.0009241152,0.00076107593],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0017864304,0.0005248461,0.042995814,0.0020728807,0.000263959,0.0019505847,0.034759738,0.09037145,0.076135665,0.080589294,0.03846635,0.630083],"study_design_scores_gemma":[0.00034468522,0.00084903295,0.048415627,0.0006736777,0.00024297599,0.0017936442,0.013874418,0.54444426,0.051509008,0.12603925,0.21157278,0.00024067934],"about_ca_topic_score_codex":0.0024529381,"about_ca_topic_score_gemma":0.0027815453,"teacher_disagreement_score":0.0066195936,"about_ca_system_score_codex":0.0003920556,"about_ca_system_score_gemma":0.0006667887,"threshold_uncertainty_score":0.022144794},"labels":[],"label_agreement":null},{"id":"W3040857534","doi":"10.1145/3385187","title":"Predicting Node Failures in an Ultra-Large-Scale Cloud Computing Platform","year":2020,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":78,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; York University","funders":"","keywords":"Cloud computing; DevOps; Computer science; Scalability; Context (archaeology); Node (physics); Scale (ratio); Data science; Software; Distributed computing; Computer security; Software engineering; Database; Operating system; Engineering","score_opus":0.059029380876341735,"score_gpt":0.2962129620419633,"score_spread":0.23718358116562155,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3040857534","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9096871,0.00056602247,0.0805252,0.0016501358,0.00022123942,0.0002370085,0.0011008766,0.0032831708,0.0027293419],"genre_scores_gemma":[0.9667594,0.00018021831,0.031177787,0.00009503107,0.00003067177,0.000043994187,0.0009888947,0.000076104356,0.0006478693],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9992766,0.00014282597,0.00005304339,0.00015778154,0.00024832963,0.00012144914],"domain_scores_gemma":[0.9969579,0.0010735727,0.0004335587,0.00030803785,0.0007888217,0.00043817222],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016370051,0.0010499151,0.00042299958,0.0012431738,0.0007862389,0.0008516601,0.0011178763,0.00075897196,0.0006278606],"category_scores_gemma":[0.0052797534,0.00031615104,0.00036502394,0.00073380774,0.0005696442,0.0014622931,0.0012221163,0.0011389203,0.0002523226],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003355827,0.00026510117,0.08701217,0.0001746616,0.000121598096,0.0007746199,0.00048746666,0.81600684,0.011625151,0.0019539723,0.008279926,0.072962955],"study_design_scores_gemma":[0.000005387969,0.000061531624,0.009726007,0.0000104476285,0.000009960492,0.00003370793,0.00019213007,0.9861972,0.0017090417,0.0012165238,0.00082590187,0.000012069702],"about_ca_topic_score_codex":0.021165546,"about_ca_topic_score_gemma":0.029929275,"teacher_disagreement_score":0.021165546,"about_ca_system_score_codex":0.0010954535,"about_ca_system_score_gemma":0.0011666268,"threshold_uncertainty_score":0.042084694},"labels":[],"label_agreement":null},{"id":"W3081249378","doi":"10.1145/3487567","title":"A Tale of Two Cities: Software Developers Working from Home during the COVID-19 Pandemic","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Innovative Approaches in Technology and Social Development","field":"Business, Management and Accounting","cited_by":207,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Pandemic; Work (physics); Thematic analysis; Coronavirus disease 2019 (COVID-19); Productivity; Space (punctuation); Narrative; Computer science; Data science; Public relations; Qualitative research; Sociology; Political science; Economic growth; Medicine; Engineering; Social science","score_opus":0.11270580278257773,"score_gpt":0.30246146702617105,"score_spread":0.1897556642435933,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3081249378","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94476837,0.0008742791,0.0022451174,0.03762777,0.00039053487,0.00009757368,0.00012316601,0.00009223143,0.013780889],"genre_scores_gemma":[0.98521626,0.00072958856,0.0012111173,0.0066447887,0.00009870499,0.00014000661,0.0000732543,0.00011738209,0.005768893],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9889704,0.0075885034,0.0001938111,0.00076701527,0.0009018865,0.0015783802],"domain_scores_gemma":[0.98579943,0.0059997668,0.0013597434,0.0010069518,0.0014922556,0.004341862],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007794626,0.00083521404,0.00092367735,0.0019094715,0.03143661,0.009609203,0.0024277535,0.0051300093,0.0033759803],"category_scores_gemma":[0.017128749,0.0013677861,0.0006587552,0.0021394151,0.020664355,0.012443587,0.01647366,0.009453491,0.00064063346],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000076910525,0.00006101689,0.005599497,0.000074673735,0.000015381429,0.0024691205,0.97769827,0.00006569352,0.00046520843,0.0023717745,0.0064065373,0.0046960018],"study_design_scores_gemma":[0.000009386795,0.00005024056,0.003102238,0.00007491107,0.0000079819365,0.00036294476,0.98042405,0.00006828813,0.00014553055,0.000568732,0.01514923,0.00003647644],"about_ca_topic_score_codex":0.029180553,"about_ca_topic_score_gemma":0.062749475,"teacher_disagreement_score":0.03143661,"about_ca_system_score_codex":0.0068534464,"about_ca_system_score_gemma":0.0061283982,"threshold_uncertainty_score":0.058021426},"labels":[],"label_agreement":null},{"id":"W3094649757","doi":"10.1145/3432690","title":"Are Multi-Language Design Smells Fault-Prone? An Empirical Study","year":2021,"lang":"en","type":"preprint","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Code smell; Computer science; Software design; Software engineering; Systems design; Software design pattern; Programming language; Software development; Software quality; Software","score_opus":0.233044127550916,"score_gpt":0.413159084720781,"score_spread":0.180114957169865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3094649757","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9987821,0.00017912682,0.0004728885,0.000083236424,0.000003274358,0.00002646178,0.00009103045,0.00001883063,0.00034297214],"genre_scores_gemma":[0.99879324,0.000121809964,0.00060199096,0.00004239833,0.0000068928675,0.000029641516,0.00017984293,0.000017789278,0.00020634006],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9918222,0.0023027244,0.001144472,0.0013480316,0.002787541,0.0005950461],"domain_scores_gemma":[0.77948684,0.1199407,0.07249368,0.008060135,0.016122278,0.0038963316],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009739994,0.00052444934,0.0005042021,0.004022757,0.0008464536,0.0017086879,0.0012306712,0.0011103997,0.0017949765],"category_scores_gemma":[0.08003575,0.0005171061,0.0006251721,0.0024804368,0.0017114124,0.003263676,0.0017772177,0.001421885,0.00050005194],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015849505,0.00035640414,0.97663134,0.00019124208,0.00006737569,0.00052348664,0.007788373,0.0002084559,0.0006928384,0.00012025399,0.00042333323,0.012838407],"study_design_scores_gemma":[0.000013546546,0.00039254627,0.98376656,0.00013364888,0.00006682906,0.0008479844,0.009726038,0.0022524598,0.0008196118,0.00018651379,0.0017605912,0.000033665827],"about_ca_topic_score_codex":0.00252023,"about_ca_topic_score_gemma":0.0033670238,"teacher_disagreement_score":0.009739994,"about_ca_system_score_codex":0.001066954,"about_ca_system_score_gemma":0.0009199586,"threshold_uncertainty_score":0.051510632},"labels":[],"label_agreement":null},{"id":"W3117161066","doi":"10.1145/3412378","title":"An Empirical Study of Developer Discussions in the Gitter Platform","year":2020,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":34,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Royal Military College of Canada; Queen's University","funders":"","keywords":"Popularity; Computer science; Thread (computing); World Wide Web; Psychology","score_opus":0.18110658069790245,"score_gpt":0.38313018133306537,"score_spread":0.20202360063516292,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3117161066","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99774206,0.000114833136,0.000918003,0.000070662296,0.0000073011356,0.00010354432,0.00026172926,0.00003606854,0.00074590463],"genre_scores_gemma":[0.9944812,0.0001481357,0.0032201235,0.000069919515,0.000030601732,0.00029034915,0.00085648557,0.00003502408,0.00086819835],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.99085265,0.0047889072,0.00068650383,0.0012361266,0.0019391247,0.0004966398],"domain_scores_gemma":[0.8524474,0.11573511,0.016007341,0.0034235085,0.00967649,0.002710197],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009413514,0.00045900588,0.0004693137,0.004232516,0.0014398029,0.0012598946,0.000718105,0.00078785996,0.00081582874],"category_scores_gemma":[0.060946953,0.00036684563,0.00029502827,0.0032079588,0.0011652923,0.0019044417,0.0012309614,0.0010687796,0.00042347502],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006651548,0.0011576838,0.73002625,0.0011201401,0.00012831677,0.0013828072,0.17201348,0.0008263594,0.009540656,0.0008344975,0.0038168244,0.07848784],"study_design_scores_gemma":[0.000046676156,0.00086724386,0.93208617,0.0002336244,0.00005205721,0.00062660803,0.04609037,0.005313574,0.003527413,0.00039960057,0.010672079,0.00008445884],"about_ca_topic_score_codex":0.0031250163,"about_ca_topic_score_gemma":0.006092117,"teacher_disagreement_score":0.009413514,"about_ca_system_score_codex":0.0012036694,"about_ca_system_score_gemma":0.0008734238,"threshold_uncertainty_score":0.049784005},"labels":[],"label_agreement":null},{"id":"W3131995106","doi":"10.1145/3434279","title":"Are Comments on Stack Overflow Well Organized for Easy Retrieval by Developers?","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; University of Manitoba; Concordia University; Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Classifier (UML); Information retrieval; Obsolescence; Artificial intelligence; Machine learning; Mechanism (biology); Point (geometry); Data mining; Data science","score_opus":0.08626475335032763,"score_gpt":0.32677058681156684,"score_spread":0.24050583346123922,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3131995106","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95414984,0.0010583075,0.027941374,0.0018144443,0.00016429379,0.0005900162,0.0015493885,0.0053802677,0.007352117],"genre_scores_gemma":[0.96641535,0.00037965467,0.028202614,0.00043081358,0.00014568446,0.00016635658,0.0015784225,0.00043835747,0.002242705],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99268544,0.0027330131,0.00082874787,0.0007781417,0.0022956517,0.0006789934],"domain_scores_gemma":[0.90665025,0.05079152,0.01810075,0.005378583,0.016783798,0.0022950885],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007877054,0.00087728014,0.00086481794,0.0039783437,0.0011950207,0.0024300192,0.00073401234,0.0015127467,0.0026725864],"category_scores_gemma":[0.09273085,0.00039593625,0.0005213152,0.0017518863,0.0008792212,0.0056396085,0.0012546009,0.00078525464,0.0022183203],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024403136,0.00057979196,0.38659722,0.002963152,0.0002156288,0.0014553189,0.013406023,0.0031702227,0.06464763,0.0022423856,0.02927184,0.49301055],"study_design_scores_gemma":[0.0005214686,0.0032540804,0.6124345,0.0021709362,0.0009306206,0.004955551,0.027960539,0.16170827,0.09233421,0.0082200905,0.08477211,0.0007376615],"about_ca_topic_score_codex":0.004639848,"about_ca_topic_score_gemma":0.0050940607,"teacher_disagreement_score":0.007877054,"about_ca_system_score_codex":0.0008575472,"about_ca_system_score_gemma":0.0020455886,"threshold_uncertainty_score":0.041658342},"labels":[],"label_agreement":null},{"id":"W3133170989","doi":"10.1145/3428076","title":"Facet-oriented Modelling","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Model-Driven Software Engineering Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Forskningsrådet om Hälsa, Arbetsliv och Välfärd","keywords":"Metamodeling; Computer science; Model transformation; Reuse; Software engineering; Model-driven architecture; Programming language; Modeling language; Software product line; Unified Modeling Language; Software; Software development; Artificial intelligence","score_opus":0.08379540016638565,"score_gpt":0.295710145808132,"score_spread":0.21191474564174634,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3133170989","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0013575414,0.00026497757,0.9795155,0.00033731005,0.000078763034,0.00016844783,0.00031668541,0.0008035645,0.017157314],"genre_scores_gemma":[0.08460553,0.0015723815,0.8853116,0.00035629512,0.00013677753,0.0007595651,0.0018529725,0.0007038371,0.024700964],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9979091,0.00055180205,0.00019104192,0.00036580174,0.00077137764,0.0002108805],"domain_scores_gemma":[0.99794894,0.00070329465,0.00014241235,0.0006857497,0.0004078196,0.00011178925],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0025355888,0.0011403364,0.0008399472,0.0021087755,0.0014909364,0.006023368,0.0029871329,0.0021969604,0.011217714],"category_scores_gemma":[0.0047166697,0.0007304985,0.0034737447,0.0025876383,0.0026433945,0.005363779,0.003539936,0.00196388,0.0032102785],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019948891,0.000020151325,0.00023924188,0.00011899772,0.000017813254,0.00011850077,0.00057110214,0.01195617,0.0010428043,0.9608914,0.0032784285,0.021725396],"study_design_scores_gemma":[0.000028130364,0.000026261923,0.00012541858,0.00014644556,0.00004559008,0.00027395663,0.0002068653,0.12574188,0.001850061,0.6788273,0.1926966,0.000031522755],"about_ca_topic_score_codex":0.008883219,"about_ca_topic_score_gemma":0.009056621,"teacher_disagreement_score":0.011217714,"about_ca_system_score_codex":0.002357936,"about_ca_system_score_gemma":0.0024604474,"threshold_uncertainty_score":0.037526965},"labels":[],"label_agreement":null},{"id":"W3134742627","doi":"10.1145/3431726","title":"Developing Cost-Effective Blockchain-Powered Applications","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":95,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Database transaction; Blockchain; Computer science; Smart contract; Flexibility (engineering); Transaction processing; Function (biology); Computer security; Unit price; Business; Database; Economics; Microeconomics","score_opus":0.051364331351637454,"score_gpt":0.3085804697335149,"score_spread":0.2572161383818774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3134742627","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7930366,0.002131789,0.1702094,0.0011445865,0.00016483356,0.00088042667,0.0004418834,0.0046523074,0.027338136],"genre_scores_gemma":[0.9083633,0.0011407691,0.08305848,0.0001511801,0.000025490264,0.00025508323,0.00077239185,0.00036158593,0.0058718063],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9987803,0.00030674067,0.00006776858,0.00009185009,0.0005173483,0.00023599058],"domain_scores_gemma":[0.9981482,0.0007108058,0.00016305238,0.0004168345,0.00041514763,0.00014597952],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0012698814,0.00046939802,0.0003244569,0.0005430717,0.0003688176,0.001130451,0.0009369905,0.0007156177,0.0020413762],"category_scores_gemma":[0.006523411,0.0003126802,0.00026504928,0.0005640142,0.0004066607,0.0022332335,0.0018234937,0.00069889013,0.0010359737],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011436485,0.0009964007,0.023101762,0.0014809837,0.00011585809,0.0022169633,0.0010266582,0.30849716,0.11061028,0.07566613,0.016539274,0.45860478],"study_design_scores_gemma":[0.00012831176,0.00056544197,0.003972579,0.00013617433,0.00005404247,0.0004940152,0.0004647066,0.87951636,0.046637055,0.025274273,0.042716525,0.00004048752],"about_ca_topic_score_codex":0.00075242476,"about_ca_topic_score_gemma":0.00095086946,"teacher_disagreement_score":0.0020413762,"about_ca_system_score_codex":0.00040371183,"about_ca_system_score_gemma":0.001241901,"threshold_uncertainty_score":0.006829083},"labels":[],"label_agreement":null},{"id":"W3159568147","doi":"10.1145/3439769","title":"Automatic API Usage Scenario Documentation from Technical Q&amp;A Sites","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; University of Saskatchewan; University of Calgary","funders":"","keywords":"Documentation; Computer science; Internal documentation; Software documentation; Code (set theory); Application programming interface; World Wide Web; Coding (social sciences); Software engineering; Information retrieval; Software; Programming language; Software development; Software development process; Set (abstract data type); Software construction","score_opus":0.09554814590246685,"score_gpt":0.34831518452753624,"score_spread":0.2527670386250694,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3159568147","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4118486,0.005461448,0.4010733,0.0031374185,0.0010252381,0.002712585,0.07070371,0.067154065,0.03688372],"genre_scores_gemma":[0.35724676,0.001569885,0.5228461,0.00042196142,0.00035233473,0.001989808,0.10310376,0.002861694,0.009607626],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9928108,0.002342563,0.00072729844,0.0012295871,0.0026306838,0.00025903148],"domain_scores_gemma":[0.9571361,0.01507274,0.0065373634,0.00342732,0.016623188,0.0012033618],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0044480837,0.0012753566,0.0007988072,0.015090626,0.0010801706,0.0026383821,0.0012276828,0.0014711525,0.0036825566],"category_scores_gemma":[0.033807054,0.0008255937,0.000772122,0.006754344,0.00037185717,0.0028198077,0.002278447,0.0013615141,0.0051872195],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044517778,0.00046763723,0.045422927,0.0028943159,0.00018965435,0.002868245,0.005199231,0.0051129553,0.021620719,0.003463793,0.17980647,0.7325089],"study_design_scores_gemma":[0.00027101955,0.00061689684,0.12918353,0.002453806,0.00035698497,0.0038791527,0.009761102,0.2983918,0.055255786,0.014106393,0.48511687,0.00060667767],"about_ca_topic_score_codex":0.002592572,"about_ca_topic_score_gemma":0.0057570906,"teacher_disagreement_score":0.015090626,"about_ca_system_score_codex":0.0010168692,"about_ca_system_score_gemma":0.00218717,"threshold_uncertainty_score":0.023523986},"labels":[],"label_agreement":null},{"id":"W3161148246","doi":"10.1145/3440757","title":"Predicting Performance Anomalies in Software Systems at Run-time","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"IBM (Canada); Thompson Rivers University; Queen's University","funders":"","keywords":"Computer science; Baseline (sea); Precision and recall; Anomaly detection; Software; Recall; Anomaly (physics); Data mining; Software system; Machine learning; Real-time computing; Artificial intelligence; Operating system","score_opus":0.037587398814696533,"score_gpt":0.26188482445949596,"score_spread":0.22429742564479943,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3161148246","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9109449,0.000950237,0.07673986,0.00039777046,0.00007719903,0.00009243692,0.0012650996,0.008147846,0.0013846417],"genre_scores_gemma":[0.98460925,0.0001688048,0.013441233,0.000046760117,0.000021215199,0.000023372646,0.001267481,0.000052629395,0.00036927452],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987141,0.00014361113,0.00011557783,0.00038521868,0.0005048815,0.00013653663],"domain_scores_gemma":[0.9951008,0.0017225185,0.0013058544,0.00044444678,0.0011946515,0.00023170256],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011378467,0.0014647528,0.00060184195,0.0034717396,0.00033490252,0.00078881066,0.000768965,0.00075198145,0.00024980825],"category_scores_gemma":[0.0061522922,0.0003655576,0.00054345466,0.0014547568,0.00035225117,0.0015259426,0.00051605824,0.0008912087,0.00027526938],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004929562,0.0008066225,0.39654544,0.00028194478,0.00029279842,0.0004457512,0.00031548683,0.31296608,0.02296334,0.00067231554,0.004734795,0.25948244],"study_design_scores_gemma":[0.0000059756244,0.00009548898,0.039383903,0.000014458371,0.000032155338,0.00011463082,0.000055521574,0.9516681,0.0071488963,0.0009249167,0.0005344736,0.000021515398],"about_ca_topic_score_codex":0.011578094,"about_ca_topic_score_gemma":0.0141518805,"teacher_disagreement_score":0.011578094,"about_ca_system_score_codex":0.0010958242,"about_ca_system_score_gemma":0.0009925886,"threshold_uncertainty_score":0.0230214},"labels":[],"label_agreement":null},{"id":"W3162893202","doi":"10.1145/3447808","title":"How Should I Improve the UI of My App?","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Digital Marketing and Social Media","field":"Social Sciences","cited_by":52,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Thompson Rivers University","funders":"","keywords":"Computer science; App store; Download; World Wide Web; Mobile apps; Perception; Internet privacy; User interface; Interface (matter); Human–computer interaction; Psychology","score_opus":0.10058065079683723,"score_gpt":0.33634803244965955,"score_spread":0.23576738165282232,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3162893202","genre_codex":"empirical","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.58592653,0.02927105,0.07723099,0.12371387,0.0110815875,0.0014856237,0.0011946876,0.0076751644,0.16242054],"genre_scores_gemma":[0.8469039,0.00901022,0.0761213,0.023495292,0.003157052,0.00041826078,0.0006898802,0.00078065053,0.039423365],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9974496,0.0009945679,0.0001651025,0.00020957472,0.0009914333,0.00018968055],"domain_scores_gemma":[0.98517793,0.00561833,0.0020050025,0.0008581077,0.00509208,0.0012486107],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037591208,0.0006160643,0.0004437962,0.0008054717,0.00085596263,0.0024574134,0.0006996185,0.0012732187,0.004294985],"category_scores_gemma":[0.04862385,0.0002637483,0.00064511196,0.00034419957,0.00090279663,0.0039375685,0.0009526223,0.0017229962,0.0040281406],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004561993,0.00040340927,0.07598789,0.0027969915,0.00019516874,0.0022979947,0.022403931,0.0003968686,0.0217653,0.0029119523,0.23737402,0.63301027],"study_design_scores_gemma":[0.0001490648,0.002831556,0.20768286,0.0036052568,0.00092291,0.01671804,0.027173832,0.0063134898,0.013733246,0.00662106,0.71366876,0.0005798793],"about_ca_topic_score_codex":0.0013367865,"about_ca_topic_score_gemma":0.0028742712,"teacher_disagreement_score":0.004294985,"about_ca_system_score_codex":0.0006596363,"about_ca_system_score_gemma":0.00092970236,"threshold_uncertainty_score":0.019880354},"labels":[],"label_agreement":null},{"id":"W3184231327","doi":"10.1145/3447876","title":"An Empirical Study of the Impact of Data Splitting Decisions on the Performance of AIOps Solutions","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Data Stream Mining Techniques","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; École de Technologie Supérieure; Polytechnique Montréal; Queen's University","funders":"","keywords":"Computer science; Concept drift; Leakage (economics); Cloud computing; Data mining; Context (archaeology); Data science; TRACE (psycholinguistics); Data stream mining","score_opus":0.25840458878557626,"score_gpt":0.41697853064057805,"score_spread":0.1585739418550018,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3184231327","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9877616,0.00091350346,0.007275653,0.00065559795,0.00010091528,0.0002189416,0.0006890076,0.0004693679,0.0019154958],"genre_scores_gemma":[0.9855802,0.000282533,0.011683316,0.00016676316,0.00004167565,0.00013778833,0.0016002143,0.00005782422,0.00044968445],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9910377,0.0033776897,0.0009790008,0.0012999937,0.0023118088,0.0009937854],"domain_scores_gemma":[0.92174727,0.053248964,0.00613886,0.0078459345,0.0076834536,0.0033356387],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011959148,0.0010431892,0.00059590413,0.0012642653,0.00108174,0.0019444682,0.0014635051,0.0013511034,0.0013061974],"category_scores_gemma":[0.07403709,0.00037922745,0.0005870179,0.0018859806,0.0009079865,0.005070276,0.0015435809,0.0031269672,0.00052414276],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0042250804,0.006101967,0.2220129,0.0017202935,0.0007717688,0.00047458458,0.001969507,0.38756773,0.020448966,0.00507567,0.017054588,0.33257696],"study_design_scores_gemma":[0.0003206305,0.0060682804,0.06746844,0.00022401106,0.00026608008,0.000501632,0.0032743912,0.8890836,0.016494447,0.006332919,0.009853081,0.00011237257],"about_ca_topic_score_codex":0.0024388616,"about_ca_topic_score_gemma":0.0021732273,"teacher_disagreement_score":0.011959148,"about_ca_system_score_codex":0.0012275242,"about_ca_system_score_gemma":0.0012606394,"threshold_uncertainty_score":0.06324679},"labels":[],"label_agreement":null},{"id":"W3203911409","doi":"10.1145/3465403","title":"CodeMatcher: Searching Code Based on Sequential Semantics of Important Query Words","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Programming language; Semantics (computer science); Information retrieval; Natural language processing","score_opus":0.08444068339043621,"score_gpt":0.3254469426044706,"score_spread":0.2410062592140344,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3203911409","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1780628,0.0029701104,0.72513103,0.0009161563,0.00015847168,0.0009104634,0.00716982,0.07825727,0.0064239176],"genre_scores_gemma":[0.4117323,0.0009371737,0.5618519,0.00061476947,0.00005592271,0.00055850885,0.015321497,0.0016720268,0.007255972],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987949,0.00011452373,0.000102757826,0.00045978941,0.00044123895,0.00008674254],"domain_scores_gemma":[0.9985311,0.0005212316,0.00019407617,0.00031292884,0.00034719572,0.00009348483],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007698426,0.0014424277,0.0011202177,0.004536153,0.00057744334,0.0011234772,0.002385695,0.0010952735,0.0030016876],"category_scores_gemma":[0.0052586216,0.00055770064,0.001066542,0.0028888308,0.00078474596,0.0045948755,0.0020141294,0.0010039691,0.0015127979],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010813017,0.00046946338,0.01876595,0.0014773442,0.00020349283,0.000585572,0.00087769056,0.03997193,0.05615087,0.015828174,0.037824843,0.82676333],"study_design_scores_gemma":[0.00012220212,0.00038428028,0.0039879205,0.000060986124,0.00010309291,0.00061859406,0.0002730038,0.9161671,0.037940763,0.018143337,0.022120127,0.0000785571],"about_ca_topic_score_codex":0.015530403,"about_ca_topic_score_gemma":0.023631757,"teacher_disagreement_score":0.015530403,"about_ca_system_score_codex":0.0010753665,"about_ca_system_score_gemma":0.0024424305,"threshold_uncertainty_score":0.030880034},"labels":[],"label_agreement":null},{"id":"W3203969037","doi":"10.1145/3470006","title":"Automatic Fault Detection for Deep Learning Programs Using Graph Transformations","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":29,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds de recherche du Québec; Institut de Valorisation des Données","keywords":"Computer science; Metamodeling; Graph; Fault detection and isolation; Precision and recall; Construct (python library); Artificial intelligence; Deep learning; Artificial neural network; Machine learning; Software; Process (computing); Data mining; Software engineering; Theoretical computer science; Programming language","score_opus":0.08467335479524384,"score_gpt":0.3304131342028107,"score_spread":0.24573977940756686,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3203969037","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09719545,0.00032596182,0.8797896,0.00029794758,0.000028860048,0.00012389805,0.00066146854,0.020601787,0.00097506616],"genre_scores_gemma":[0.66511524,0.00022344186,0.33071285,0.00016353696,0.000013885203,0.00017931582,0.0019639637,0.0006599888,0.0009677524],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991436,0.00020019844,0.000066122906,0.00026301388,0.0002543865,0.00007269775],"domain_scores_gemma":[0.9977944,0.0012144358,0.00036380647,0.00034766574,0.0002463458,0.00003332853],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006114245,0.0013155343,0.00044679042,0.0018897186,0.00027079292,0.0006801109,0.0012528081,0.0007514284,0.0011743071],"category_scores_gemma":[0.004135099,0.0004635462,0.0014656605,0.00063870853,0.0007707631,0.00152025,0.0008397371,0.0011220135,0.00027502596],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00031667747,0.00025141297,0.015196503,0.0005430505,0.00014471496,0.00064741273,0.00024580717,0.5558929,0.038669586,0.012777075,0.003667376,0.37164748],"study_design_scores_gemma":[0.000010156953,0.000034169887,0.0004710676,0.00001705417,0.000017265978,0.000053332336,0.000022666523,0.9765786,0.012749745,0.009288052,0.00075221265,0.0000056319723],"about_ca_topic_score_codex":0.006261319,"about_ca_topic_score_gemma":0.009909898,"teacher_disagreement_score":0.006261319,"about_ca_system_score_codex":0.0014061407,"about_ca_system_score_gemma":0.0010896132,"threshold_uncertainty_score":0.012449741},"labels":[],"label_agreement":null},{"id":"W3209468076","doi":"10.1145/3530785","title":"On Wasted Contributions: Understanding the Dynamics of Contributor-Abandoned Pull Requests–A Mixed-Methods Study of 10 Large Open-Source Projects","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Carleton University; Concordia University","funders":"","keywords":"Computer science; Open source; Dynamics (music); Open source software; Management science; Operations research; Software; Economics; Sociology; Engineering; Programming language","score_opus":0.08502718441017452,"score_gpt":0.3720169809182515,"score_spread":0.286989796508077,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3209468076","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9905545,0.0008787806,0.0063423566,0.0003167501,0.000019223438,0.00023391978,0.00075276126,0.000056990964,0.0008446557],"genre_scores_gemma":[0.9873848,0.00041900962,0.009071165,0.00018074735,0.000044264783,0.0004910801,0.0013640112,0.000081883445,0.00096303027],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.9656782,0.022567462,0.0025044759,0.0037934156,0.004324904,0.0011315872],"domain_scores_gemma":[0.55422914,0.36320633,0.045820754,0.012448997,0.020635838,0.003658929],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.047358554,0.00054650276,0.0007350277,0.006567546,0.0013439935,0.0031876748,0.001685757,0.0013223572,0.0009841243],"category_scores_gemma":[0.17870174,0.0006034715,0.00090782036,0.0045151543,0.0013215622,0.0047545996,0.0025691777,0.001416825,0.0005716469],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00075856294,0.0007369439,0.86121756,0.0016792149,0.0004759785,0.0007572922,0.05649165,0.0013661883,0.0027715305,0.0019794176,0.0027182442,0.06904727],"study_design_scores_gemma":[0.000087056134,0.0010351517,0.8791635,0.0007802125,0.00030811937,0.0011290279,0.059321575,0.038918436,0.002858546,0.0029345548,0.013247912,0.0002158533],"about_ca_topic_score_codex":0.0047682384,"about_ca_topic_score_gemma":0.0084212795,"teacher_disagreement_score":0.9526414,"about_ca_system_score_codex":0.0016120571,"about_ca_system_score_gemma":0.0019515789,"threshold_uncertainty_score":0.25045896},"labels":[],"label_agreement":null},{"id":"W3211925781","doi":"10.1145/3491211","title":"An Empirical Study of the Effectiveness of an Ensemble of Stand-alone Sentiment Detection Tools for Software Engineering Datasets","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; University of Saskatchewan; Concordia University; University of Calgary","funders":"","keywords":"Computer science; Sentiment analysis; Software; Majority rule; Artificial intelligence; Machine learning; Empirical research; Detector; Data mining","score_opus":0.0814680668414065,"score_gpt":0.3610115127954521,"score_spread":0.2795434459540456,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3211925781","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.943196,0.0045340112,0.021558283,0.0007559565,0.0005173709,0.00090922817,0.022640996,0.0023525315,0.0035357587],"genre_scores_gemma":[0.7963675,0.0010067715,0.08142878,0.00039795498,0.00033118145,0.0008456191,0.117794424,0.00020964882,0.0016180889],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99256,0.0026872454,0.0010593253,0.0016960952,0.0015936858,0.00040361012],"domain_scores_gemma":[0.9788519,0.011313766,0.0013011462,0.002693726,0.0050651133,0.00077436096],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.015920976,0.0019057917,0.0015332126,0.004577098,0.0012011977,0.0017054189,0.0015052274,0.00157105,0.0007482875],"category_scores_gemma":[0.02032055,0.00032974067,0.001771274,0.003305621,0.0007376197,0.0028864173,0.0017134781,0.0016744633,0.0009227493],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0035145415,0.003937093,0.43388045,0.0028286255,0.004152777,0.0005851057,0.0012483855,0.03728474,0.025140656,0.0012219655,0.06610791,0.42009774],"study_design_scores_gemma":[0.0007220311,0.0045704283,0.36932158,0.0005284521,0.0019228252,0.0022737978,0.0029115425,0.5461591,0.036497828,0.002833791,0.031965096,0.00029355142],"about_ca_topic_score_codex":0.0029433907,"about_ca_topic_score_gemma":0.0050652656,"teacher_disagreement_score":0.015920976,"about_ca_system_score_codex":0.0009285034,"about_ca_system_score_gemma":0.00082476344,"threshold_uncertainty_score":0.08419919},"labels":[],"label_agreement":null},{"id":"W3211945175","doi":"10.1145/3488269","title":"Towards a Consistent Interpretation of AIOps Models","year":2021,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Imbalanced Data Classification Techniques","field":"Computer Science","cited_by":28,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; Huawei Technologies (Canada); Queen's University","funders":"","keywords":"Consistency (knowledge bases); Interpretation (philosophy); Computer science; Machine learning; Randomness; Consistency model; Artificial intelligence; Hyperparameter; Data mining; Econometrics; Statistics; Algorithm; Mathematics; Correctness","score_opus":0.09577591198243182,"score_gpt":0.316722677026575,"score_spread":0.22094676504414315,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W3211945175","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0537554,0.00049320754,0.9346203,0.004406387,0.00017777497,0.00027921848,0.0013578575,0.0013921915,0.0035176459],"genre_scores_gemma":[0.50449145,0.0003271323,0.48860943,0.0011418503,0.00017368929,0.0005792277,0.0032770324,0.00066711183,0.00073307427],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.95479614,0.0279416,0.0034010292,0.0051226197,0.007862473,0.0008760442],"domain_scores_gemma":[0.8314191,0.10239896,0.012327401,0.024341777,0.027773777,0.0017390216],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.040732432,0.0019873732,0.0018298537,0.0050916327,0.0014785702,0.010251732,0.00484883,0.002548229,0.001527709],"category_scores_gemma":[0.2084255,0.0015162671,0.00196705,0.0034009796,0.0031072854,0.00972414,0.0050560627,0.0070998836,0.0007668622],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007277566,0.0005022928,0.07407919,0.0012458869,0.0009865839,0.0012555823,0.010844084,0.4097506,0.0055110524,0.26716495,0.024450807,0.20348111],"study_design_scores_gemma":[0.000069430396,0.000088206005,0.0029263017,0.00029505652,0.00009338446,0.00012364495,0.0011117475,0.67215884,0.0020014995,0.3135206,0.007536311,0.00007494169],"about_ca_topic_score_codex":0.0034345686,"about_ca_topic_score_gemma":0.0046688127,"teacher_disagreement_score":0.040732432,"about_ca_system_score_codex":0.0031619454,"about_ca_system_score_gemma":0.0053718244,"threshold_uncertainty_score":0.21541625},"labels":[],"label_agreement":null},{"id":"W4205529760","doi":"10.1145/3502297","title":"Automated, Cost-effective, and Update-driven App Testing","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Centre National de la Recherche Scientifique; European Commission","keywords":"Computer science; Dependability; Oracle; Code coverage; Random testing; Android (operating system); Model-based testing; Test case; Cyclomatic complexity; Automation; Source code; Code (set theory); Test suite; Random oracle; Set (abstract data type); Software engineering; Distributed computing; Programming language; Machine learning; Software; Operating system; Encryption","score_opus":0.07249387697246641,"score_gpt":0.3145700562029212,"score_spread":0.24207617923045482,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4205529760","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15705416,0.0014665993,0.8192918,0.0007029131,0.000061269944,0.0006426671,0.00042658296,0.015524265,0.0048298026],"genre_scores_gemma":[0.76754314,0.00033325196,0.22929019,0.00020029784,0.00003506151,0.0003196809,0.0006312526,0.0005701125,0.0010769832],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99109155,0.0033184458,0.00043463722,0.0008976819,0.0038285137,0.00042908857],"domain_scores_gemma":[0.97000265,0.01731408,0.0023362862,0.0074348277,0.0024849675,0.00042728084],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002919322,0.0017710633,0.0009345435,0.0029032836,0.00053294405,0.0016592053,0.0032552117,0.0015475117,0.0018451982],"category_scores_gemma":[0.025245955,0.0007453322,0.0012695156,0.0011310703,0.0012171279,0.0034729457,0.002508893,0.0013167345,0.0007290658],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006953966,0.0009925532,0.0249818,0.0008876346,0.00028825575,0.0007789396,0.000841223,0.21976617,0.06923053,0.011543869,0.0048825303,0.6651111],"study_design_scores_gemma":[0.00012228572,0.00084165623,0.006297393,0.00013791726,0.0002141048,0.00096718495,0.00025169842,0.91972655,0.048994407,0.01738507,0.004977495,0.00008426043],"about_ca_topic_score_codex":0.0033490767,"about_ca_topic_score_gemma":0.005227069,"teacher_disagreement_score":0.0033490767,"about_ca_system_score_codex":0.0010048145,"about_ca_system_score_gemma":0.0026046678,"threshold_uncertainty_score":0.0154390335},"labels":[],"label_agreement":null},{"id":"W4210455774","doi":"10.1145/3490489","title":"NPC: <u>N</u> euron <u>P</u> ath <u>C</u> overage via Characterizing Decision Logic of Deep Neural Networks","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":57,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; Natural Sciences and Engineering Research Council of Canada; JST-Mirai Program; National Satellite of Excellence in Trustworthy Software Systems, National University of Singapore; National Research Foundation; Bộ Giáo dục và Ðào tạo; Ministry of Education - Singapore; National Research Foundation Singapore; Canadian Institute for Advanced Research","keywords":"Computer science; Artificial intelligence; Path (computing); Machine learning; Artificial neural network; Graph; Deep neural networks; Decision tree; Mirroring; Theoretical computer science","score_opus":0.039775669112155874,"score_gpt":0.2872848317002451,"score_spread":0.24750916258808922,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4210455774","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1438845,0.00035517899,0.8338144,0.001359008,0.000107520005,0.00019565236,0.0009630001,0.0019549709,0.01736581],"genre_scores_gemma":[0.8934185,0.000321288,0.09914385,0.0003842722,0.00005123688,0.00022044599,0.0008509186,0.00026525723,0.005344263],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9988047,0.00022451808,0.000056356515,0.00034051793,0.00038156717,0.00019232971],"domain_scores_gemma":[0.99758554,0.0013562379,0.00023740384,0.00017990047,0.00055225904,0.00008869036],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0008836144,0.0005439384,0.00030671497,0.0010314785,0.00045551668,0.0015150417,0.0006761031,0.0007424562,0.0044388534],"category_scores_gemma":[0.006607119,0.00024203383,0.0006829772,0.00054262805,0.0015467855,0.0019020756,0.0010843591,0.0012460081,0.0005135263],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00049490086,0.00017729598,0.011840859,0.0004433319,0.00010315845,0.0013237328,0.00046947118,0.30113497,0.023277601,0.35572568,0.008406529,0.2966025],"study_design_scores_gemma":[0.000012473661,0.00004477341,0.0009185545,0.0000421397,0.000023716673,0.00013548059,0.000060929822,0.7820353,0.008500022,0.20464961,0.0035591202,0.000017871478],"about_ca_topic_score_codex":0.008501739,"about_ca_topic_score_gemma":0.008574104,"teacher_disagreement_score":0.008501739,"about_ca_system_score_codex":0.0015008107,"about_ca_system_score_gemma":0.0012280148,"threshold_uncertainty_score":0.016904533},"labels":[],"label_agreement":null},{"id":"W4210647952","doi":"10.1145/3506695","title":"An Empirical Study of the Impact of Hyperparameter Tuning and Model Optimization on the Performance Properties of Deep Neural Networks","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Advanced Neural Network Applications","field":"Computer Science","cited_by":198,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; University of Alberta; Concordia University","funders":"","keywords":"Hyperparameter; Computer science; Artificial intelligence; Artificial neural network; Machine learning; Inference; Hyperparameter optimization; Pruning; Deep learning; Support vector machine","score_opus":0.09199875935483277,"score_gpt":0.3210991654779176,"score_spread":0.22910040612308485,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4210647952","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.91190225,0.013961572,0.06469898,0.0013377811,0.00021220198,0.00023551605,0.0016792413,0.0011857285,0.004786675],"genre_scores_gemma":[0.9755578,0.001842328,0.018386237,0.0002692707,0.00010167826,0.00015276064,0.0027539076,0.00027898117,0.0006571066],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99269414,0.0032025415,0.000838019,0.0014494461,0.0013984757,0.00041724998],"domain_scores_gemma":[0.8668156,0.11213253,0.0060527213,0.009233994,0.004864391,0.0009007998],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016062906,0.002106975,0.0010409078,0.0016870197,0.00090365513,0.0017793289,0.0015790684,0.0017945408,0.0010738131],"category_scores_gemma":[0.09762495,0.0007106608,0.0010499236,0.002230996,0.0015562108,0.00526086,0.0015908814,0.0038154852,0.00047056816],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001200786,0.0008832611,0.08498777,0.0010604177,0.00074698345,0.00046624555,0.00040695287,0.7582662,0.0057379548,0.004976009,0.008687315,0.13258016],"study_design_scores_gemma":[0.00014238094,0.0013586124,0.03651041,0.0003567464,0.00030368372,0.00070421095,0.00046206478,0.9354222,0.011426145,0.008763809,0.0044205743,0.00012924393],"about_ca_topic_score_codex":0.0044515184,"about_ca_topic_score_gemma":0.0047540967,"teacher_disagreement_score":0.016062906,"about_ca_system_score_codex":0.0015033545,"about_ca_system_score_gemma":0.001140683,"threshold_uncertainty_score":0.08494973},"labels":[],"label_agreement":null},{"id":"W4210772589","doi":"10.1145/3511887","title":"Towards Robustness of Deep Program Processing Models—Detection, Estimation, and Enhancement","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":58,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"JST-Mirai Program; National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Robustness (evolution); Computer science; Source code; Source lines of code; Computer engineering; Machine learning; Artificial intelligence; Data mining; Software; Programming language","score_opus":0.048802842295263535,"score_gpt":0.3117413535294528,"score_spread":0.26293851123418926,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4210772589","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025788931,0.00035860186,0.9708238,0.0004226537,0.000034404362,0.000044386125,0.0000750671,0.0016985134,0.0007536971],"genre_scores_gemma":[0.8357656,0.0005593926,0.1608645,0.0004845276,0.00011048561,0.00014432581,0.00035047083,0.00042991946,0.0012907354],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9962649,0.0012069707,0.00023350911,0.00082269125,0.0011512332,0.00032074237],"domain_scores_gemma":[0.98503,0.0083828,0.0019093544,0.002945339,0.0014003294,0.0003322997],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00524672,0.0023678795,0.0010602984,0.0016237143,0.0004100419,0.0016187084,0.001978723,0.0016484775,0.0011185803],"category_scores_gemma":[0.03553505,0.0008693297,0.0012511004,0.0006359069,0.0024151145,0.0035298816,0.004789027,0.0044800784,0.0005099883],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029112794,0.00008787291,0.0025916542,0.00014395491,0.000112920075,0.00010061803,0.0001061132,0.87381303,0.014311723,0.011348681,0.0012059687,0.09588628],"study_design_scores_gemma":[0.0000029399866,0.000046005447,0.00016668244,0.000008947631,0.00000908053,0.000021879956,0.0000053008403,0.99084014,0.004503388,0.004197934,0.00019092037,0.000006708653],"about_ca_topic_score_codex":0.0023773024,"about_ca_topic_score_gemma":0.0015317869,"teacher_disagreement_score":0.00524672,"about_ca_system_score_codex":0.0016706468,"about_ca_system_score_gemma":0.0012354914,"threshold_uncertainty_score":0.027747631},"labels":[],"label_agreement":null},{"id":"W4214849268","doi":"10.1145/3494516","title":"Buddy Stacks: Protecting Return Addresses with Efficient Thread-Local Storage and Runtime Re-Randomization","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Security and Verification in Computing","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"Australian Research Council","keywords":"Computer science; Thread (computing); Pointer (user interface); Operating system; Call stack; Embedded system; Parallel computing; Computer hardware; Stack (abstract data type)","score_opus":0.05372739604091537,"score_gpt":0.2789406530306056,"score_spread":0.22521325698969025,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4214849268","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1696485,0.0018101876,0.7938747,0.00046076233,0.0003170714,0.00026713306,0.00031265168,0.025778024,0.007530976],"genre_scores_gemma":[0.83286524,0.00051242404,0.15640537,0.0003087945,0.00009675898,0.0003414128,0.00047313306,0.0019112995,0.0070854453],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9982015,0.0003139759,0.00018776809,0.00025789792,0.0005838607,0.0004549649],"domain_scores_gemma":[0.9955504,0.0005342882,0.0006052191,0.0025974803,0.00049371726,0.00021888292],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013167139,0.0010135951,0.0008636737,0.0011503734,0.0012692738,0.0023004923,0.0025138275,0.000773825,0.0026890785],"category_scores_gemma":[0.003921162,0.0006492544,0.00070360856,0.0011069297,0.0019943614,0.00449039,0.00468188,0.0014876944,0.0013324565],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031580569,0.00035177072,0.010959267,0.00075097964,0.00036107804,0.00082205626,0.0023304448,0.067668304,0.23798905,0.21194956,0.025468666,0.43819076],"study_design_scores_gemma":[0.0003361532,0.00085672404,0.0024642542,0.00016796212,0.00036194146,0.0007326901,0.0005781696,0.43543202,0.38277173,0.09909854,0.07690643,0.00029343553],"about_ca_topic_score_codex":0.0018029972,"about_ca_topic_score_gemma":0.0021118664,"teacher_disagreement_score":0.0026890785,"about_ca_system_score_codex":0.0010969756,"about_ca_system_score_gemma":0.002390756,"threshold_uncertainty_score":0.008995891},"labels":[],"label_agreement":null},{"id":"W4220958110","doi":"10.1145/3487571","title":"Context- and Fairness-Aware In-Process Crowdworker Recommendation","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"National Key Research and Development Program of China; Youth Innovation Promotion Association of the Chinese Academy of Sciences; Youth Innovation Promotion Association; Chinese Academy of Sciences; National Natural Science Foundation of China","keywords":"Computer science; Popularity; Process (computing); Context (archaeology); Matching (statistics); Task (project management); Ranking (information retrieval); Recommender system; Software bug; Software; World Wide Web; Machine learning; Psychology","score_opus":0.0526966237025728,"score_gpt":0.2985543421583204,"score_spread":0.24585771845574758,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4220958110","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18577132,0.0020739797,0.7889159,0.0009937829,0.0003006595,0.001422931,0.00054121757,0.008380751,0.011599507],"genre_scores_gemma":[0.7878539,0.00048922125,0.20474665,0.0002984017,0.00016616484,0.00065045373,0.0005927215,0.000252156,0.004950284],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99362516,0.0019241516,0.00038514548,0.0019023404,0.001618221,0.0005448707],"domain_scores_gemma":[0.98549795,0.0060427953,0.0015173551,0.0030902307,0.0024688037,0.0013827157],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051858323,0.0017895619,0.0017314713,0.0017260268,0.0018139732,0.0024260955,0.0027143864,0.0016584505,0.0033033043],"category_scores_gemma":[0.023216294,0.00065773237,0.00090541027,0.0010740812,0.0006503121,0.0031426284,0.0026729424,0.0017821532,0.0016161704],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00207278,0.0025671697,0.07253531,0.0015417663,0.00042456356,0.0012547849,0.0056140097,0.06328258,0.053223617,0.0094897235,0.013491431,0.7745022],"study_design_scores_gemma":[0.00038814158,0.0013448418,0.033327065,0.0003118432,0.0006422238,0.00092353055,0.002729575,0.866033,0.031932186,0.023233203,0.03874488,0.00038943076],"about_ca_topic_score_codex":0.005428318,"about_ca_topic_score_gemma":0.00850973,"teacher_disagreement_score":0.005428318,"about_ca_system_score_codex":0.00079633866,"about_ca_system_score_gemma":0.002394102,"threshold_uncertainty_score":0.027425587},"labels":[],"label_agreement":null},{"id":"W4221142711","doi":"10.1145/3638243","title":"Building Domain-Specific Machine Learning Workflows: A Conceptual Framework for the State of the Practice","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Scientific Computing and Data Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal; Polytechnique Montréal","funders":"Université de Montréal; McMaster University; McGill University","keywords":"Workflow; Executable; Computer science; Domain (mathematical analysis); Software engineering; Subject-matter expert; Key (lock); Automation; Software; Data science; Domain engineering; Artificial intelligence; Knowledge management; Software development; Expert system; Component-based software engineering; Engineering; Programming language; Database","score_opus":0.20623520377135818,"score_gpt":0.4093535291417279,"score_spread":0.20311832537036972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4221142711","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0019420364,0.00074845797,0.9830357,0.00822595,0.000068815556,0.00016711395,0.000059001566,0.00045607815,0.0052969554],"genre_scores_gemma":[0.04551832,0.001271089,0.95041704,0.0008223749,0.00010789269,0.00044541425,0.00021503895,0.00017145151,0.001031373],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97375613,0.015979597,0.0024507844,0.0031143806,0.0035537167,0.0011454156],"domain_scores_gemma":[0.9560553,0.02235654,0.0024857873,0.012570895,0.004617197,0.0019142157],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04997301,0.0016471023,0.0012735382,0.008333741,0.003913264,0.020599876,0.009773659,0.006941212,0.003272598],"category_scores_gemma":[0.038474444,0.0019414093,0.0029135146,0.0073974165,0.029206902,0.03453114,0.010152794,0.0104620345,0.0019167153],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000010849267,0.000047682996,0.00025699864,0.00020812344,0.000016994163,0.00006709295,0.0016655212,0.006444837,0.00027335793,0.9738462,0.00096882303,0.016193468],"study_design_scores_gemma":[0.000017416689,0.00003977575,0.00012832522,0.0005367681,0.000017888438,0.00012320784,0.0014190376,0.032525547,0.0008768041,0.9119448,0.0523174,0.000053045384],"about_ca_topic_score_codex":0.006136571,"about_ca_topic_score_gemma":0.002982551,"teacher_disagreement_score":0.950027,"about_ca_system_score_codex":0.0062383376,"about_ca_system_score_gemma":0.014520758,"threshold_uncertainty_score":0.26428568},"labels":[],"label_agreement":null},{"id":"W4224214784","doi":"10.1145/3511598","title":"An Empirical Study on Data Distribution-Aware Test Selection for Deep Learning Enhancement","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Advanced Neural Network Applications","field":"Computer Science","cited_by":40,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Retraining; Computer science; Artificial intelligence; Machine learning; Selection (genetic algorithm); Test data; Artificial neural network; Software deployment; Metric (unit); Data mining; Data set; Test set; Process (computing); Model selection; Deep learning; Distribution (mathematics); Set (abstract data type); Empirical research; Statistics; Mathematics; Engineering","score_opus":0.12392267985852783,"score_gpt":0.38876204613595167,"score_spread":0.2648393662774238,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4224214784","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9261658,0.002355196,0.065553136,0.000692505,0.00013383903,0.00031951786,0.0006597347,0.0014509957,0.0026692718],"genre_scores_gemma":[0.9669216,0.00022493498,0.030624159,0.00017505714,0.000037324884,0.00013405913,0.0011741745,0.000101444944,0.0006072318],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98811066,0.006427385,0.0009183824,0.0017302785,0.0024056628,0.00040760706],"domain_scores_gemma":[0.90357834,0.068475224,0.0052293497,0.01185768,0.009297406,0.0015619631],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016330857,0.0014538503,0.00068181346,0.0016221204,0.0006124081,0.0011433722,0.001842155,0.001071738,0.0007652937],"category_scores_gemma":[0.0871673,0.00032684495,0.00049184257,0.0014521899,0.0012088001,0.0032042237,0.0015714325,0.0018607449,0.00035002508],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0031346923,0.0030549997,0.24640164,0.0009582599,0.0004753898,0.00060836633,0.00082042365,0.24508397,0.022778615,0.004151051,0.013110861,0.45942166],"study_design_scores_gemma":[0.0002517436,0.0027956339,0.046067946,0.00013331017,0.00022469151,0.0007425127,0.0005388322,0.90537274,0.034869142,0.0034710036,0.005460335,0.00007214529],"about_ca_topic_score_codex":0.0021917005,"about_ca_topic_score_gemma":0.0030982208,"teacher_disagreement_score":0.016330857,"about_ca_system_score_codex":0.0014012874,"about_ca_system_score_gemma":0.0011024326,"threshold_uncertainty_score":0.08636683},"labels":[],"label_agreement":null},{"id":"W4224287853","doi":"10.1145/3508479","title":"Just-In-Time Defect Prediction on JavaScript Projects: A Replication Study","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"National Natural Science Foundation of China; National Research Foundation Singapore","keywords":"JavaScript; Computer science; Java; Unobtrusive JavaScript; Machine learning; Artificial intelligence; Replication (statistics); Java Programming Language; Natural language processing; Software engineering; Programming language; Rich Internet application","score_opus":0.10556455715455794,"score_gpt":0.3332004721546199,"score_spread":0.22763591500006197,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4224287853","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.983952,0.00093919877,0.009098691,0.0002717773,0.00011020262,0.00028127033,0.0030414695,0.0010666032,0.0012387136],"genre_scores_gemma":[0.97171164,0.00041198585,0.015022888,0.0002163601,0.000098673314,0.00031937062,0.010810049,0.00016401455,0.0012450606],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9923062,0.002552067,0.0006836667,0.0023124795,0.0017678206,0.00037783667],"domain_scores_gemma":[0.9323357,0.026679413,0.007964983,0.015758444,0.015398904,0.0018627098],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.008426437,0.0011905794,0.001016876,0.0036038447,0.0007784706,0.0014538171,0.0018141805,0.0011895417,0.0008220244],"category_scores_gemma":[0.03992362,0.000445402,0.0015137431,0.002513006,0.00078963867,0.003664635,0.0012365668,0.0021052845,0.0009972227],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001134696,0.003721781,0.7818263,0.001358107,0.0008315272,0.00096700585,0.0025191668,0.014382587,0.0065614786,0.0009155834,0.016441718,0.16934006],"study_design_scores_gemma":[0.00033717928,0.0034063014,0.7386821,0.0004982477,0.000870725,0.0018859748,0.0040490525,0.21871385,0.010007953,0.0021932651,0.019085465,0.00026996373],"about_ca_topic_score_codex":0.009919353,"about_ca_topic_score_gemma":0.010368378,"teacher_disagreement_score":0.9915736,"about_ca_system_score_codex":0.0009346705,"about_ca_system_score_gemma":0.0013236507,"threshold_uncertainty_score":0.04456383},"labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"simulation_or_modeling","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high"},{"model":"gpt","categories":[],"domain":null,"study_design":"observational","genre":"empirical","about_ca_system":false,"about_ca_topic":false,"confidence":"high"}],"label_agreement":"split"},{"id":"W4225143092","doi":"10.1145/3533021","title":"A Machine Learning Approach for Automated Filling of Categorical Fields in Data Entry Forms","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Categorical variable; Computer science; Field (mathematics); Machine learning; Data mining; Artificial intelligence; Set (abstract data type); Data field; Software; Mathematics","score_opus":0.3306833255399733,"score_gpt":0.4220707860804389,"score_spread":0.09138746054046559,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4225143092","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009367939,0.00026819407,0.969764,0.0005324262,0.00004230552,0.0002709133,0.00080955774,0.018035894,0.00090877654],"genre_scores_gemma":[0.07354237,0.00010322438,0.9230017,0.00020937284,0.000050832252,0.00027963688,0.0012183196,0.0001652905,0.0014291584],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9945523,0.0019081937,0.0004692055,0.0012210262,0.0016244167,0.00022480178],"domain_scores_gemma":[0.98129773,0.011816653,0.0015648723,0.002591624,0.0023294596,0.00039973794],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049319393,0.0016865549,0.0012507794,0.0042731087,0.0011708468,0.0014742456,0.0032120978,0.0018803512,0.0044172565],"category_scores_gemma":[0.027385328,0.0007033753,0.0013876766,0.003322736,0.0010588937,0.004979424,0.001940931,0.0028037971,0.0030507066],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00026057553,0.0004191875,0.004663484,0.00028788365,0.00004684005,0.00022116318,0.00058200536,0.03144934,0.0048773787,0.004623132,0.017891267,0.9346777],"study_design_scores_gemma":[0.00006709532,0.00018116621,0.0017422028,0.00007590486,0.000022769202,0.0004358819,0.00022777378,0.9602975,0.007303168,0.016060345,0.013496632,0.00008952218],"about_ca_topic_score_codex":0.0070824954,"about_ca_topic_score_gemma":0.012130314,"teacher_disagreement_score":0.0070824954,"about_ca_system_score_codex":0.0014245762,"about_ca_system_score_gemma":0.0031740628,"threshold_uncertainty_score":0.026082933},"labels":[],"label_agreement":null},{"id":"W4225163285","doi":"10.1145/3522587","title":"There’s no Such Thing as a Free Lunch: Lessons Learned from Exploring the Overhead Introduced by the Greenkeeper Dependency Bot in Npm","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); Queen's University","funders":"","keywords":"Computer science; Dependency (UML); Overhead (engineering); Process (computing); Computer security; Pipeline (software); Action (physics); Software; Software engineering; Risk analysis (engineering); Process management","score_opus":0.17598726110486262,"score_gpt":0.33581605162678857,"score_spread":0.15982879052192595,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4225163285","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.911591,0.0015797759,0.06517582,0.008626724,0.000091552356,0.00017295126,0.0003104874,0.0012726876,0.011179024],"genre_scores_gemma":[0.9670911,0.0004262068,0.030011233,0.0006114077,0.000033751006,0.00006917422,0.0002398861,0.00032911816,0.0011881243],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99277675,0.0038164507,0.00022346804,0.0009927703,0.0016003802,0.0005901828],"domain_scores_gemma":[0.9248485,0.058243386,0.0048120446,0.005973252,0.0046509397,0.0014719627],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010034121,0.00070069043,0.0006704211,0.0018619364,0.0016492617,0.0032756173,0.002032736,0.0014213563,0.0014573794],"category_scores_gemma":[0.06302603,0.0006346079,0.00042537195,0.0013360521,0.002999458,0.009369755,0.0023772244,0.002755816,0.0004077997],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011589151,0.0020397657,0.44316465,0.0018383249,0.00024426,0.00444571,0.064522915,0.02900301,0.014792513,0.037343502,0.018041825,0.38340467],"study_design_scores_gemma":[0.00018698088,0.0015201119,0.286957,0.0019015085,0.00029267187,0.0036272937,0.09060454,0.44250762,0.012490805,0.10750028,0.05201939,0.0003918338],"about_ca_topic_score_codex":0.0097435005,"about_ca_topic_score_gemma":0.01735904,"teacher_disagreement_score":0.010034121,"about_ca_system_score_codex":0.0024891603,"about_ca_system_score_gemma":0.0022131347,"threshold_uncertainty_score":0.053066075},"labels":[],"label_agreement":null},{"id":"W4280562623","doi":"10.1145/3529318","title":"Testing Feedforward Neural Networks Training Programs","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Debugging; Hyperparameter; Artificial neural network; Machine learning; Artificial intelligence; Software; Deep neural networks; Training (meteorology); Deep learning; Test data; Software engineering; Programming language","score_opus":0.1199166601035608,"score_gpt":0.30557276794227944,"score_spread":0.18565610783871866,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4280562623","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6682219,0.00036797847,0.3031165,0.0007074854,0.00018762308,0.00019782846,0.0011772593,0.020868171,0.00515529],"genre_scores_gemma":[0.90135956,0.00009857047,0.094984815,0.0002094924,0.000013796848,0.00016954212,0.0012246176,0.00050738786,0.0014322796],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.998387,0.00048323075,0.00013360898,0.00043455279,0.0003903679,0.00017123122],"domain_scores_gemma":[0.9854833,0.010276351,0.00074858614,0.0016238784,0.0016988401,0.0001690707],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002495747,0.0011182879,0.0003660788,0.0005617309,0.00028893407,0.00055189076,0.0018660235,0.0009599824,0.0041976073],"category_scores_gemma":[0.02015126,0.00044473936,0.00054490566,0.00033346756,0.0010180789,0.0015373228,0.0007865357,0.001041575,0.0006368725],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006553139,0.00041776817,0.018162066,0.0005268664,0.00010984268,0.0003860383,0.00025588242,0.7647144,0.020340586,0.0056269094,0.004996429,0.18380794],"study_design_scores_gemma":[0.000028993749,0.0001202529,0.0007253119,0.000031493102,0.0000118668495,0.0000315047,0.000024091418,0.977572,0.018792417,0.0020251751,0.00062881777,0.000008080959],"about_ca_topic_score_codex":0.0052579236,"about_ca_topic_score_gemma":0.0061716703,"teacher_disagreement_score":0.0052579236,"about_ca_system_score_codex":0.0013739655,"about_ca_system_score_gemma":0.0012741789,"threshold_uncertainty_score":0.0140423775},"labels":[],"label_agreement":null},{"id":"W4282033849","doi":"10.1145/3542944","title":"Towards Learning Generalizable Code Embeddings Using Task-agnostic Graph Convolutional Networks","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Concordia University","funders":"","keywords":"Computer science; Source code; Downstream (manufacturing); Graph; Benchmarking; Code (set theory); Abstract syntax; Embedding; Task (project management); Artificial intelligence; Syntax; Theoretical computer science; Programming language","score_opus":0.07716174346765048,"score_gpt":0.31820709951536275,"score_spread":0.24104535604771227,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4282033849","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.177948,0.0017265733,0.7967585,0.0010004874,0.00023111278,0.000176278,0.0019299305,0.015858592,0.0043705516],"genre_scores_gemma":[0.75645584,0.0009612111,0.22072338,0.00090991886,0.00012710557,0.00027758064,0.011774821,0.0009123827,0.00785773],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99942243,0.00012058632,0.0000264867,0.00026322732,0.000088998604,0.00007830877],"domain_scores_gemma":[0.99861455,0.0005073707,0.0001759083,0.00035916825,0.0002671485,0.000075828844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006479251,0.0024446878,0.00069667335,0.0012819347,0.00033135666,0.00075102324,0.0013798375,0.0012474385,0.0012020733],"category_scores_gemma":[0.0036987402,0.0006007485,0.0011991725,0.0011664835,0.00089131866,0.0028712908,0.0014209768,0.0022308773,0.0010412163],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00023243434,0.00034213348,0.0068368977,0.00030226685,0.00018156621,0.00020878052,0.00017882026,0.513875,0.015933335,0.0071939486,0.01601021,0.43870455],"study_design_scores_gemma":[0.000013285043,0.00005679006,0.00053901545,0.000014003113,0.000018921835,0.000038935206,0.000025158895,0.9854665,0.002681197,0.009862641,0.0012731707,0.000010455082],"about_ca_topic_score_codex":0.007650756,"about_ca_topic_score_gemma":0.016184883,"teacher_disagreement_score":0.007650756,"about_ca_system_score_codex":0.0010432334,"about_ca_system_score_gemma":0.0011069787,"threshold_uncertainty_score":0.015212476},"labels":[],"label_agreement":null},{"id":"W4283323653","doi":"10.1145/3544790","title":"Toward More Efficient Statistical Debugging with Abstraction Refinement","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Office of Naval Research; Natural Science Foundation of Jiangsu Province; National Natural Science Foundation of China; National Science Foundation","keywords":"Debugging; Computer science; Algorithmic program debugging; Abstraction; Programming language; Pruning; Discriminative model; Software engineering; Machine learning","score_opus":0.07935506400377887,"score_gpt":0.3095148429020107,"score_spread":0.23015977889823183,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4283323653","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.007685265,0.0001368255,0.9894302,0.00015760402,0.00001411993,0.00006014936,0.000036952133,0.0021636565,0.0003152491],"genre_scores_gemma":[0.20179886,0.00024081589,0.79577863,0.00028281342,0.00004325046,0.00019988991,0.00032588057,0.0007381718,0.00059171],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.98449576,0.0062291147,0.0009504119,0.0015206581,0.0059493096,0.000854819],"domain_scores_gemma":[0.9642211,0.016460415,0.0032757698,0.01123442,0.00444098,0.00036729133],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010319101,0.0015169284,0.0014262837,0.0027609447,0.0009093215,0.0017841392,0.0031483236,0.0010307786,0.0011319132],"category_scores_gemma":[0.04409249,0.0010501927,0.0017673941,0.0018867197,0.0017630114,0.0040756967,0.0037183862,0.0031970886,0.0007225902],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004855957,0.00040742752,0.019159801,0.00060663524,0.00031698158,0.0006652528,0.0012804056,0.26398683,0.07003285,0.09568154,0.006852462,0.54052424],"study_design_scores_gemma":[0.00008183279,0.0002103275,0.0016877621,0.00009864586,0.000115066665,0.00036221617,0.00013220758,0.8957353,0.02301731,0.0720583,0.006444765,0.00005614976],"about_ca_topic_score_codex":0.0026624925,"about_ca_topic_score_gemma":0.004811287,"teacher_disagreement_score":0.010319101,"about_ca_system_score_codex":0.0009712569,"about_ca_system_score_gemma":0.004262997,"threshold_uncertainty_score":0.054573238},"labels":[],"label_agreement":null},{"id":"W4284962442","doi":"10.1145/3546941","title":"Estimating Probabilistic Safe WCET Ranges of Real-Time Systems at Design Stages","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Real-Time Systems Scheduling","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; European Commission; Université du Luxembourg","keywords":"Computer science; Probabilistic logic; Worst-case execution time; Scheduling (production processes); Task (project management); Set (abstract data type); Execution time; Point (geometry); Real-time computing; Distributed computing; Mathematical optimization; Artificial intelligence; Programming language; Systems engineering","score_opus":0.07499825737973446,"score_gpt":0.29155091299379854,"score_spread":0.21655265561406406,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4284962442","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.16761526,0.00042926506,0.8286357,0.00017011806,0.000013928669,0.000089665074,0.00031370472,0.0014758416,0.0012564918],"genre_scores_gemma":[0.8803867,0.00014915175,0.118134364,0.000042850996,0.000013082696,0.00012378175,0.00059072516,0.00019129468,0.00036792504],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99641645,0.001135772,0.00022243072,0.00067562016,0.0013236078,0.00022605398],"domain_scores_gemma":[0.97490466,0.017954748,0.0031809774,0.0021019906,0.001642004,0.00021553959],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0039632884,0.001216654,0.00063768914,0.0026527718,0.0004220874,0.0009967022,0.0012361117,0.00096751854,0.00071810914],"category_scores_gemma":[0.03246803,0.0008248047,0.0009318228,0.0012708615,0.0008666744,0.0018983759,0.00096259214,0.0013287492,0.00027346343],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001104954,0.000041039828,0.008032635,0.0000738207,0.000043902768,0.00015369586,0.00010845272,0.96237135,0.005468798,0.0024135844,0.00024125671,0.02094094],"study_design_scores_gemma":[0.000007333282,0.00003573283,0.0018010573,0.0000087938815,0.000010379884,0.00006738521,0.000021716423,0.9892497,0.004010113,0.0044840993,0.00029163025,0.000012015525],"about_ca_topic_score_codex":0.0037484036,"about_ca_topic_score_gemma":0.0038350767,"teacher_disagreement_score":0.0039632884,"about_ca_system_score_codex":0.0009528951,"about_ca_system_score_gemma":0.0011918134,"threshold_uncertainty_score":0.020960152},"labels":[],"label_agreement":null},{"id":"W4285394558","doi":"10.1145/3546945","title":"Assessing the Alignment between the Information Needs of Developers and the Documentation of Programming Languages: A Case Study on Rust","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Huawei Technologies (Canada)","funders":"","keywords":"Documentation; Computer science; Technical documentation; Application programming interface; World Wide Web; Internal documentation; Set (abstract data type); Baseline (sea); Software engineering; Information retrieval; Programming language; Software development; Software","score_opus":0.0778948728389709,"score_gpt":0.3699735239963255,"score_spread":0.2920786511573546,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285394558","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9933356,0.00020043358,0.00465961,0.00023015244,0.000009150506,0.00010000279,0.00036140668,0.00009770216,0.0010059358],"genre_scores_gemma":[0.9797323,0.00018608687,0.017196352,0.00010595369,0.000024998764,0.00016205621,0.0016497283,0.00008695422,0.00085549604],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.9932881,0.004024745,0.00048730045,0.00094117346,0.0009986331,0.00026003813],"domain_scores_gemma":[0.91566086,0.06723779,0.0062963655,0.003176809,0.0060944343,0.0015337202],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0083435,0.00036065897,0.00060069206,0.003043409,0.0011932311,0.0014867116,0.00072943035,0.0011273103,0.0006952808],"category_scores_gemma":[0.041403756,0.00030461125,0.00046338805,0.0038491474,0.0008770524,0.0024436074,0.0015817342,0.001471685,0.00034133586],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016319362,0.0036469891,0.5815503,0.0019433225,0.0002971357,0.0062808534,0.09014054,0.015105577,0.0215234,0.0032133826,0.009938231,0.26472828],"study_design_scores_gemma":[0.00032473635,0.0030150865,0.7171318,0.00053263205,0.0003224562,0.0044911588,0.07279516,0.13957971,0.024449207,0.005724139,0.031375613,0.0002583308],"about_ca_topic_score_codex":0.0068254224,"about_ca_topic_score_gemma":0.012328102,"teacher_disagreement_score":0.0083435,"about_ca_system_score_codex":0.0013886563,"about_ca_system_score_gemma":0.0009960762,"threshold_uncertainty_score":0.0441252},"labels":[],"label_agreement":null},{"id":"W4285794979","doi":"10.1145/3549542","title":"Is My Transaction Done Yet? An Empirical Study of Transaction Processing Times in the Ethereum Blockchain Platform","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":45,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); Queen's University","funders":"","keywords":"Database transaction; Computer science; Transaction processing system; Blockchain; Transaction processing; Online transaction processing; Distributed transaction; Transaction cost; Computer security; Database; Business; Finance","score_opus":0.07742992799670803,"score_gpt":0.3326076058421708,"score_spread":0.25517767784546275,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4285794979","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9958859,0.00027922206,0.0007680585,0.00036115936,0.00001502294,0.00003566072,0.00054545817,0.0000308398,0.0020786556],"genre_scores_gemma":[0.9971796,0.00016953587,0.0006268166,0.000086221444,0.000022064109,0.00003473182,0.0010088964,0.000026000404,0.00084608496],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9961856,0.0016089136,0.00034381475,0.0004969037,0.00090939953,0.0004553131],"domain_scores_gemma":[0.7843049,0.16694471,0.027845014,0.006481326,0.0094563635,0.004967661],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007852411,0.00040244937,0.0004310197,0.0014619046,0.0006564614,0.0020723576,0.0012310259,0.0011105288,0.0066364906],"category_scores_gemma":[0.09137406,0.00034870382,0.00046686627,0.0021763274,0.0009852715,0.004011074,0.00090041617,0.003963162,0.0024430763],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002192385,0.002243026,0.9264598,0.00030254075,0.00021518902,0.0006132518,0.0028006358,0.0087661,0.001278271,0.0039157225,0.0070625795,0.044150513],"study_design_scores_gemma":[0.0001423018,0.0015219031,0.8905242,0.00018506736,0.0001405612,0.0008097499,0.008173958,0.08330265,0.0018865813,0.0039280243,0.009266447,0.00011859807],"about_ca_topic_score_codex":0.0045601553,"about_ca_topic_score_gemma":0.0033944952,"teacher_disagreement_score":0.007852411,"about_ca_system_score_codex":0.00083715224,"about_ca_system_score_gemma":0.0009044834,"threshold_uncertainty_score":0.041527987},"labels":[],"label_agreement":null},{"id":"W4286485305","doi":"10.1145/3550271","title":"Black-box Safety Analysis and Retraining of DNNs based on Feature Extraction and Clustering","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg; Université du Luxembourg","keywords":"Computer science; Cluster analysis; Retraining; Black box; Deep neural networks; Artificial intelligence; Machine learning; Artificial neural network; Feature (linguistics); Root (linguistics); Root cause; Pattern recognition (psychology); Reliability engineering","score_opus":0.037010912796732284,"score_gpt":0.30333438913844046,"score_spread":0.26632347634170817,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4286485305","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06660454,0.0002599859,0.92854273,0.00015187626,0.00005835583,0.00008912724,0.00007449005,0.0026999256,0.0015189748],"genre_scores_gemma":[0.7867711,0.00019328215,0.20929885,0.00019753595,0.000024126457,0.000113818736,0.0002897043,0.00029079936,0.0028208836],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99947673,0.000073646544,0.0000344773,0.00016927828,0.00017127891,0.0000745608],"domain_scores_gemma":[0.99864024,0.0005035862,0.00021137278,0.00022952243,0.00036978113,0.000045493773],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013532748,0.001648578,0.00062156457,0.0009567377,0.00041615064,0.00054150017,0.0016225685,0.000904542,0.0015091979],"category_scores_gemma":[0.004245228,0.00050747156,0.00068347243,0.00032304114,0.00088859326,0.0013136688,0.0011411133,0.0015391213,0.00044732122],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019650765,0.00009052072,0.0023794896,0.00009470807,0.00007516816,0.0002667471,0.00014254836,0.74816304,0.025645562,0.0037386322,0.0013672699,0.21783985],"study_design_scores_gemma":[0.0000026163805,0.000032974076,0.00034778827,0.000007517243,0.000008073981,0.000027462576,0.00000872413,0.9868024,0.010728138,0.0017321537,0.0002964691,0.0000058285746],"about_ca_topic_score_codex":0.0056365184,"about_ca_topic_score_gemma":0.005879547,"teacher_disagreement_score":0.0056365184,"about_ca_system_score_codex":0.0013885642,"about_ca_system_score_gemma":0.0010806065,"threshold_uncertainty_score":0.011207402},"labels":[],"label_agreement":null},{"id":"W4286488146","doi":"10.1145/3550270","title":"Parameter Coverage for Testing of Autonomous Driving Systems under Uncertainty","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Reliability and Analysis Research","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":false,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"","funders":"Japan Society for the Promotion of Science; JST-Mirai Program; European Regional Development Fund; Japan Science and Technology Agency; University of Waterloo; Exploratory Research for Advanced Technology; Science Foundation Ireland","keywords":"Computer science; Correctness; Heuristics; Trustworthiness; Process (computing); Cover (algebra); Operations research; Reliability engineering; Mathematical optimization; Risk analysis (engineering); Algorithm; Computer security","score_opus":0.10734772623091261,"score_gpt":0.3267423154097948,"score_spread":0.21939458917888222,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4286488146","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4356611,0.0011310616,0.5555952,0.00049956987,0.00004053768,0.00020932875,0.00076799496,0.0020724745,0.004022628],"genre_scores_gemma":[0.9582231,0.00012394095,0.040716864,0.000046888737,0.000015628839,0.00009059693,0.00045884424,0.00014548516,0.00017860574],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9916127,0.0037362212,0.0006027718,0.0008537215,0.002601002,0.0005937177],"domain_scores_gemma":[0.9401591,0.05098559,0.0027228992,0.003584355,0.0018581783,0.0006898283],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005188096,0.0010320437,0.0011743865,0.0026489925,0.00054001133,0.0017175464,0.0015393009,0.0011477612,0.0015833579],"category_scores_gemma":[0.056494456,0.00044810175,0.0013408279,0.0011003793,0.0017112024,0.0024714707,0.0016642562,0.0010090441,0.00018065977],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00035941243,0.00007051609,0.010838004,0.00018283381,0.000115176335,0.00020079469,0.00017860327,0.94400936,0.005049877,0.0077353106,0.00031473546,0.030945411],"study_design_scores_gemma":[0.000016573227,0.0001739305,0.0022367768,0.000045257144,0.00003073967,0.0001256316,0.000066312,0.97595507,0.0057468032,0.015060229,0.00051572506,0.000026837197],"about_ca_topic_score_codex":0.0028071469,"about_ca_topic_score_gemma":0.0017563684,"teacher_disagreement_score":0.005188096,"about_ca_system_score_codex":0.0012859679,"about_ca_system_score_gemma":0.0012169842,"threshold_uncertainty_score":0.027437568},"labels":[],"label_agreement":null},{"id":"W4287378204","doi":"10.1145/3533700","title":"The Co-evolution of the WordPress Platform and Its Plugins","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal; Queen's University","funders":"","keywords":"Plug-in; Computer science; Software; World Wide Web; Software engineering; Operating system","score_opus":0.05189968053057521,"score_gpt":0.2961285702044682,"score_spread":0.244228889673893,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4287378204","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.88934046,0.0012178102,0.08132104,0.0014313556,0.00020009736,0.00019818435,0.00032033885,0.0064913677,0.019479325],"genre_scores_gemma":[0.93689036,0.0009062315,0.043223366,0.00042663864,0.00009527737,0.00017792382,0.0014966216,0.0028591652,0.013924467],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.99426645,0.0011878171,0.00043304398,0.0010573886,0.0023437522,0.00071150146],"domain_scores_gemma":[0.9808462,0.0068667396,0.0029838213,0.0051686508,0.0030943265,0.0010401859],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005134798,0.00083784445,0.00039144035,0.002283014,0.001088302,0.0037226083,0.0013655943,0.0011936886,0.0013299038],"category_scores_gemma":[0.026837558,0.0010263431,0.00072487857,0.0022243897,0.0016244465,0.010135051,0.0045939074,0.0026784649,0.0011968822],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007444685,0.0010265434,0.31785783,0.000952301,0.00044789852,0.0129178865,0.01596214,0.026157504,0.06266431,0.045336638,0.018081512,0.49785098],"study_design_scores_gemma":[0.00009267796,0.0009971297,0.40942267,0.0005694155,0.00056453404,0.014357537,0.006636496,0.16185983,0.075674355,0.023519013,0.3058849,0.00042143845],"about_ca_topic_score_codex":0.0027847202,"about_ca_topic_score_gemma":0.0023501844,"teacher_disagreement_score":0.005134798,"about_ca_system_score_codex":0.00091379404,"about_ca_system_score_gemma":0.0015449955,"threshold_uncertainty_score":0.027155697},"labels":[],"label_agreement":null},{"id":"W4288363128","doi":"10.1145/3306607","title":"Status Quo in Requirements Engineering","year":2019,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Techniques and Practices","field":"Computer Science","cited_by":112,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"Fundação de Amparo à Pesquisa do Estado do Rio Grande do Sul; Conselho Nacional de Desenvolvimento Científico e Tecnológico; Bundesministerium für Wissenschaft, Forschung und Wirtschaft; Eesti Teadusagentuur; Österreichische Nationalstiftung für Forschung, Technologie und Entwicklung","keywords":"Computer science; Requirements engineering; Context (archaeology); Non-functional requirement; Process (computing); Management science; Requirements elicitation; Empirical research; Status quo; Software engineering; Software development; Software; Engineering; Software construction","score_opus":0.08131317937373102,"score_gpt":0.32950755697898276,"score_spread":0.24819437760525176,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4288363128","genre_codex":"other","genre_gemma":"other","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"other","genre_consensus":"other","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.01178243,0.061923012,0.14023134,0.3221159,0.014030093,0.00017094423,0.00037683712,0.00044223506,0.4489272],"genre_scores_gemma":[0.70723784,0.054025088,0.0852646,0.061293945,0.016110493,0.0011150731,0.00072019384,0.0008604234,0.07337235],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"not_applicable","domain_scores_codex":[0.9454261,0.031074552,0.0031073631,0.006325327,0.012812258,0.0012544326],"domain_scores_gemma":[0.9156161,0.061333954,0.003012093,0.009081216,0.009825093,0.0011314864],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04314915,0.0006958088,0.00096466707,0.0028668395,0.003980434,0.013669412,0.002328775,0.007676096,0.010582245],"category_scores_gemma":[0.07721739,0.0005085829,0.00066735613,0.0040662372,0.021787237,0.02204623,0.005425394,0.009399225,0.0034116697],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000024526697,0.000010971884,0.00021687165,0.00022892997,0.000007357553,0.00003333967,0.001610643,0.00030682518,0.00012690005,0.942093,0.010177403,0.04516319],"study_design_scores_gemma":[0.000021644963,0.00006640378,0.0003701403,0.0011574831,0.000010044324,0.00015949017,0.0015330188,0.0021804972,0.00026725934,0.5955668,0.39862758,0.000039697286],"about_ca_topic_score_codex":0.0024194391,"about_ca_topic_score_gemma":0.0009851416,"teacher_disagreement_score":0.04314915,"about_ca_system_score_codex":0.0069575887,"about_ca_system_score_gemma":0.004953264,"threshold_uncertainty_score":0.22819728},"labels":[],"label_agreement":null},{"id":"W4293235851","doi":"10.1145/3517193","title":"Bash in the Wild: Language Usage, Code Smells, and Bugs","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Security and Verification in Computing","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Scripting language; Programming language; Unix; Software","score_opus":0.07204278200437231,"score_gpt":0.31282443255324427,"score_spread":0.24078165054887196,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293235851","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9957734,0.00027573638,0.0021520623,0.00020049556,0.000009529515,0.00001704355,0.00019060433,0.00053263543,0.0008486108],"genre_scores_gemma":[0.9943984,0.00019286016,0.0033495682,0.00013697159,0.000013652291,0.000024669413,0.0005163317,0.00057295966,0.000794631],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9906607,0.0030644692,0.0007416383,0.0012788571,0.0037741454,0.0004802222],"domain_scores_gemma":[0.8772019,0.07732384,0.025174905,0.009077398,0.009326702,0.0018953393],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00722738,0.00054376165,0.00046366552,0.0049821185,0.0009345382,0.0026522675,0.0009995946,0.00080999784,0.0010149862],"category_scores_gemma":[0.07248842,0.0006690788,0.0003705934,0.0035299002,0.0025640798,0.0053536994,0.0023560456,0.0013909759,0.0004758634],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044424672,0.00030354774,0.77549356,0.00041310117,0.00015448274,0.002272271,0.048135784,0.0014127931,0.007810724,0.0012921045,0.00391293,0.1583544],"study_design_scores_gemma":[0.000026852016,0.0004436031,0.9275613,0.0003885989,0.00015959115,0.0048908736,0.02566991,0.014168364,0.009755937,0.0033098042,0.013420595,0.00020465405],"about_ca_topic_score_codex":0.0031537707,"about_ca_topic_score_gemma":0.0054090214,"teacher_disagreement_score":0.00722738,"about_ca_system_score_codex":0.0008936372,"about_ca_system_score_gemma":0.0006984763,"threshold_uncertainty_score":0.03822249},"labels":[],"label_agreement":null},{"id":"W4293452506","doi":"10.1145/3560263","title":"TokenAware: Accurate and Efficient Bookkeeping Recognition for Token Smart Contracts","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Guelph","funders":"National Key Research and Development Program of China; Natural Science Foundation of Sichuan Province; National Natural Science Foundation of China","keywords":"Bookkeeping; Security token; Computer science; Overhead (engineering); Distributed computing; Transfer (computing); Artificial intelligence; Computer security; Programming language; Operating system; Accounting","score_opus":0.06725597527650143,"score_gpt":0.288131035906911,"score_spread":0.2208750606304096,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4293452506","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.13777642,0.0006036713,0.7776161,0.0003172157,0.00020987004,0.00028608515,0.0019237858,0.07690141,0.004365501],"genre_scores_gemma":[0.7093635,0.00027822008,0.2767235,0.000247159,0.000058685146,0.00022362114,0.0034516677,0.00096071226,0.008692976],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989857,0.000109612505,0.00009114153,0.00032269733,0.00036950604,0.00012130488],"domain_scores_gemma":[0.99831116,0.00035194392,0.0003143427,0.0006568137,0.00025958475,0.00010628059],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0006173121,0.0006917201,0.0006256626,0.0010304428,0.0004217656,0.0010916948,0.0013060115,0.00062577,0.0038921006],"category_scores_gemma":[0.0029800634,0.00044418577,0.00037064002,0.0007436168,0.0005269676,0.0030178013,0.0015288412,0.0008263863,0.0026378112],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0014950339,0.00034606358,0.025998432,0.000359146,0.00010493932,0.00043871542,0.00045036388,0.02872114,0.10811072,0.011027589,0.026403133,0.79654473],"study_design_scores_gemma":[0.000079684716,0.00032477663,0.0059147635,0.000038920432,0.000049218295,0.0005204339,0.0001596412,0.8068296,0.15036526,0.01171689,0.02387881,0.00012202027],"about_ca_topic_score_codex":0.0023843085,"about_ca_topic_score_gemma":0.004536881,"teacher_disagreement_score":0.0038921006,"about_ca_system_score_codex":0.00077359274,"about_ca_system_score_gemma":0.0018250385,"threshold_uncertainty_score":0.013020337},"labels":[],"label_agreement":null},{"id":"W4296132134","doi":"10.1145/3563214","title":"Video Game Bad Smells: What They Are and How Developers Perceive Them","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":22,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Game Developer; Video game development; Video game; Code smell; Relevance (law); Game design; Game development tool; Game testing; Animation; World Wide Web; Software; Game art design; Software development; Game design document; Multimedia; Data science; Software quality","score_opus":0.08737181515015419,"score_gpt":0.29583185583088095,"score_spread":0.20846004068072677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4296132134","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9702262,0.0035343163,0.013462633,0.0020156468,0.00014180123,0.00017702904,0.00025047944,0.0005213987,0.00967046],"genre_scores_gemma":[0.9872953,0.0012153015,0.0066795773,0.00059345487,0.000054095697,0.00009349198,0.00037876246,0.00022210412,0.0034678166],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9912799,0.0033861147,0.000712414,0.00067348283,0.0032986486,0.0006494749],"domain_scores_gemma":[0.958,0.02350099,0.0096844975,0.0016080064,0.005489659,0.001716777],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061164643,0.00077374204,0.00056011707,0.0034213152,0.0012236696,0.0036323573,0.0007179268,0.0016976222,0.0011953304],"category_scores_gemma":[0.059260238,0.00052094215,0.00040135437,0.0016561978,0.001564093,0.0045250645,0.0029956077,0.0012160856,0.0004547568],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00073863164,0.00025747583,0.4279123,0.002121306,0.00014069115,0.0050270776,0.27440208,0.0010327891,0.032361854,0.0052368836,0.015784107,0.2349848],"study_design_scores_gemma":[0.000052152027,0.000757522,0.5259196,0.002658885,0.0001992089,0.011237011,0.31297794,0.00927815,0.0093540605,0.008792397,0.11836229,0.0004107481],"about_ca_topic_score_codex":0.00306757,"about_ca_topic_score_gemma":0.0044915094,"teacher_disagreement_score":0.0061164643,"about_ca_system_score_codex":0.0012871069,"about_ca_system_score_gemma":0.0007787681,"threshold_uncertainty_score":0.03234732},"labels":[],"label_agreement":null},{"id":"W4296592009","doi":"10.1145/3563210","title":"Arachne: Search-Based Repair of Deep Neural Networks","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":65,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Institute for Information and Communications Technology Promotion; Ministry of Science and ICT, South Korea; National Research Foundation of Korea; National Research Foundation","keywords":"Computer science; Debiasing; Deep neural networks; Convolutional neural network; Retraining; Artificial neural network; Artificial intelligence; Machine learning","score_opus":0.04776573394678502,"score_gpt":0.2979497318344487,"score_spread":0.25018399788766366,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4296592009","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.099513814,0.0007705787,0.88189,0.00051065546,0.00020281946,0.00018135803,0.0002532061,0.013163443,0.0035141397],"genre_scores_gemma":[0.7189787,0.0002866495,0.27126163,0.00053146906,0.000054493838,0.00026312814,0.000677752,0.0014523802,0.0064937514],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990558,0.00021024059,0.00008193794,0.00023779547,0.00029494017,0.00011933479],"domain_scores_gemma":[0.9963242,0.0017780953,0.00044440513,0.0009415817,0.00041843642,0.00009326523],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017833071,0.001133157,0.00078043033,0.00073734706,0.00043021532,0.0006413773,0.0025205344,0.0012888697,0.003468681],"category_scores_gemma":[0.0102388365,0.00057328015,0.00083705847,0.00031619053,0.0014484215,0.0018427147,0.0021085166,0.0018094874,0.00076011405],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047269557,0.00017678374,0.0041367644,0.00037363247,0.00015605755,0.0005285449,0.00029944352,0.6809931,0.023708126,0.020468285,0.0072728945,0.26141366],"study_design_scores_gemma":[0.000024085153,0.00011462354,0.00019332903,0.000023704557,0.000019524143,0.00009104164,0.000032091662,0.977948,0.009424293,0.010153447,0.0019654902,0.000010357107],"about_ca_topic_score_codex":0.0029697304,"about_ca_topic_score_gemma":0.004262932,"teacher_disagreement_score":0.003468681,"about_ca_system_score_codex":0.000909056,"about_ca_system_score_gemma":0.0010114902,"threshold_uncertainty_score":0.011603951},"labels":[],"label_agreement":null},{"id":"W4307811455","doi":"10.1145/3569935","title":"Simulator-based Explanation and Debugging of Hazard-triggering Events in DNN-based Safety-critical Systems","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg; European Commission; Université du Luxembourg","keywords":"Debugging; Computer science; Hazard; Software engineering; Programming language","score_opus":0.04752392328673987,"score_gpt":0.3100093273240562,"score_spread":0.26248540403731635,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4307811455","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.21430057,0.000409033,0.77590644,0.0005996611,0.000109603716,0.00009715538,0.00024696588,0.0060854154,0.0022451465],"genre_scores_gemma":[0.9327576,0.0001085774,0.06599442,0.000113035145,0.000008886765,0.000044935696,0.00019483338,0.00012291878,0.0006547268],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99948275,0.00019743777,0.000034083536,0.000120378856,0.00011969885,0.000045587767],"domain_scores_gemma":[0.9971601,0.0017158532,0.000322927,0.00040166042,0.00030859376,0.00009091331],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013055911,0.00080411544,0.00028638216,0.00037095076,0.00019234576,0.0004260976,0.0013521044,0.00072622974,0.0013586695],"category_scores_gemma":[0.007053859,0.00039832148,0.00032263636,0.00013610873,0.00067887397,0.0010359451,0.0008433737,0.0011863698,0.00020747003],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00020516352,0.0000595412,0.0020175588,0.000074605945,0.00002654703,0.00021306872,0.00015527995,0.9368846,0.008124391,0.0029635073,0.0008303709,0.048445374],"study_design_scores_gemma":[0.000005209825,0.000024553949,0.00014302623,0.0000055286096,0.0000035689873,0.000015475453,0.000007971963,0.99342185,0.004429131,0.0016864726,0.00025363723,0.0000036135652],"about_ca_topic_score_codex":0.0046351226,"about_ca_topic_score_gemma":0.0065135756,"teacher_disagreement_score":0.0046351226,"about_ca_system_score_codex":0.0011406173,"about_ca_system_score_gemma":0.0010559814,"threshold_uncertainty_score":0.009216309},"labels":[],"label_agreement":null},{"id":"W4307812665","doi":"10.1145/3569934","title":"What Is the Intended Usage Context of This Model? An Exploratory Study of Pre-Trained Models on Various Model Repositories","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Reuse; Benchmark (surveying); Leverage (statistics); Software engineering; Software; Machine learning; Artificial intelligence; Domain engineering; Code reuse; Context (archaeology); Software development; Software construction; Programming language","score_opus":0.1167313584876961,"score_gpt":0.33314403356947164,"score_spread":0.21641267508177553,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4307812665","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reporting","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reporting","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6800328,0.0035100824,0.238929,0.006705214,0.000526867,0.0006059091,0.01618797,0.02316587,0.030336289],"genre_scores_gemma":[0.83940613,0.0011536148,0.1215656,0.0010314344,0.00010228091,0.0006759895,0.023027012,0.0073623057,0.005675585],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9879892,0.004557082,0.0010523193,0.0023148388,0.00328029,0.0008062051],"domain_scores_gemma":[0.943551,0.024017356,0.0018440903,0.022491828,0.007403007,0.00069276424],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.012548388,0.0012431039,0.0011468759,0.0025113847,0.0013622813,0.0039128475,0.0030680366,0.0018609681,0.0035488245],"category_scores_gemma":[0.07779439,0.000896118,0.001693462,0.0030515115,0.0017557497,0.011906111,0.0023388434,0.0037600067,0.0021176946],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001788959,0.0013260284,0.18019573,0.0028059534,0.00078379654,0.0033674797,0.008119728,0.118346855,0.018500783,0.07009395,0.08310604,0.5115646],"study_design_scores_gemma":[0.00015814458,0.0006088787,0.029205216,0.00093645096,0.00037849226,0.0022134706,0.0033404615,0.80046177,0.03100394,0.03623262,0.09520467,0.00025589718],"about_ca_topic_score_codex":0.010522769,"about_ca_topic_score_gemma":0.014819662,"teacher_disagreement_score":0.9874516,"about_ca_system_score_codex":0.0020288725,"about_ca_system_score_gemma":0.003277014,"threshold_uncertainty_score":0.06636304},"labels":[],"label_agreement":null},{"id":"W4309529433","doi":"10.1145/3571848","title":"On the Discoverability of npm Vulnerabilities in Node.js Projects","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; Concordia University","funders":"","keywords":"Discoverability; Computer science; Dependency (UML); Vulnerability (computing); Computer security; World Wide Web; Software engineering","score_opus":0.0900358588716811,"score_gpt":0.3147622559724983,"score_spread":0.2247263971008172,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4309529433","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99775994,0.000086263964,0.0007491445,0.00012373536,0.0000016410868,0.000017942102,0.00019260202,0.000026732123,0.0010419834],"genre_scores_gemma":[0.9982805,0.0000978716,0.0010451672,0.000018200748,0.0000045027246,0.000021995662,0.00030867133,0.000015952433,0.00020716396],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9916524,0.0019843203,0.00089455943,0.0012555424,0.003366329,0.0008467596],"domain_scores_gemma":[0.6846758,0.23171626,0.061443567,0.008010489,0.011465554,0.0026884477],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011836533,0.0003482178,0.00024109783,0.008141704,0.0011247993,0.00151825,0.00078213523,0.00088373653,0.0015822961],"category_scores_gemma":[0.09871123,0.00035961354,0.0004966157,0.0048430124,0.0020736223,0.004311473,0.0024530524,0.0013171874,0.00025859452],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000076131626,0.00011746355,0.98090404,0.00008018626,0.000036891604,0.0003514042,0.0034550594,0.0015135262,0.0008238605,0.0007058804,0.00029042966,0.011645206],"study_design_scores_gemma":[0.0000048006145,0.00012097512,0.98674345,0.00006219194,0.00003124393,0.0004988566,0.0030316105,0.0069321427,0.0009025509,0.0008802328,0.0007698279,0.000022007876],"about_ca_topic_score_codex":0.009378598,"about_ca_topic_score_gemma":0.017957311,"teacher_disagreement_score":0.011836533,"about_ca_system_score_codex":0.0012927861,"about_ca_system_score_gemma":0.0014421304,"threshold_uncertainty_score":0.06259829},"labels":[],"label_agreement":null},{"id":"W4311938462","doi":"10.1145/3571852","title":"Open Source License Inconsistencies on GitHub","year":2022,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"License; Open source; Computer science; Source code; Open source software; MIT License; Code (set theory); Software; Computer security; World Wide Web; Software engineering; Database; Operating system; Programming language; Set (abstract data type)","score_opus":0.10022283026906519,"score_gpt":0.31426406166282855,"score_spread":0.21404123139376335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4311938462","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9744829,0.0016431182,0.007113569,0.0006340611,0.00008649211,0.00015491372,0.006149779,0.002738986,0.006996147],"genre_scores_gemma":[0.9597013,0.0006124329,0.011123266,0.00027936127,0.00005504875,0.00018314774,0.02328964,0.0020450577,0.0027108728],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.97014284,0.00501876,0.0031647643,0.003642722,0.016255226,0.0017757028],"domain_scores_gemma":[0.8951404,0.046267573,0.022039304,0.014146072,0.020874843,0.0015317608],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.010924285,0.0006443592,0.00074690575,0.017177183,0.0016633439,0.003451245,0.0015892818,0.0007983913,0.0022138509],"category_scores_gemma":[0.09689597,0.0006924846,0.0008396143,0.027756175,0.0024793255,0.0050059194,0.005356524,0.0015250064,0.00083952514],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0007142951,0.00028429565,0.7582278,0.0017412812,0.0005220014,0.0038112088,0.02134182,0.0038935435,0.004926627,0.014409888,0.031525876,0.15860134],"study_design_scores_gemma":[0.000040909363,0.00008060055,0.89072675,0.00082296133,0.00022555025,0.003241875,0.008510874,0.012474208,0.0062297178,0.008223357,0.069250464,0.00017276443],"about_ca_topic_score_codex":0.0146512445,"about_ca_topic_score_gemma":0.013786104,"teacher_disagreement_score":0.9984107,"about_ca_system_score_codex":0.0025521016,"about_ca_system_score_gemma":0.0025207275,"threshold_uncertainty_score":0.05777383},"labels":[],"label_agreement":null},{"id":"W4317209725","doi":"10.1145/3576037","title":"I Depended on You and You Broke Me: An Empirical Study of Manifesting Breaking Changes in Client Packages","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Software versioning; Breaking strength; Change impact analysis; Dependency (UML); Software; Software engineering; Operating system","score_opus":0.14032214583660246,"score_gpt":0.3833204554910334,"score_spread":0.24299830965443092,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4317209725","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9977622,0.00009842802,0.0007596356,0.00020927726,0.0000052673686,0.0000385314,0.000071173126,0.00003561544,0.0010199656],"genre_scores_gemma":[0.9982059,0.00007836037,0.0009085274,0.00013329294,0.000008717967,0.00003930329,0.00016086541,0.000051636012,0.000413389],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9838095,0.008996811,0.0012568786,0.0015312281,0.0035316,0.00087397435],"domain_scores_gemma":[0.71232516,0.2042191,0.04604715,0.018457396,0.014126418,0.0048248502],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.016221542,0.0004553901,0.00040603275,0.0017882022,0.0016456963,0.0025197829,0.0017825171,0.001538369,0.0021473067],"category_scores_gemma":[0.1562494,0.00074580545,0.0003840088,0.0020672497,0.0031198661,0.006903557,0.00275074,0.0036624137,0.0007701556],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042777666,0.0015571672,0.86349803,0.00026240695,0.000118349424,0.0013487634,0.089617066,0.0006134678,0.0020487986,0.0012983491,0.0019867886,0.037223063],"study_design_scores_gemma":[0.000041837873,0.00071066455,0.9283539,0.00020657231,0.00007734775,0.001557556,0.053700846,0.0062521813,0.0014211325,0.0014087552,0.0061733094,0.000095933305],"about_ca_topic_score_codex":0.0057940655,"about_ca_topic_score_gemma":0.0059509533,"teacher_disagreement_score":0.016221542,"about_ca_system_score_codex":0.0014964467,"about_ca_system_score_gemma":0.001192208,"threshold_uncertainty_score":0.08578873},"labels":[],"label_agreement":null},{"id":"W4319296048","doi":"10.1145/3579640","title":"<scp>Katana</scp> : Dual Slicing Based Context for Learning Bug Fixes","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":18,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Program slicing; Debugging; Leverage (statistics); Slicing; Context (archaeology); Program comprehension; Software; Statement (logic); Software engineering; Programming language; Artificial intelligence; Software system; World Wide Web","score_opus":0.0980160634797238,"score_gpt":0.31874940585467515,"score_spread":0.22073334237495135,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319296048","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04069626,0.0020716668,0.87386256,0.002125311,0.00040482296,0.00022349572,0.0043118256,0.071790166,0.00451392],"genre_scores_gemma":[0.34204182,0.0011467064,0.63339144,0.0011374474,0.0003060734,0.00036991911,0.011309786,0.0042566443,0.0060401806],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9991762,0.00018092083,0.0000472297,0.0002851449,0.00025202043,0.00005841157],"domain_scores_gemma":[0.9971419,0.0010551257,0.0002836091,0.0009223632,0.00045186945,0.00014518568],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00084606046,0.0013433011,0.0005894408,0.001503031,0.00062282593,0.00090414606,0.0018718535,0.001400503,0.005967494],"category_scores_gemma":[0.007060108,0.0005835134,0.00083364965,0.00131787,0.00086790987,0.0024818233,0.0018622386,0.002096198,0.0023915388],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0006950329,0.00027714897,0.008787753,0.00073253375,0.0002993096,0.00047040277,0.0004400311,0.08547427,0.035307363,0.00952936,0.13832362,0.71966314],"study_design_scores_gemma":[0.00009130482,0.00028570785,0.0052410127,0.000109537155,0.00011506819,0.00043511434,0.000089190595,0.8986032,0.036872316,0.024046592,0.03401493,0.000096050884],"about_ca_topic_score_codex":0.012689078,"about_ca_topic_score_gemma":0.024255028,"teacher_disagreement_score":0.012689078,"about_ca_system_score_codex":0.0007179716,"about_ca_system_score_gemma":0.0011419692,"threshold_uncertainty_score":0.025230467},"labels":[],"label_agreement":null},{"id":"W4319451761","doi":"10.1145/3583564","title":"Finding Deviated Behaviors of the Compressed DNN Models for Image Classifications","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; Hong Kong University of Science and Technology; University of Waterloo; Cisco Systems","keywords":"Computer science; Artificial intelligence; Task (project management); Image (mathematics); Machine learning; Markov chain; Fitness function; Artificial neural network; Pattern recognition (psychology); Data mining; Genetic algorithm","score_opus":0.1374384112985051,"score_gpt":0.34852953556130783,"score_spread":0.21109112426280274,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319451761","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6567275,0.0015781122,0.3192862,0.0015100603,0.00024494994,0.00022099582,0.0009859035,0.014522641,0.004923607],"genre_scores_gemma":[0.9081795,0.0002649821,0.087611675,0.00042133307,0.000029622483,0.00012725929,0.0016088878,0.0003777068,0.0013789588],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.998982,0.00021715993,0.00007367652,0.00029657737,0.00030350397,0.00012692013],"domain_scores_gemma":[0.9952484,0.0028549836,0.00037757927,0.0007676631,0.00059598114,0.00015556865],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019207406,0.0018729914,0.00077537436,0.0008672229,0.00047353373,0.0010821456,0.0020424519,0.0015064455,0.0017782181],"category_scores_gemma":[0.013020454,0.0006530997,0.0009605799,0.00050080515,0.0008988621,0.00299729,0.0012319565,0.002681914,0.00065269024],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008454837,0.00045034674,0.016400902,0.00033773814,0.00025127034,0.00065195106,0.00028470988,0.6445417,0.02904478,0.0042944797,0.0077838697,0.29511276],"study_design_scores_gemma":[0.000015019484,0.00005316416,0.00057313894,0.000010748303,0.00001362051,0.000045668326,0.000022184786,0.99001837,0.0075811404,0.0013405961,0.0003167194,0.000009547821],"about_ca_topic_score_codex":0.012194132,"about_ca_topic_score_gemma":0.013611826,"teacher_disagreement_score":0.012194132,"about_ca_system_score_codex":0.0016274807,"about_ca_system_score_gemma":0.0016216232,"threshold_uncertainty_score":0.024246275},"labels":[],"label_agreement":null},{"id":"W4319594569","doi":"10.1145/3583566","title":"COMET: Coverage-guided Model Generation For Deep Learning Library Testing","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Key Research and Development Program of China; National Natural Science Foundation of China","keywords":"Comet; Computer science; Layer (electronics); Set (abstract data type); Test set; Artificial intelligence; Machine learning; Algorithm; Data mining; Programming language; Chemistry","score_opus":0.21094581494621803,"score_gpt":0.3372706176748377,"score_spread":0.12632480272861965,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319594569","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12574884,0.0012233091,0.80798423,0.0009854222,0.00014501235,0.00036680378,0.0026354545,0.054515738,0.0063951467],"genre_scores_gemma":[0.6080656,0.0003630286,0.37744516,0.0008249438,0.000037304002,0.0005813542,0.007109942,0.003129092,0.0024435895],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9978654,0.0006125343,0.00016461495,0.0003964342,0.0007428766,0.00021814884],"domain_scores_gemma":[0.99350834,0.0040619373,0.00038195815,0.0010908467,0.00079067,0.00016623904],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001818279,0.0021134617,0.00073282106,0.0016081937,0.00048621465,0.001281762,0.003289008,0.0016291079,0.0043526115],"category_scores_gemma":[0.013057856,0.00095710106,0.0020959089,0.0007716082,0.0011127113,0.0024396693,0.0019710774,0.0019334106,0.0010597672],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005086184,0.00040969087,0.015528831,0.0007710829,0.00026011438,0.0007387292,0.0002712176,0.68451536,0.020827383,0.012064461,0.01845272,0.24565172],"study_design_scores_gemma":[0.00004323537,0.00007853045,0.00027136694,0.000022636254,0.000023622893,0.00008396732,0.000021826156,0.98379576,0.008532496,0.0051069534,0.0020077345,0.0000119321885],"about_ca_topic_score_codex":0.008751678,"about_ca_topic_score_gemma":0.013676884,"teacher_disagreement_score":0.008751678,"about_ca_system_score_codex":0.0019704548,"about_ca_system_score_gemma":0.0033161256,"threshold_uncertainty_score":0.017401516},"labels":[],"label_agreement":null},{"id":"W4319964781","doi":"10.1145/3579642","title":"A Survey on Automated Driving System Testing: Landscapes and Trends","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Vehicular Ad Hoc Networks (VANETs)","field":"Engineering","cited_by":123,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"JST-Mirai Program; Basic Research Program of Jiangsu Province; National Natural Science Foundation of China","keywords":"Computer science; Context (archaeology); Software deployment; Test strategy; Integration testing; Data science; Systems engineering; Software engineering; Risk analysis (engineering); Software; Engineering","score_opus":0.06982938188292201,"score_gpt":0.28702639865301505,"score_spread":0.21719701677009304,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4319964781","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.023359,0.9019483,0.036120255,0.0069615874,0.0009380767,0.00024515504,0.0015496492,0.0010404646,0.027837561],"genre_scores_gemma":[0.08897756,0.8634783,0.030664138,0.0037364955,0.0013261532,0.00021808912,0.0049552144,0.00039781595,0.0062462846],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9934603,0.0014441966,0.0010106597,0.0008258044,0.0029754646,0.00028355158],"domain_scores_gemma":[0.9596898,0.027585143,0.0023342955,0.001101841,0.00864864,0.0006403265],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0049499064,0.0012136677,0.0009895896,0.011184105,0.000575962,0.0024887838,0.0021427688,0.0017048329,0.00539148],"category_scores_gemma":[0.019746684,0.00071329414,0.0008887728,0.012586691,0.0009424525,0.006831521,0.0012933495,0.0012322929,0.0022670806],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00007140407,0.00009456084,0.00962685,0.00907733,0.000060419112,0.00015286889,0.0004903942,0.0018622772,0.0014839468,0.006267944,0.027125977,0.9436861],"study_design_scores_gemma":[0.000026895634,0.00059828296,0.0312369,0.020803962,0.00029979294,0.0035306711,0.0026222535,0.0075410935,0.0039007948,0.01121279,0.9180697,0.00015686685],"about_ca_topic_score_codex":0.0032943478,"about_ca_topic_score_gemma":0.0035323563,"teacher_disagreement_score":0.011184105,"about_ca_system_score_codex":0.0010405197,"about_ca_system_score_gemma":0.0021149695,"threshold_uncertainty_score":0.026177943},"labels":[],"label_agreement":null},{"id":"W4322717938","doi":"10.1145/3624742","title":"Stress Testing Control Loops in Cyber-physical Systems","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Reliability and Analysis Research","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Knut och Alice Wallenbergs Stiftelse; Lunds Universitet; European Commission","keywords":"Leverage (statistics); Computer science; Cyber-physical system; Software; Stress testing (software); Control (management); Control system; Control engineering; Systems engineering; Engineering; Artificial intelligence","score_opus":0.10009437022392521,"score_gpt":0.3358311838909224,"score_spread":0.23573681366699722,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4322717938","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25207093,0.00031823944,0.7435434,0.0003043226,0.00005527852,0.0001313126,0.00010290893,0.0015227563,0.0019509231],"genre_scores_gemma":[0.9688733,0.00006810428,0.030509422,0.000054235672,0.00001474788,0.00008138874,0.00008072995,0.00006646704,0.00025156385],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9961976,0.0014764962,0.00021859117,0.00062175497,0.0011869606,0.00029854284],"domain_scores_gemma":[0.98018897,0.014420227,0.001756941,0.0022123607,0.001087408,0.00033411177],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028407425,0.0013093599,0.00057893066,0.0010844081,0.00038367914,0.0012314938,0.0015633374,0.00127943,0.0016074813],"category_scores_gemma":[0.021181135,0.0003842562,0.0010001751,0.0004121391,0.002834447,0.0024180997,0.0016757062,0.0011871125,0.0001577019],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00047866115,0.00020921172,0.009945535,0.00028470787,0.00007929551,0.00045022718,0.00050886814,0.863649,0.023511598,0.028358523,0.00040925783,0.072115086],"study_design_scores_gemma":[0.000033413435,0.00040883303,0.0015010238,0.000053991567,0.00002033277,0.000131322,0.00009816579,0.9533994,0.016024163,0.027625468,0.00067794725,0.000025929809],"about_ca_topic_score_codex":0.0019237678,"about_ca_topic_score_gemma":0.0011051801,"teacher_disagreement_score":0.0028407425,"about_ca_system_score_codex":0.0011017108,"about_ca_system_score_gemma":0.0009134754,"threshold_uncertainty_score":0.01502347},"labels":[],"label_agreement":null},{"id":"W4362721716","doi":"10.1145/3591870","title":"PatchCensor: Patch Robustness Certification for Transformers via Exhaustive Testing","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"JST-Mirai Program; National Key Research and Development Program of China; Japan Society for the Promotion of Science; Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; Canadian Institute for Advanced Research","keywords":"Computer science; Robustness (evolution); Artificial intelligence; Transformer; Convolutional neural network; Computer security; Software deployment; Computer engineering; Machine learning; Real-time computing; Software engineering; Electrical engineering","score_opus":0.13330348676148915,"score_gpt":0.33313640744718626,"score_spread":0.1998329206856971,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4362721716","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.08708649,0.0023197576,0.8522493,0.001089546,0.00049344136,0.00066060515,0.0010376738,0.04225977,0.012803308],"genre_scores_gemma":[0.80239546,0.000648586,0.1855823,0.00079870125,0.00016385746,0.00044885153,0.0024584117,0.0027868575,0.0047170524],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99450374,0.001316194,0.00038086477,0.0010850584,0.0021154864,0.0005986715],"domain_scores_gemma":[0.9823601,0.0085227,0.0012434222,0.0050570406,0.0022296829,0.0005869571],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0051216413,0.0019654038,0.001565165,0.0018448435,0.00070688745,0.0013479461,0.0033831159,0.0021402114,0.011923277],"category_scores_gemma":[0.032948557,0.0007353997,0.0016930689,0.000662706,0.0028030027,0.004531494,0.0035724943,0.002491381,0.003412743],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00252742,0.0006745541,0.015516168,0.0017890633,0.0005094242,0.0011465122,0.00030646738,0.26173392,0.053823195,0.044276718,0.044998,0.5726986],"study_design_scores_gemma":[0.0002518859,0.0008043732,0.0015236796,0.00014862999,0.00008216672,0.00063182175,0.00011188397,0.9323044,0.02379449,0.033028275,0.007259633,0.00005889936],"about_ca_topic_score_codex":0.0017684147,"about_ca_topic_score_gemma":0.0024984395,"teacher_disagreement_score":0.011923277,"about_ca_system_score_codex":0.0011329758,"about_ca_system_score_gemma":0.0026540842,"threshold_uncertainty_score":0.03988737},"labels":[],"label_agreement":null},{"id":"W4366825753","doi":"10.1145/3593802","title":"Predicting the Change Impact of Resolving Defects by Leveraging the Topics of Issue Reports in Open Source Software Systems","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Queen's University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Leverage (statistics); Data science; Open source; Metric (unit); Source code; Software bug; Change impact analysis; Eclipse; Data mining; Software; Information retrieval; Machine learning","score_opus":0.11775826133334773,"score_gpt":0.35457027254143647,"score_spread":0.23681201120808876,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4366825753","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9853727,0.0007073839,0.011665532,0.0002131856,0.000033874476,0.000055687036,0.0009966599,0.00047518694,0.0004798122],"genre_scores_gemma":[0.98601264,0.00030693188,0.009385046,0.00003800974,0.00006871526,0.00004728774,0.0037349402,0.000049107784,0.0003571521],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9984371,0.00038175102,0.00014833004,0.00039237822,0.0004743649,0.00016609501],"domain_scores_gemma":[0.9770641,0.015569097,0.004146455,0.0009568412,0.0017374066,0.000526124],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036787344,0.0008970299,0.0005759919,0.0048333327,0.00038040458,0.0011808269,0.0007274609,0.0010347093,0.00037723835],"category_scores_gemma":[0.019941809,0.0003271177,0.0008907805,0.0031826862,0.00034351952,0.002265292,0.00084811467,0.0011193948,0.00035386696],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005603969,0.00064600335,0.744312,0.0004925133,0.00019798917,0.0004477146,0.00159155,0.074272655,0.007487223,0.0006131772,0.0044561857,0.16492274],"study_design_scores_gemma":[0.000027603128,0.0003207715,0.39321887,0.000058426576,0.00016250063,0.00030768022,0.0008108505,0.5966714,0.0044548623,0.0012037451,0.002720088,0.00004320094],"about_ca_topic_score_codex":0.0060377303,"about_ca_topic_score_gemma":0.005577978,"teacher_disagreement_score":0.0060377303,"about_ca_system_score_codex":0.0005857543,"about_ca_system_score_gemma":0.00048553868,"threshold_uncertainty_score":0.019455254},"labels":[],"label_agreement":null},{"id":"W4376505229","doi":"10.1145/3597208","title":"An Empirical Study on GitHub Pull Requests’ Reactions","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; École de Technologie Supérieure","funders":"","keywords":"Computer science; Leverage (statistics); Source code; Set (abstract data type); Code review; Open source; Software; Empirical research; Code (set theory); Process (computing); Static program analysis; Software engineering; World Wide Web; Software development; Programming language; Artificial intelligence","score_opus":0.16806167547143225,"score_gpt":0.40876858222722956,"score_spread":0.2407069067557973,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4376505229","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9912498,0.00020528086,0.0028694742,0.0006376542,0.000033632037,0.00029955475,0.00029976454,0.000120596014,0.0042842817],"genre_scores_gemma":[0.9920844,0.00032808588,0.0033875396,0.0006443077,0.00005461203,0.000678525,0.0004967323,0.00012312306,0.0022027337],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9676914,0.018039312,0.0022112692,0.0023867965,0.008037245,0.0016340072],"domain_scores_gemma":[0.65195394,0.2472757,0.04651977,0.0085525485,0.039596867,0.0061010895],"candidate_categories":["metaresearch","bibliometrics"],"consensus_categories":[],"category_scores_codex":[0.023799697,0.00075500726,0.0004951917,0.0038120197,0.0015008964,0.0035212093,0.0011883344,0.0017121118,0.0030114083],"category_scores_gemma":[0.15617275,0.00057331845,0.0004408941,0.002774784,0.0019558596,0.0038499467,0.002793472,0.0026892999,0.0016484093],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00085761613,0.0016739968,0.51862043,0.0023496677,0.00010924943,0.0019553693,0.35882372,0.00053889805,0.011210864,0.0019949493,0.007592035,0.094273195],"study_design_scores_gemma":[0.00007704987,0.0014346876,0.649215,0.0011147591,0.00008664751,0.0012042723,0.29661113,0.0053727184,0.0069591794,0.0011755174,0.036508683,0.00024042116],"about_ca_topic_score_codex":0.0019550687,"about_ca_topic_score_gemma":0.002114036,"teacher_disagreement_score":0.996188,"about_ca_system_score_codex":0.0020232669,"about_ca_system_score_gemma":0.0017144049,"threshold_uncertainty_score":0.12586635},"labels":[],"label_agreement":null},{"id":"W4379014622","doi":"10.1145/3603110","title":"Dependency Update Strategies and Package Characteristics","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec à Montréal; Concordia University","funders":"","keywords":"Computer science; Dependency (UML); Software versioning; Software engineering; Dilemma; Software; Programming language","score_opus":0.07638883686026998,"score_gpt":0.32458871075136997,"score_spread":0.2481998738911,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4379014622","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98067284,0.00021998509,0.014158143,0.00019947423,0.000014565683,0.00008283663,0.00052968215,0.000340938,0.00378138],"genre_scores_gemma":[0.99187696,0.00008012418,0.0061546317,0.000027342243,0.000007102756,0.000042986067,0.00068527507,0.0001331982,0.0009924062],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99552745,0.0015024488,0.00035357068,0.00081350066,0.0014810555,0.00032198933],"domain_scores_gemma":[0.8791574,0.081227206,0.018567218,0.008705438,0.010157777,0.0021849296],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061071836,0.0005479517,0.00034075064,0.002146082,0.0006246971,0.001935728,0.0009451137,0.0005555774,0.0022877129],"category_scores_gemma":[0.07588558,0.0005679655,0.0004859971,0.002056025,0.00078942836,0.0039345887,0.0012484863,0.0009952011,0.00072505034],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014909216,0.00016686115,0.91723734,0.00010772584,0.00009707988,0.00021786802,0.0018064934,0.010751565,0.0019489719,0.0016038961,0.001970578,0.06394254],"study_design_scores_gemma":[0.000022863383,0.0003578591,0.9054407,0.00007349465,0.00016406797,0.0008961281,0.00211362,0.0743191,0.0040715844,0.0035100945,0.008936467,0.000094059826],"about_ca_topic_score_codex":0.0050614844,"about_ca_topic_score_gemma":0.008134998,"teacher_disagreement_score":0.0061071836,"about_ca_system_score_codex":0.0010712668,"about_ca_system_score_gemma":0.0007990565,"threshold_uncertainty_score":0.032298267},"labels":[],"label_agreement":null},{"id":"W4382987313","doi":"10.1145/3607186","title":"What Constitutes the Deployment and Runtime Configuration System? An Empirical Study on OpenStack Projects","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université du Québec","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Software deployment; Computer science; Configuration Management (ITSM); Software configuration management; Operating system; Software; Leverage (statistics); System deployment; Software system; Software construction","score_opus":0.1711848459207054,"score_gpt":0.38614101168772763,"score_spread":0.21495616576702223,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4382987313","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9958882,0.00018846852,0.000979735,0.00036007617,0.000008489584,0.000054842916,0.00012876409,0.000022818973,0.0023685803],"genre_scores_gemma":[0.9980568,0.00016182917,0.0009933984,0.00006905436,0.000008880333,0.00006571509,0.00022564772,0.000040577426,0.00037802398],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9816839,0.010393904,0.0013494,0.0021638835,0.0033346447,0.0010741615],"domain_scores_gemma":[0.79061586,0.15101002,0.02859617,0.008070932,0.015699018,0.0060079913],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.020025196,0.0003549319,0.0003447891,0.0028656977,0.0024279498,0.0047092303,0.0016046873,0.0013960712,0.0023340143],"category_scores_gemma":[0.14775184,0.00053542596,0.00026638084,0.0038496233,0.0045640487,0.012098288,0.0024887493,0.0022261855,0.0004968802],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004575031,0.0010423646,0.72709584,0.00071967894,0.000070746195,0.0011487397,0.18752904,0.0008929113,0.0021925922,0.006861565,0.00482216,0.067166805],"study_design_scores_gemma":[0.000027570164,0.00048404405,0.7494814,0.00046310667,0.000033491713,0.00068191695,0.22462374,0.004087943,0.0009013017,0.0018629928,0.017250443,0.00010211995],"about_ca_topic_score_codex":0.0043965965,"about_ca_topic_score_gemma":0.00557425,"teacher_disagreement_score":0.020025196,"about_ca_system_score_codex":0.0024308157,"about_ca_system_score_gemma":0.0017370981,"threshold_uncertainty_score":0.10590464},"labels":[],"label_agreement":null},{"id":"W4383067302","doi":"10.1145/3607179","title":"A Systematic Review of Automated Query Reformulations in Source Code Search","year":2023,"lang":"en","type":"review","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Saskatchewan; Dalhousie University","funders":"Natural Sciences and Engineering Research Council of Canada; Dalhousie University","keywords":"Computer science; Information retrieval; Query expansion; Web search query; Vocabulary; Term (time); Weighting; Software; Data mining; Search engine; Programming language","score_opus":0.17950256895553524,"score_gpt":0.4152454057738076,"score_spread":0.23574283681827235,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4383067302","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0020059494,0.9916237,0.0023802675,0.0007307176,0.00013893063,0.0011938413,0.000799968,0.00006979678,0.0010568236],"genre_scores_gemma":[0.01736971,0.9631407,0.013228681,0.0011055791,0.00009411327,0.003123106,0.0015646603,0.000059241243,0.00031418217],"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","domain_scores_codex":[0.9491001,0.02657936,0.0136892805,0.0023171033,0.007689528,0.0006245845],"domain_scores_gemma":[0.7873412,0.179024,0.012812592,0.004158499,0.015992884,0.0006707929],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.04711116,0.0023591174,0.006188067,0.03154381,0.0012685824,0.0041101715,0.004326847,0.0021336733,0.00527941],"category_scores_gemma":[0.2089029,0.0017225253,0.006208858,0.030893443,0.0018922772,0.008180022,0.003702886,0.0017781418,0.0011133787],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024019567,0.00004647575,0.00072317285,0.7064533,0.0020920052,0.000119561984,0.0014267216,0.0003319982,0.00045019534,0.0010947805,0.005431553,0.2815901],"study_design_scores_gemma":[0.00031920173,0.00048355578,0.004254841,0.87161624,0.018600902,0.0006003876,0.0016620487,0.000518346,0.0011085256,0.0019290486,0.09878546,0.00012140354],"about_ca_topic_score_codex":0.0102202585,"about_ca_topic_score_gemma":0.02706047,"teacher_disagreement_score":0.04711116,"about_ca_system_score_codex":0.006258567,"about_ca_system_score_gemma":0.026865771,"threshold_uncertainty_score":0.24915063},"labels":[],"label_agreement":null},{"id":"W4383555717","doi":"10.1145/3607185","title":"Programming by Example Made Easy","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Fundamental Research Funds for the Central Universities; Natural Science Foundation of Jiangsu Province; National Natural Science Foundation of China; Impact Fund; National Science Foundation","keywords":"Computer science; Usability; Domain (mathematical analysis); Table (database); Exploit; Constraint programming; Domain-specific language; Software engineering; Range (aeronautics); Programming language; Theoretical computer science; Data mining; Human–computer interaction; Mathematical optimization","score_opus":0.10711553213041111,"score_gpt":0.3310986989582408,"score_spread":0.22398316682782968,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4383555717","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.006089012,0.00040650088,0.9353393,0.0011641376,0.0002068541,0.00033081777,0.0006030229,0.01313688,0.042723477],"genre_scores_gemma":[0.06440716,0.0007435026,0.90602314,0.000710994,0.00008328006,0.0004505436,0.0016062873,0.003607291,0.022367865],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.99702924,0.0009352061,0.00024520047,0.00068566087,0.0009122901,0.00019234775],"domain_scores_gemma":[0.992863,0.0037890451,0.0002982686,0.0020211525,0.00083652535,0.00019195655],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002497311,0.0011586419,0.0006402926,0.00086709816,0.0009375008,0.002741359,0.0022422087,0.0010370901,0.040814526],"category_scores_gemma":[0.014669173,0.00079852826,0.0012877138,0.0008779922,0.0012454707,0.0056747356,0.0035912183,0.0023038161,0.010097653],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028706618,0.00023450203,0.0014155505,0.0012881043,0.00009460193,0.0005352995,0.0014167378,0.0084985215,0.020156376,0.23100646,0.066913076,0.6681537],"study_design_scores_gemma":[0.00010828804,0.00016767406,0.0007105792,0.00042165318,0.00008123614,0.0012824503,0.00040456842,0.055640686,0.02551927,0.18340474,0.73218215,0.00007667532],"about_ca_topic_score_codex":0.0009045368,"about_ca_topic_score_gemma":0.0017794373,"teacher_disagreement_score":0.040814526,"about_ca_system_score_codex":0.0004929002,"about_ca_system_score_gemma":0.0016569517,"threshold_uncertainty_score":0.13653815},"labels":[],"label_agreement":null},{"id":"W4385222146","doi":"10.1145/3585005","title":"<i>ArchRepair</i> : Block-Level Architecture-Oriented Repairing for Deep Neural Networks","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"JST-Mirai Program","keywords":"Computer science; Robustness (evolution); Block (permutation group theory); Deep neural networks; Overfitting; Architecture; Artificial neural network; Artificial intelligence; Network architecture; Machine learning; Retraining; Distributed computing; Computer engineering; Computer architecture; Computer network","score_opus":0.06776237405796683,"score_gpt":0.31016859040160144,"score_spread":0.24240621634363463,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4385222146","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.008349636,0.0013461608,0.9770735,0.0011584546,0.0004663121,0.00007572007,0.0002504905,0.006161443,0.0051182387],"genre_scores_gemma":[0.4406445,0.002922006,0.5218148,0.0022475112,0.00058521604,0.00031913308,0.0025226986,0.0028116205,0.02613249],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99897206,0.00016536427,0.000088087516,0.00018861219,0.0004795353,0.00010626488],"domain_scores_gemma":[0.9974968,0.0004245411,0.0002579477,0.0012532942,0.00047775885,0.00008965598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0015215324,0.0012769264,0.00066965615,0.0006586635,0.00075404486,0.001776526,0.0031164286,0.0017082195,0.0075925174],"category_scores_gemma":[0.0061769984,0.0005235306,0.00089365843,0.0006918916,0.0016655567,0.0031141657,0.0036878632,0.003442863,0.0034556496],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037371955,0.00010788774,0.0024372155,0.0004873104,0.00019987099,0.0006474789,0.00032704603,0.2891122,0.047010083,0.08330197,0.074690364,0.50130475],"study_design_scores_gemma":[0.000017801956,0.0001188815,0.00043298875,0.00009133129,0.00004830571,0.0004392004,0.000051474686,0.88103443,0.039974954,0.040725086,0.03701906,0.000046484573],"about_ca_topic_score_codex":0.0034820905,"about_ca_topic_score_gemma":0.0047303126,"teacher_disagreement_score":0.0075925174,"about_ca_system_score_codex":0.0011673104,"about_ca_system_score_gemma":0.0012905145,"threshold_uncertainty_score":0.025399566},"labels":[],"label_agreement":null},{"id":"W4386031629","doi":"10.1145/3617171","title":"<scp>StubCoder</scp> : Automated Generation and Repair of Stub Code for Mock Objects","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China; Hong Kong University of Science and Technology; Impact Fund; National Science Foundation","keywords":"Stub (electronics); Computer science; Unit testing; Regression testing; Leverage (statistics); Test case; Programming language; Software; Software development; Artificial intelligence; Engineering; Structural engineering; Machine learning; Software construction","score_opus":0.12252016812216412,"score_gpt":0.3394841656674571,"score_spread":0.21696399754529297,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386031629","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015290251,0.00013265044,0.77185965,0.00032820785,0.00007480434,0.00036536134,0.0012941601,0.20523731,0.005417568],"genre_scores_gemma":[0.18242802,0.0003031058,0.7517792,0.0004901771,0.00006828165,0.00048812127,0.008092695,0.04224616,0.0141042555],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99857664,0.00030401628,0.00009884827,0.00028028857,0.00061523094,0.00012510187],"domain_scores_gemma":[0.9919104,0.0028066572,0.0007797426,0.0029720436,0.0013073395,0.00022383375],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020528086,0.0016229153,0.0005106226,0.0015195705,0.0008248613,0.0012525164,0.0024315445,0.0016820151,0.011652511],"category_scores_gemma":[0.008709342,0.00089261984,0.0008550024,0.0007322828,0.001895882,0.0023701773,0.001873587,0.0012497546,0.006893845],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005238002,0.00041625247,0.010164193,0.00116461,0.0001384721,0.0023130884,0.001301034,0.035731673,0.12691456,0.025014063,0.16953117,0.626787],"study_design_scores_gemma":[0.0002528722,0.0005934507,0.0071265306,0.0002588009,0.00007812029,0.003621212,0.00017164557,0.43789604,0.32542413,0.020513449,0.20381963,0.00024407925],"about_ca_topic_score_codex":0.0036788129,"about_ca_topic_score_gemma":0.0051545417,"teacher_disagreement_score":0.011652511,"about_ca_system_score_codex":0.0006945487,"about_ca_system_score_gemma":0.0014055255,"threshold_uncertainty_score":0.038981497},"labels":[],"label_agreement":null},{"id":"W4386136237","doi":"10.1145/3617168","title":"Faire: Repairing Fairness of Neural Networks via Neuron Condition Synthesis","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"Natural Sciences and Engineering Research Council of Canada; National Research Foundation; Bộ Giáo dục và Ðào tạo; Ministry of Education - Singapore; National Research Foundation Singapore; Canadian Institute for Advanced Research","keywords":"Computer science; Deep neural networks; Retraining; Robustness (evolution); Overhead (engineering); Trustworthiness; Artificial intelligence; Machine learning; Artificial neural network; Computer security; Programming language","score_opus":0.04843400212564353,"score_gpt":0.30030590532651313,"score_spread":0.2518719032008696,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386136237","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.031200582,0.0002891309,0.9623219,0.0002454819,0.00013097726,0.0001136085,0.00009943956,0.0035289372,0.0020700223],"genre_scores_gemma":[0.76457274,0.00019888929,0.23049535,0.000297684,0.000075816984,0.00027322906,0.0002570185,0.00046111917,0.0033681851],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983228,0.00027648453,0.00012859183,0.0004927421,0.00055359903,0.00022570656],"domain_scores_gemma":[0.99610436,0.0022063064,0.00038758866,0.00064605224,0.0005186504,0.00013709017],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026106504,0.0012178905,0.0010664769,0.00082181726,0.00072296866,0.0011007157,0.0019642874,0.0012632458,0.004812771],"category_scores_gemma":[0.012721211,0.00054831983,0.001058636,0.0003343366,0.0019816058,0.0021280695,0.0022753682,0.0020546936,0.00046597555],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044501593,0.00010942,0.0020163925,0.00022292169,0.0000871185,0.0003920613,0.00019613218,0.74020344,0.01824021,0.03708401,0.0024200575,0.19858328],"study_design_scores_gemma":[0.000024263192,0.000055889843,0.00008437199,0.0000151503655,0.000018624005,0.000037450445,0.000012931178,0.9739595,0.0075417873,0.017402165,0.00083744345,0.000010445107],"about_ca_topic_score_codex":0.0038231032,"about_ca_topic_score_gemma":0.004042197,"teacher_disagreement_score":0.004812771,"about_ca_system_score_codex":0.0013517998,"about_ca_system_score_gemma":0.0023581088,"threshold_uncertainty_score":0.016100347},"labels":[],"label_agreement":null},{"id":"W4386191524","doi":"10.1145/3617176","title":"Probabilistic Safe WCET Estimation for Weakly Hard Real-time Systems at Design Stages","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Real-Time Systems Scheduling","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Mitacs; European Commission; Université du Luxembourg","keywords":"Computer science; Leverage (statistics); Probabilistic logic; Worst-case execution time; Scheduling (production processes); Reliability engineering; Execution time; Distributed computing; Mathematical optimization; Machine learning; Artificial intelligence","score_opus":0.11398602655715183,"score_gpt":0.31539242308225535,"score_spread":0.2014063965251035,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386191524","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06271837,0.00033400848,0.93501365,0.00022872238,0.000011193448,0.000039048482,0.00010436739,0.00064499566,0.000905634],"genre_scores_gemma":[0.89095664,0.00019121556,0.107070595,0.00009482997,0.00003217808,0.000097628246,0.00028830042,0.00016520088,0.0011033803],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977956,0.00089395285,0.00010909129,0.00049146096,0.00052370084,0.00018618495],"domain_scores_gemma":[0.98967934,0.007600859,0.0012641258,0.00066097354,0.00057511276,0.00021953013],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027074984,0.001202572,0.00095945754,0.0015165197,0.00036419908,0.0011448271,0.0013609119,0.0009967497,0.0011115851],"category_scores_gemma":[0.016344452,0.00067248486,0.0008444014,0.00083305925,0.0011230583,0.0014380608,0.0012590446,0.0017019794,0.000297059],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000081252336,0.00002204115,0.0022170339,0.0000517404,0.000022190961,0.0000640482,0.000045921934,0.97659814,0.002155515,0.0020031822,0.00017987686,0.016559014],"study_design_scores_gemma":[0.0000035453406,0.000013739649,0.00040437395,0.000005381124,0.0000039299525,0.000012743456,0.000008360632,0.9951035,0.00081452174,0.00349794,0.00012783948,0.000004285152],"about_ca_topic_score_codex":0.0050542857,"about_ca_topic_score_gemma":0.004923653,"teacher_disagreement_score":0.0050542857,"about_ca_system_score_codex":0.0010862764,"about_ca_system_score_gemma":0.0013948512,"threshold_uncertainty_score":0.014318764},"labels":[],"label_agreement":null},{"id":"W4386442953","doi":"10.1145/3617172","title":"On the Caching Schemes to Speed Up Program Reduction","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Reduction (mathematics); Debugging; Compiler; Cache; Parallel computing; Process (computing); ENCODE; Memory footprint; Encoding (memory); Compile time; Computation; Theoretical computer science; Computer engineering; Algorithm; Programming language; Artificial intelligence","score_opus":0.1289523923715804,"score_gpt":0.35603718592683653,"score_spread":0.22708479355525613,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386442953","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17682482,0.012198086,0.786121,0.0029232777,0.00027658432,0.0004778983,0.00031968465,0.008951737,0.01190693],"genre_scores_gemma":[0.58480906,0.0037329884,0.40402672,0.0007432545,0.00024121368,0.00040698136,0.00048651957,0.0010158087,0.0045374567],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9955525,0.0011731021,0.00033661007,0.0006451344,0.0016188261,0.00067389815],"domain_scores_gemma":[0.97883534,0.009344569,0.0016499566,0.00794138,0.0019233073,0.00030549493],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028958225,0.0012965748,0.0010309805,0.0025250558,0.0012544427,0.0019169544,0.0037762655,0.0012105054,0.0034800195],"category_scores_gemma":[0.02042489,0.00081539236,0.0011056093,0.0033947695,0.0028206003,0.00988454,0.002476002,0.0024638718,0.00090893515],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010784277,0.00040533204,0.007951087,0.0010507398,0.0001427265,0.0003756725,0.0009589081,0.10121077,0.059328657,0.25530916,0.013601387,0.5585872],"study_design_scores_gemma":[0.00019734597,0.0009363694,0.0028312942,0.0004631087,0.0003573363,0.0010548909,0.00035812714,0.73592263,0.10947251,0.11066031,0.037577163,0.00016884983],"about_ca_topic_score_codex":0.005671298,"about_ca_topic_score_gemma":0.0061427946,"teacher_disagreement_score":0.005671298,"about_ca_system_score_codex":0.0028819838,"about_ca_system_score_gemma":0.0042944956,"threshold_uncertainty_score":0.020910382},"labels":[],"label_agreement":null},{"id":"W4386830495","doi":"10.1145/3624740","title":"<i>LoGenText-Plus</i> : Improving Neural Machine Translation Based Logging Texts Generation with Syntactic Templates","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":15,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; University of Waterloo","funders":"","keywords":"Computer science; Logging; Source code; Code (set theory); Context (archaeology); Template; Natural language processing; Artificial intelligence; Database; Programming language; Set (abstract data type)","score_opus":0.09871842548096321,"score_gpt":0.3114374323806604,"score_spread":0.2127190068996972,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386830495","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06686246,0.0011915951,0.7669296,0.0019524817,0.0011298975,0.0006817658,0.005412722,0.14091447,0.014924954],"genre_scores_gemma":[0.19076154,0.00052544713,0.7572302,0.0013230776,0.00026731478,0.0005903712,0.02183441,0.0060637044,0.021403858],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987048,0.00039604882,0.00011322591,0.0003736582,0.00033190614,0.000080375234],"domain_scores_gemma":[0.9963819,0.0015133884,0.0001780248,0.0009165876,0.0008940443,0.00011601343],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011888337,0.002038308,0.00073983136,0.001419848,0.00072149094,0.0016993895,0.0019559555,0.0015359246,0.009202589],"category_scores_gemma":[0.007228444,0.00054064766,0.0010393952,0.0010364604,0.0007756677,0.0033640552,0.0017889547,0.0021751567,0.0071123783],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00044118922,0.0005734803,0.0017354698,0.0009358338,0.00014716099,0.00076550175,0.00060973095,0.030376699,0.042102322,0.0063924,0.11898388,0.7969364],"study_design_scores_gemma":[0.00022475966,0.00033205844,0.0016425181,0.00010324437,0.00012479327,0.00061510515,0.00028168576,0.8262599,0.0904217,0.008872005,0.070989765,0.00013244126],"about_ca_topic_score_codex":0.0061825044,"about_ca_topic_score_gemma":0.010846771,"teacher_disagreement_score":0.009202589,"about_ca_system_score_codex":0.0007677202,"about_ca_system_score_gemma":0.0021435786,"threshold_uncertainty_score":0.03078574},"labels":[],"label_agreement":null},{"id":"W4386845625","doi":"10.1145/3624739","title":"Understanding the Helpfulness of Stale Bot for Pull-Based Development: An Empirical Study of 20 Large Open-Source Projects","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Helpfulness; Computer science; Abandonment (legal); Process (computing); Workflow; Process management; Code refactoring; Risk analysis (engineering); Business; Operating system; Database; Psychology","score_opus":0.3482458636295181,"score_gpt":0.39795572836972043,"score_spread":0.049709864740202336,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386845625","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99781346,0.000089869725,0.0009511517,0.00016095006,0.0000038929993,0.000060457438,0.000056081197,0.000050431147,0.0008136175],"genre_scores_gemma":[0.99649966,0.000115582065,0.0024514014,0.000105643005,0.0000071501095,0.0000971886,0.00014451245,0.000036631347,0.00054212613],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98591137,0.0075088907,0.00096109306,0.0016792368,0.003135319,0.00080415333],"domain_scores_gemma":[0.66575664,0.25033948,0.04244902,0.009040871,0.023963304,0.008450756],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.025691511,0.00047581102,0.00042424919,0.0029711712,0.0016675569,0.0032044658,0.0013747031,0.0012990829,0.0015243522],"category_scores_gemma":[0.1375386,0.0006544658,0.00029619195,0.0019071274,0.0018714124,0.0059376713,0.002356291,0.0020190033,0.0006205282],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005596717,0.002269581,0.81263196,0.0007993707,0.00014425007,0.0012199631,0.09814208,0.0010078526,0.0048742266,0.0010301063,0.0024437567,0.07487712],"study_design_scores_gemma":[0.00009177447,0.0023615162,0.8824184,0.0004281593,0.00013755502,0.001067411,0.083818935,0.015459381,0.002551041,0.0012307427,0.010262652,0.00017243544],"about_ca_topic_score_codex":0.0033614966,"about_ca_topic_score_gemma":0.0073905755,"teacher_disagreement_score":0.025691511,"about_ca_system_score_codex":0.0013794814,"about_ca_system_score_gemma":0.0018841111,"threshold_uncertainty_score":0.13587129},"labels":[],"label_agreement":null},{"id":"W4386889431","doi":"10.1145/3624745","title":"Search-Based Software Testing Driven by Automatically Generated and Manually Defined Fitness Functions","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University","funders":"Natural Sciences and Engineering Research Council of Canada; Compute Canada; McMaster University","keywords":"Fitness function; Computer science; Domain (mathematical analysis); Software; Set (abstract data type); Function (biology); Search-based software engineering; Software engineering; Artificial intelligence; Machine learning; Data mining; Software system; Programming language; Software construction; Genetic algorithm","score_opus":0.09860297203966012,"score_gpt":0.31119659442560965,"score_spread":0.21259362238594953,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4386889431","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1290902,0.00023592387,0.8542691,0.00015449277,0.000030423955,0.00023685559,0.00020836455,0.012783983,0.0029906922],"genre_scores_gemma":[0.6064823,0.00007485714,0.3902706,0.00010249268,0.000011853182,0.00028655076,0.00067996513,0.0009614552,0.0011299254],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9958812,0.0017187543,0.00023686494,0.0006239066,0.0012732113,0.00026602452],"domain_scores_gemma":[0.985352,0.010019418,0.0009893216,0.001964724,0.001461238,0.00021332063],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0035800338,0.0016629044,0.0010458486,0.0023210568,0.00032805404,0.0009935308,0.002078093,0.0012159137,0.0022480728],"category_scores_gemma":[0.018396992,0.0005741015,0.0009089412,0.00087214395,0.0009922803,0.0015472951,0.0013851032,0.0008504344,0.000624601],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00066605845,0.0006795098,0.013999213,0.0004405019,0.00020724225,0.00036491704,0.00041563218,0.5785672,0.05101606,0.009334039,0.0035717303,0.3407378],"study_design_scores_gemma":[0.000051508076,0.00011113146,0.00085007946,0.000020209765,0.000017827659,0.00007115717,0.000022623484,0.9882684,0.008176557,0.0017706408,0.0006242563,0.00001555579],"about_ca_topic_score_codex":0.0043889075,"about_ca_topic_score_gemma":0.0056358185,"teacher_disagreement_score":0.0043889075,"about_ca_system_score_codex":0.001024836,"about_ca_system_score_gemma":0.001634885,"threshold_uncertainty_score":0.018933237},"labels":[],"label_agreement":null},{"id":"W4387735187","doi":"10.1145/3628159","title":"Generation-based Differential Fuzzing for Deep Learning Libraries","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"JST-Mirai Program; Science, Technology and Innovation Commission of Shenzhen Municipality; National Natural Science Foundation of China","keywords":"Fuzz testing; Computer science; Context (archaeology); Task (project management); Machine learning; Artificial intelligence; Deep learning; Benchmark (surveying); Software engineering; Software; Programming language","score_opus":0.14059899809814053,"score_gpt":0.32340425284178986,"score_spread":0.18280525474364934,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387735187","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25136557,0.00085195,0.7299873,0.00079611305,0.000099010096,0.0003395923,0.0004245159,0.012764764,0.003371082],"genre_scores_gemma":[0.840384,0.00012835726,0.15633734,0.00049445545,0.000019891617,0.00020805292,0.00059695804,0.00033870927,0.0014921062],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9977336,0.00054848084,0.00018822984,0.0005739673,0.0006290601,0.000326608],"domain_scores_gemma":[0.9938128,0.003779516,0.0005130239,0.0009491626,0.00078264,0.0001629168],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00228105,0.0013306475,0.0008500577,0.0015238872,0.00062123203,0.0010012706,0.002987944,0.0014829491,0.0022390191],"category_scores_gemma":[0.0113723725,0.0006433474,0.0014733388,0.0006277882,0.0018695082,0.0028148955,0.001957968,0.001946809,0.00034414395],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00055571424,0.0004470185,0.020447277,0.0004871466,0.00015043005,0.00094709115,0.00042321248,0.50343144,0.031036515,0.019024286,0.0040436764,0.41900608],"study_design_scores_gemma":[0.000035841033,0.000103626124,0.0006717679,0.00002772432,0.000031692525,0.0001152516,0.000028775195,0.9716476,0.013594061,0.012870636,0.00085548573,0.000017436385],"about_ca_topic_score_codex":0.005240703,"about_ca_topic_score_gemma":0.0076090195,"teacher_disagreement_score":0.005240703,"about_ca_system_score_codex":0.0022477435,"about_ca_system_score_gemma":0.0024167842,"threshold_uncertainty_score":0.016308665},"labels":[],"label_agreement":null},{"id":"W4387869001","doi":"10.1145/3630009","title":"The Good, the Bad, and the Missing: Neural Code Generation for Machine Learning Tasks","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada; National Natural Science Foundation of China","keywords":"Computer science; Snippet; Artificial intelligence; Code (set theory); Code generation; Machine learning; Artificial neural network; Construct (python library); Set (abstract data type); Programming language; Natural language processing","score_opus":0.10990026194072332,"score_gpt":0.33788673840129657,"score_spread":0.22798647646057324,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4387869001","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.53225744,0.009701511,0.4262698,0.0042275484,0.00043004417,0.0004561643,0.0015269882,0.015477096,0.009653416],"genre_scores_gemma":[0.7354818,0.0015009484,0.25550506,0.0007565064,0.00008323233,0.0004297448,0.0028238953,0.0006267389,0.0027921475],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9968021,0.0014979339,0.00024360412,0.00056551356,0.00072766305,0.00016320779],"domain_scores_gemma":[0.9866261,0.009543783,0.00076598045,0.0014828017,0.0013181923,0.0002630992],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0041296105,0.0010442173,0.0005683819,0.0013173411,0.0005649148,0.0012346414,0.0017305302,0.0014082105,0.0011788377],"category_scores_gemma":[0.021779567,0.00038056145,0.00065346947,0.0011591071,0.0011058542,0.0028582963,0.0013751588,0.0021661092,0.00060907705],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00096134737,0.0005498669,0.016872736,0.001091159,0.0001668452,0.000297874,0.0004746491,0.26278526,0.010054864,0.009590357,0.015297879,0.6818571],"study_design_scores_gemma":[0.00010104086,0.0002597223,0.0026518821,0.00009743898,0.000059071437,0.00012610875,0.00009445785,0.969383,0.009819735,0.013327458,0.0040429775,0.00003709741],"about_ca_topic_score_codex":0.0056711156,"about_ca_topic_score_gemma":0.0080198385,"teacher_disagreement_score":0.0056711156,"about_ca_system_score_codex":0.00155198,"about_ca_system_score_gemma":0.0015435441,"threshold_uncertainty_score":0.021839678},"labels":[],"label_agreement":null},{"id":"W4388848588","doi":"10.1145/3631972","title":"Improving Automated Program Repair with Domain Adaptation","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Adaptation (eye); Software engineering; Domain (mathematical analysis)","score_opus":0.07110201268549876,"score_gpt":0.31706774702128526,"score_spread":0.2459657343357865,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4388848588","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.44869503,0.0023910818,0.518744,0.001013421,0.00021714016,0.00035410447,0.00068420573,0.022433562,0.0054674293],"genre_scores_gemma":[0.836714,0.0004856016,0.1574062,0.00055016164,0.000051646955,0.0002282085,0.0017624474,0.00031920723,0.0024825272],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987023,0.00047588395,0.000079459765,0.0004385847,0.00018879803,0.0001149727],"domain_scores_gemma":[0.9961838,0.0019198692,0.000316813,0.00083871715,0.00058959564,0.00015118015],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018919746,0.0013656821,0.0007650778,0.0011124415,0.0003322243,0.0007605039,0.0017923223,0.0012954583,0.0012956804],"category_scores_gemma":[0.009198681,0.0004237179,0.000970391,0.0008138012,0.0005095524,0.0022518837,0.0016844687,0.0018891034,0.00075080997],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00019496735,0.0006968865,0.010850823,0.00024902695,0.000110477755,0.00015395676,0.00027455974,0.5600482,0.011114928,0.0011909998,0.006024066,0.40909117],"study_design_scores_gemma":[0.000033220946,0.000121159814,0.0012843115,0.0000186991,0.000031494736,0.000055210716,0.00007981269,0.9902293,0.0041747773,0.0017346401,0.0022206015,0.000016707727],"about_ca_topic_score_codex":0.005605346,"about_ca_topic_score_gemma":0.004951067,"teacher_disagreement_score":0.005605346,"about_ca_system_score_codex":0.0008632364,"about_ca_system_score_gemma":0.0014545594,"threshold_uncertainty_score":0.011145413},"labels":[],"label_agreement":null},{"id":"W4389386408","doi":"10.1145/3635708","title":"Learning-based Relaxation of Completeness Requirements for Data Entry Forms","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"BNP Paribas Cardif; Université du Luxembourg","keywords":"Completeness (order theory); Computer science; Field (mathematics); Artificial intelligence; Machine learning; Data mining","score_opus":0.5502892581731978,"score_gpt":0.47503419077972964,"score_spread":0.07525506739346821,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4389386408","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18134321,0.00063507305,0.777308,0.0019168056,0.00016510877,0.0012316514,0.0042996733,0.029850356,0.0032501074],"genre_scores_gemma":[0.49688146,0.0002552138,0.47969428,0.0013666741,0.00010385746,0.0007601255,0.015705628,0.0013805901,0.003852285],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9790354,0.0062759197,0.0025814825,0.0054315943,0.005629085,0.0010465705],"domain_scores_gemma":[0.90381,0.06414021,0.007609151,0.012053724,0.010815058,0.0015718148],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013536688,0.0022080936,0.0019629193,0.002065706,0.0014128105,0.0030613933,0.0039165383,0.0025861703,0.0032131027],"category_scores_gemma":[0.10164584,0.0014947476,0.0025013075,0.001669324,0.0021525463,0.008897182,0.0041716825,0.0065281037,0.0018857501],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0020235053,0.0016330323,0.044955622,0.0014941749,0.00025186012,0.0009229907,0.0021186303,0.26462212,0.018730724,0.013402298,0.029704507,0.62014055],"study_design_scores_gemma":[0.00012672279,0.00022957368,0.001966941,0.00008412722,0.00005380559,0.00026921966,0.00029791662,0.9671108,0.012275387,0.0093077505,0.008226271,0.000051501174],"about_ca_topic_score_codex":0.00979409,"about_ca_topic_score_gemma":0.01609866,"teacher_disagreement_score":0.013536688,"about_ca_system_score_codex":0.0026269373,"about_ca_system_score_gemma":0.008229097,"threshold_uncertainty_score":0.07158971},"labels":[],"label_agreement":null},{"id":"W4390051453","doi":"10.1145/3638246","title":"Test Generation Strategies for Building Failure Models and Explaining Spurious Failures","year":2023,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Spurious relationship; Computer science; Test (biology); Machine learning; Surrogate model; Test case; Reliability engineering; Artificial intelligence; Engineering","score_opus":0.13746197668598853,"score_gpt":0.34626461896241467,"score_spread":0.20880264227642614,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390051453","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.02616278,0.00016228696,0.97008634,0.00026004584,0.00001726051,0.000111910915,0.00024824284,0.0021916097,0.0007595545],"genre_scores_gemma":[0.52967227,0.0001812563,0.46688232,0.00027619879,0.00003051978,0.00038982197,0.0013566695,0.0003894409,0.0008215365],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9975466,0.0010613941,0.00016516092,0.00046512485,0.0006240522,0.00013757622],"domain_scores_gemma":[0.98557633,0.010522386,0.0009777109,0.0015159397,0.001222555,0.00018498614],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0029061462,0.0019345657,0.00093775586,0.0021804464,0.0004374303,0.0011406628,0.002534576,0.001631492,0.0018448514],"category_scores_gemma":[0.021620609,0.000809661,0.0015713628,0.00087121205,0.0012031484,0.0017023804,0.0014022406,0.0018030617,0.0005477121],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000135098,0.00014970308,0.006780132,0.00014743292,0.00009617845,0.00035349728,0.00018702824,0.88171,0.0043941187,0.010472181,0.001723122,0.09385153],"study_design_scores_gemma":[0.0000102388485,0.000028172992,0.00018129365,0.000011199006,0.000011831283,0.000050078295,0.000011495719,0.991126,0.001534839,0.0067493706,0.00027950853,0.0000059560966],"about_ca_topic_score_codex":0.0042165327,"about_ca_topic_score_gemma":0.0057320944,"teacher_disagreement_score":0.0042165327,"about_ca_system_score_codex":0.0012402211,"about_ca_system_score_gemma":0.0016939944,"threshold_uncertainty_score":0.015369356},"labels":[],"label_agreement":null},{"id":"W4390838289","doi":"10.1145/3640331","title":"Method-level Bug Prediction: Problems and Promises","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":19,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia; York University; University of Manitoba","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Computer science; Software bug; Java; Class (philosophy); Software; Software regression; Granularity; Predictive modelling; Data science; Data mining; Machine learning; Software engineering; Artificial intelligence; Software development; Software quality; Programming language","score_opus":0.13377957027849616,"score_gpt":0.33783300584779025,"score_spread":0.2040534355692941,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4390838289","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.109705456,0.08647993,0.6149586,0.15281011,0.0034725927,0.00037032086,0.01050763,0.014015221,0.0076800855],"genre_scores_gemma":[0.560373,0.020079495,0.37502876,0.011103251,0.0052519757,0.0006139664,0.01949944,0.0022209717,0.005829174],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.959857,0.015673386,0.001787013,0.010585613,0.011046789,0.0010501917],"domain_scores_gemma":[0.7298616,0.19185795,0.008964104,0.038232043,0.027086576,0.003997786],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.053355675,0.0033034536,0.0036300975,0.005008307,0.0019274217,0.0067918557,0.006931013,0.0048242267,0.0021280614],"category_scores_gemma":[0.17654042,0.0015844858,0.0026470518,0.0069938367,0.0043069473,0.020224754,0.004503762,0.012202567,0.00426308],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005330553,0.0005648086,0.14641513,0.0014599749,0.0006345958,0.00015028534,0.0013514619,0.03893546,0.0018658553,0.012642147,0.0783708,0.71707636],"study_design_scores_gemma":[0.00016373149,0.00076648453,0.07051471,0.0018496772,0.000351188,0.00078442926,0.0026004922,0.6442244,0.005063422,0.18907116,0.084130295,0.00047999332],"about_ca_topic_score_codex":0.02095682,"about_ca_topic_score_gemma":0.012135274,"teacher_disagreement_score":0.053355675,"about_ca_system_score_codex":0.002392705,"about_ca_system_score_gemma":0.0048680203,"threshold_uncertainty_score":0.28217518},"labels":[],"label_agreement":null},{"id":"W4391110914","doi":"10.1145/3641847","title":"Battling against Protocol Fuzzing: Protecting Networked Embedded Devices from Dynamic Fuzzers","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Fuzz testing; Computer science; Protocol (science); Computer security; Obfuscation; Encryption; Code (set theory); Overhead (engineering); Embedded system; Operating system; Programming language; Software; Set (abstract data type)","score_opus":0.050501100712676196,"score_gpt":0.33744814811419427,"score_spread":0.28694704740151805,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391110914","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.33903623,0.0018989316,0.637582,0.0008364419,0.00013515331,0.00036325495,0.00010546389,0.012987208,0.007055262],"genre_scores_gemma":[0.94385046,0.00032432156,0.053917073,0.000285751,0.00002411911,0.00008047781,0.00007683759,0.000151493,0.0012894545],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99761945,0.00055041857,0.00019098938,0.00046922575,0.00084550446,0.0003243879],"domain_scores_gemma":[0.99083215,0.0028912367,0.0014340014,0.0039311294,0.0006934701,0.00021801224],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019590287,0.0009742197,0.0006655517,0.001185387,0.00064559514,0.0013695824,0.0019312422,0.0011597836,0.0011419137],"category_scores_gemma":[0.009971757,0.00042542894,0.0006881327,0.00031193672,0.0016964864,0.004000071,0.002497439,0.0016944209,0.0004018915],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001779491,0.000647067,0.019384332,0.00072495366,0.00034738833,0.0012756385,0.0013491191,0.14778255,0.25826743,0.05169127,0.006107071,0.5106437],"study_design_scores_gemma":[0.00006350717,0.000863725,0.002934565,0.00015382643,0.00016790703,0.00092031615,0.00015063323,0.795432,0.17889561,0.013270441,0.007051953,0.00009553697],"about_ca_topic_score_codex":0.0011994352,"about_ca_topic_score_gemma":0.0012479139,"teacher_disagreement_score":0.0019590287,"about_ca_system_score_codex":0.0008368483,"about_ca_system_score_gemma":0.0011749599,"threshold_uncertainty_score":0.01036042},"labels":[],"label_agreement":null},{"id":"W4391136370","doi":"10.1145/3641541","title":"Learning Failure-Inducing Models for Testing Software-Defined Networks","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software-Defined Networks and 5G","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg; Science Foundation Ireland","keywords":"Computer science; Software testing; Software engineering; Software; Programming language","score_opus":0.08763148271628124,"score_gpt":0.29222353392917,"score_spread":0.20459205121288876,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391136370","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1425781,0.0007630056,0.85102594,0.0006269271,0.000055659395,0.00014881711,0.00063509314,0.0032379837,0.00092846033],"genre_scores_gemma":[0.88481545,0.00019174666,0.112276435,0.00023217572,0.000043024436,0.0001943545,0.0014848331,0.000114767514,0.0006471327],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970394,0.0010847336,0.00023867507,0.0007934445,0.00064871093,0.00019507774],"domain_scores_gemma":[0.9652856,0.028351884,0.0021324398,0.0017499351,0.0019569804,0.0005230164],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003825496,0.0018809008,0.0011160228,0.0018043107,0.00044357774,0.0010164798,0.0025103695,0.0015594653,0.0008975818],"category_scores_gemma":[0.030034311,0.00058917963,0.0010759783,0.0006307512,0.0013677126,0.0020398214,0.0014367843,0.0025498269,0.00023326436],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014303932,0.00013188248,0.0074954666,0.000113250244,0.00006289232,0.00006763939,0.000075834556,0.9506246,0.0012836367,0.0028544243,0.0007259841,0.036421373],"study_design_scores_gemma":[0.00000632751,0.000019596771,0.0001876689,0.000006321973,0.000004424676,0.000009035984,0.0000054841303,0.9957028,0.0005654817,0.0034214421,0.000068218345,0.000003209551],"about_ca_topic_score_codex":0.0071858233,"about_ca_topic_score_gemma":0.0100178495,"teacher_disagreement_score":0.0071858233,"about_ca_system_score_codex":0.002330793,"about_ca_system_score_gemma":0.0015377793,"threshold_uncertainty_score":0.020231426},"labels":[],"label_agreement":null},{"id":"W4391444574","doi":"10.1145/3641543","title":"Beyond Fidelity: Explaining Vulnerability Localization of Learning-Based Detectors","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Fidelity; Vulnerability (computing); Detector; Computer security; Telecommunications","score_opus":0.04447907595058553,"score_gpt":0.31488102608238366,"score_spread":0.27040195013179813,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391444574","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.15438074,0.0009353022,0.83634573,0.0011757443,0.00007136853,0.00012532281,0.00056567526,0.00441859,0.001981409],"genre_scores_gemma":[0.88906056,0.00030418145,0.10842996,0.00026047885,0.000039534414,0.00005189999,0.00085156166,0.0001994009,0.0008024033],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99769336,0.00077479944,0.00014977805,0.0005247498,0.00062960247,0.00022773135],"domain_scores_gemma":[0.97497004,0.016906492,0.0029424543,0.0032802105,0.0015965352,0.00030415333],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0042236885,0.0016721667,0.00089159637,0.003258075,0.00044525828,0.0015854299,0.0014822023,0.001840652,0.0014659746],"category_scores_gemma":[0.032102596,0.00044629409,0.0010156415,0.0013330874,0.0015902164,0.0047733597,0.0023601097,0.0024820263,0.0004319641],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00039099812,0.00020195114,0.056093905,0.00042943232,0.00028311563,0.00054667966,0.00061595114,0.66259027,0.0077697923,0.021464923,0.0046706772,0.24494235],"study_design_scores_gemma":[0.0000138130235,0.000073112096,0.0026332706,0.000039300667,0.000046277037,0.0001752034,0.00007783039,0.9688626,0.0050714216,0.021831643,0.0011492558,0.000026339561],"about_ca_topic_score_codex":0.0038423028,"about_ca_topic_score_gemma":0.0038983577,"teacher_disagreement_score":0.0042236885,"about_ca_system_score_codex":0.0012201953,"about_ca_system_score_gemma":0.0013150522,"threshold_uncertainty_score":0.022337258},"labels":[],"label_agreement":null},{"id":"W4391614448","doi":"10.1145/3643671","title":"Supporting Safety Analysis of Image-processing DNNs through Clustering-based Approaches","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Fonds National de la Recherche Luxembourg; Université du Luxembourg","keywords":"Computer science; Cluster analysis; Artificial intelligence; Data mining; Software engineering","score_opus":0.08917448517718657,"score_gpt":0.349973502449047,"score_spread":0.26079901727186044,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391614448","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.040853646,0.00020910766,0.95260715,0.00020113659,0.000036072153,0.00011703976,0.00022519464,0.004165196,0.0015854648],"genre_scores_gemma":[0.60348123,0.00019368053,0.39307132,0.0002254841,0.000036208916,0.00016472574,0.0007346067,0.00047045795,0.0016222744],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990427,0.00018957374,0.000062704064,0.0003004587,0.00030137014,0.00010320391],"domain_scores_gemma":[0.99639916,0.0014011074,0.00048959325,0.0005678683,0.0010598366,0.00008231231],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019064113,0.0018681741,0.0006622476,0.0025564702,0.0006219557,0.0011625953,0.002233391,0.0013494138,0.0022522886],"category_scores_gemma":[0.0083551975,0.0005545314,0.0009145601,0.00078296184,0.0010819816,0.0017676449,0.00168954,0.0016999292,0.0007485036],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0002765169,0.00012665046,0.0035570257,0.00018447825,0.000116306386,0.00021744978,0.00020253901,0.7646832,0.017227197,0.0059362883,0.0026291271,0.20484321],"study_design_scores_gemma":[0.0000033442489,0.00001629951,0.00031875915,0.000008353158,0.000006918351,0.000019515846,0.000020662972,0.98753786,0.007281841,0.0044894796,0.00029087556,0.0000060675725],"about_ca_topic_score_codex":0.008583513,"about_ca_topic_score_gemma":0.011980025,"teacher_disagreement_score":0.008583513,"about_ca_system_score_codex":0.0021594441,"about_ca_system_score_gemma":0.0015077731,"threshold_uncertainty_score":0.017067134},"labels":[],"label_agreement":null},{"id":"W4391614764","doi":"10.1145/3644388","title":"DeepGD: A Multi-Objective Black-Box Test Selection Approach for Deep Neural Networks","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":27,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Selection (genetic algorithm); Black box; Artificial neural network; Artificial intelligence; Deep neural networks; Machine learning; Test (biology)","score_opus":0.0672593941211773,"score_gpt":0.31481957568626506,"score_spread":0.24756018156508774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4391614764","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.049873948,0.00068242097,0.94371265,0.0004627037,0.000051144514,0.00023231767,0.00023119648,0.0034657202,0.0012878875],"genre_scores_gemma":[0.64526355,0.00018682358,0.34937075,0.0007682182,0.0000613251,0.0005431295,0.0010149236,0.00040206118,0.0023892242],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99776804,0.00093854335,0.00011121885,0.00041667675,0.00052277796,0.0002427777],"domain_scores_gemma":[0.993494,0.004433055,0.0004658075,0.00035852398,0.00096739276,0.00028128232],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0036264842,0.0024101313,0.0018573055,0.00183162,0.00044283015,0.00083255686,0.0030960464,0.0016944687,0.0027045864],"category_scores_gemma":[0.00862855,0.00096080184,0.0010263422,0.00073695404,0.0013239811,0.0015369436,0.0021699932,0.002335218,0.00036490892],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000260118,0.00022038464,0.0037563501,0.00015573484,0.00013656185,0.00019892516,0.00006972935,0.8263527,0.0041572843,0.0033590177,0.002566647,0.1587666],"study_design_scores_gemma":[0.000022434702,0.00006834147,0.0001812957,0.000007921684,0.000009743893,0.000019740393,0.000007552403,0.9961738,0.0012414162,0.00206919,0.00019304828,0.000005542646],"about_ca_topic_score_codex":0.0065109967,"about_ca_topic_score_gemma":0.0091484785,"teacher_disagreement_score":0.0065109967,"about_ca_system_score_codex":0.0020721043,"about_ca_system_score_gemma":0.0026796015,"threshold_uncertainty_score":0.019178867},"labels":[],"label_agreement":null},{"id":"W4392347631","doi":"10.1145/3649598","title":"Communicating Study Design Trade-offs in Software Engineering","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; University of Victoria; McGill University","funders":"Natural Sciences and Engineering Research Council of Canada; Science Foundation Ireland; European Commission; National Science Foundation","keywords":"Computer science; Work (physics); Reflection (computer programming); Process (computing); Risk analysis (engineering); Management science; Strengths and weaknesses; Engineering ethics; Psychology; Business; Engineering; Social psychology","score_opus":0.12728410333757412,"score_gpt":0.3491946088004384,"score_spread":0.22191050546286428,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392347631","genre_codex":"methods","genre_gemma":"methods","domain_codex":"methods","domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":"methods","prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.06548037,0.01105646,0.7223025,0.14250295,0.006421748,0.02700455,0.0003035513,0.0012804379,0.023647549],"genre_scores_gemma":[0.28138644,0.0018049192,0.6168484,0.030882267,0.0023195036,0.06456007,0.00011165973,0.0005075057,0.0015792415],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.03621928,0.8435737,0.07400297,0.009981275,0.034387335,0.001835453],"domain_scores_gemma":[0.014647949,0.8792941,0.030895647,0.054165337,0.018982721,0.00201419],"candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.9067478,0.004290369,0.0048906775,0.013317633,0.012499801,0.026581358,0.010011335,0.024767516,0.005899589],"category_scores_gemma":[0.94183356,0.005997211,0.0053601125,0.008446526,0.048024204,0.04351353,0.034710906,0.026730098,0.0024552343],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0030426676,0.0006847073,0.013890049,0.011341111,0.0021512657,0.0019907306,0.27467874,0.0046074893,0.006706508,0.27313396,0.017447483,0.39032528],"study_design_scores_gemma":[0.0024257845,0.0026931793,0.0074143745,0.02647537,0.0010199089,0.002178523,0.037261423,0.019695371,0.010564144,0.8026391,0.086474776,0.0011580526],"about_ca_topic_score_codex":0.0009162107,"about_ca_topic_score_gemma":0.0015838072,"teacher_disagreement_score":0.09325218,"about_ca_system_score_codex":0.025189694,"about_ca_system_score_gemma":0.025854979,"threshold_uncertainty_score":0.18276489},"labels":[],"label_agreement":null},{"id":"W4392593558","doi":"10.1145/3649597","title":"Lessons Learned from Developing a Sustainability Awareness Framework for Software Engineering Using Design Science","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Green IT and Sustainability","field":"Engineering","cited_by":16,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Vector Institute","funders":"Engineering and Physical Sciences Research Council","keywords":"Sustainability; Computer science; Process (computing); Work (physics); Process management; Engineering management; Sustainable development; Knowledge management; Management science; Risk analysis (engineering); Software engineering; Engineering; Business","score_opus":0.2091774411769321,"score_gpt":0.38913597300434755,"score_spread":0.17995853182741545,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4392593558","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0072767343,0.0032423045,0.93802434,0.033187173,0.00042082847,0.00028820304,0.000051550927,0.0004465092,0.017062292],"genre_scores_gemma":[0.072702035,0.0026002754,0.9200067,0.0017964228,0.00020677948,0.00040486985,0.000075503645,0.00019023805,0.0020172314],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"qualitative","domain_scores_codex":[0.97656894,0.0153689915,0.0015507474,0.0010726261,0.0047007673,0.00073792756],"domain_scores_gemma":[0.9661451,0.023804586,0.0010286684,0.0032582448,0.004259873,0.0015035218],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0367243,0.0019145564,0.0010597521,0.004830909,0.0036107444,0.009830129,0.0030967498,0.004307256,0.0020036022],"category_scores_gemma":[0.021788245,0.0014555064,0.0016352648,0.0023014513,0.015809085,0.014237175,0.008555009,0.010980608,0.00066868257],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000019371437,0.0001756222,0.0015193503,0.0009828461,0.000048829817,0.00039145953,0.011713162,0.011814733,0.0015590985,0.87363195,0.005171686,0.09297191],"study_design_scores_gemma":[0.000031267686,0.00007936716,0.00036684304,0.0010955943,0.00003287841,0.00037050206,0.004389595,0.020925526,0.0016502823,0.82831275,0.14267175,0.00007375235],"about_ca_topic_score_codex":0.0059858714,"about_ca_topic_score_gemma":0.009414399,"teacher_disagreement_score":0.0367243,"about_ca_system_score_codex":0.0054863705,"about_ca_system_score_gemma":0.011009303,"threshold_uncertainty_score":0.194219},"labels":[],"label_agreement":null},{"id":"W4396773582","doi":"10.1145/3664606","title":"Unveiling Code Pre-Trained Models: Investigating Syntax and Semantics Capacities","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":25,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"National Research Foundation Singapore","keywords":"Computer science; Syntax; Abstract syntax tree; Abstract syntax; Programming language; Semantics (computer science); Syntax error; Artificial intelligence; Natural language processing; Code (set theory); Source code","score_opus":0.12010431418201437,"score_gpt":0.3292689790807466,"score_spread":0.20916466489873223,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396773582","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8486233,0.00048764102,0.14188397,0.00091341103,0.000109787004,0.00012777036,0.0008036361,0.0029508234,0.0040997895],"genre_scores_gemma":[0.9653939,0.00013121555,0.0307589,0.00027426207,0.000015532432,0.00012554083,0.0017018141,0.00025732126,0.0013415],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989538,0.0003099672,0.00005898156,0.0003637743,0.0001836192,0.00012978488],"domain_scores_gemma":[0.98874027,0.0073338975,0.000491947,0.0017437435,0.001314523,0.000375536],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002183212,0.0014705306,0.0006558885,0.00078139507,0.00039390047,0.0016733068,0.0018349148,0.0014766719,0.001535535],"category_scores_gemma":[0.021103341,0.0006705432,0.0011003657,0.00056231127,0.001281117,0.0057439497,0.0019641754,0.0041824565,0.0006663609],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000807485,0.00049003697,0.042223092,0.0004541068,0.000353724,0.00044484646,0.0014005534,0.6726412,0.02609305,0.011306734,0.00717723,0.236608],"study_design_scores_gemma":[0.000015829168,0.00009972345,0.0018568499,0.000020053307,0.000041788426,0.00003618636,0.00009231923,0.9869927,0.0057485728,0.004454479,0.00062287407,0.000018634533],"about_ca_topic_score_codex":0.011739079,"about_ca_topic_score_gemma":0.011804596,"teacher_disagreement_score":0.011739079,"about_ca_system_score_codex":0.0017800751,"about_ca_system_score_gemma":0.0018708551,"threshold_uncertainty_score":0.023341477},"labels":[],"label_agreement":null},{"id":"W4396871622","doi":"10.1145/3664599","title":"On the Model Update Strategies for Supervised Learning in AIOps Solutions","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Data Stream Mining Techniques","field":"Computer Science","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; Polytechnique Montréal; Queen's University","funders":"","keywords":"Computer science; Artificial intelligence; Machine learning","score_opus":0.11905621126172586,"score_gpt":0.32602654953751875,"score_spread":0.2069703382757929,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396871622","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.12552096,0.0020747543,0.8663084,0.0014018677,0.00013940941,0.00021883838,0.00017089615,0.0018510037,0.0023139038],"genre_scores_gemma":[0.84075224,0.0005587543,0.15618761,0.00052034564,0.00010622915,0.00018507036,0.0004426936,0.00014719635,0.0010998307],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99718434,0.0012583386,0.00021874635,0.000606283,0.0005494554,0.00018284997],"domain_scores_gemma":[0.9837352,0.010714638,0.0010200045,0.0019698883,0.0022226146,0.0003376557],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0063502155,0.0014074469,0.001235026,0.0010107933,0.0008516922,0.0013672952,0.0025717167,0.0015708348,0.00096419745],"category_scores_gemma":[0.030469956,0.00074875937,0.00077271496,0.000727541,0.0011300343,0.0039911065,0.0018734977,0.0031837318,0.0003952883],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00022408756,0.0003965607,0.006914852,0.00016108422,0.00012763427,0.00009622834,0.00030477665,0.77881885,0.0024035631,0.007342285,0.0024149423,0.20079513],"study_design_scores_gemma":[0.000007874125,0.000048627455,0.00023433047,0.000009485055,0.000009457118,0.000018518293,0.000019928559,0.9963768,0.00065779267,0.0023678842,0.00024429618,0.0000050539647],"about_ca_topic_score_codex":0.010196948,"about_ca_topic_score_gemma":0.010459522,"teacher_disagreement_score":0.010196948,"about_ca_system_score_codex":0.0011696094,"about_ca_system_score_gemma":0.0019659381,"threshold_uncertainty_score":0.03358358},"labels":[],"label_agreement":null},{"id":"W4396871671","doi":"10.1145/3664601","title":"Testing Updated Apps by Adapting Learned Models","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Natural Sciences and Engineering Research Council of Canada; Centre National de la Recherche Scientifique","keywords":"Computer science; Correctness; Adaptation (eye); Software; Software inspection; Software performance testing; Machine learning; Software engineering; Human–computer interaction; Software development; Software quality; Operating system; Software construction; Programming language","score_opus":0.1504107154579445,"score_gpt":0.32533558583462185,"score_spread":0.17492487037667737,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4396871671","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3614489,0.0016958071,0.5704099,0.0007734447,0.0002379842,0.00066036556,0.0015069564,0.057163887,0.006102831],"genre_scores_gemma":[0.772447,0.0004720799,0.21854551,0.0004923003,0.000060064605,0.00042518158,0.0025553375,0.0019062405,0.0030963183],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9936081,0.0016730445,0.00036164344,0.0016257791,0.0023644639,0.00036702715],"domain_scores_gemma":[0.9754047,0.011732799,0.0015686437,0.007786157,0.003073535,0.00043409705],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0028974065,0.0025087127,0.0009318402,0.0014849557,0.00034999187,0.0017243244,0.0043767644,0.0014997227,0.002371946],"category_scores_gemma":[0.038369454,0.0012579897,0.0012146246,0.0007159737,0.0008685481,0.004483138,0.0030872915,0.0026229313,0.0014015343],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010268135,0.0011238393,0.0422955,0.00076854625,0.00033207992,0.00063784997,0.0008145503,0.25262776,0.03093397,0.0025280418,0.008327145,0.65858394],"study_design_scores_gemma":[0.00006108849,0.00031231984,0.0030061281,0.00007535555,0.000106833744,0.00026880548,0.000092469774,0.9750768,0.012505013,0.003851343,0.0045943805,0.00004948427],"about_ca_topic_score_codex":0.007080216,"about_ca_topic_score_gemma":0.011327507,"teacher_disagreement_score":0.007080216,"about_ca_system_score_codex":0.0011082771,"about_ca_system_score_gemma":0.0023426474,"threshold_uncertainty_score":0.015323162},"labels":[],"label_agreement":null},{"id":"W4399601886","doi":"10.1145/3672457","title":"<i>GIST</i> : Generated Inputs Sets Transferability in Deep Learning","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Scratch; Test set; GiST; Testability; Test (biology); Set (abstract data type); Artificial intelligence; Property (philosophy); Artificial neural network; Machine learning; Transfer of learning; Data mining; Reliability engineering; Programming language","score_opus":0.04779516566455851,"score_gpt":0.31123246473109833,"score_spread":0.26343729906653984,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399601886","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.018536162,0.00028916568,0.97440594,0.00048418934,0.00005188402,0.00013780444,0.00012677826,0.0037710594,0.002197132],"genre_scores_gemma":[0.6581245,0.00034580566,0.33666217,0.000649351,0.00009589452,0.00043511417,0.00068726,0.00084698235,0.0021529815],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9949999,0.0019446595,0.00033652526,0.00088217645,0.00153961,0.00029724246],"domain_scores_gemma":[0.98166245,0.010469561,0.001319248,0.005122036,0.00115137,0.00027532608],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005787595,0.0013952391,0.00075191766,0.0012816199,0.00041899396,0.0015507687,0.002919201,0.0016250414,0.003737976],"category_scores_gemma":[0.031027248,0.0005961552,0.0011464789,0.00081888225,0.003451832,0.0036680591,0.0035957561,0.0034771531,0.0007125886],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005541619,0.00026002838,0.0046275104,0.00044643946,0.00017306747,0.0006205254,0.00041416282,0.478938,0.027521197,0.078738965,0.0071542184,0.40055168],"study_design_scores_gemma":[0.000031772393,0.00020059984,0.00045554253,0.00006191641,0.000024456598,0.00016576347,0.0000235881,0.9196472,0.02626254,0.050759688,0.0023449413,0.000021960397],"about_ca_topic_score_codex":0.0015667984,"about_ca_topic_score_gemma":0.0012417329,"teacher_disagreement_score":0.005787595,"about_ca_system_score_codex":0.001588322,"about_ca_system_score_gemma":0.0011343743,"threshold_uncertainty_score":0.030608118},"labels":[],"label_agreement":null},{"id":"W4399619591","doi":"10.1145/3672449","title":"An Empirical Study on the Characteristics of Database Access Bugs in Java Applications","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Database; SQL; Java; Database schema; Stored procedure; Query by Example; View; Commit; Database model; Database design; World Wide Web; Programming language; Web search query","score_opus":0.16223068351507128,"score_gpt":0.4138237030113616,"score_spread":0.2515930194962903,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4399619591","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9978023,0.0002369537,0.0008453349,0.000117253614,0.000006153577,0.000043296008,0.00032578866,0.000036783167,0.00058600865],"genre_scores_gemma":[0.9977277,0.00018102038,0.0011109817,0.000047532616,0.000009224236,0.00004319618,0.00064411253,0.000018976047,0.00021724912],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98892444,0.0026481408,0.0021577864,0.0014540509,0.004128757,0.0006868587],"domain_scores_gemma":[0.669917,0.1965958,0.09217562,0.008647342,0.029428454,0.0032358598],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0061616264,0.0003742577,0.0003194572,0.004793358,0.00062282174,0.0015716726,0.0009658005,0.0009133967,0.00094332354],"category_scores_gemma":[0.10643455,0.0004452152,0.0003747347,0.0045067132,0.0010266257,0.0030110432,0.0010011808,0.0011580465,0.0002466711],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000609809,0.00020502655,0.9826988,0.00016020046,0.00003600805,0.00024600865,0.0024348537,0.00024018611,0.000844613,0.000119679804,0.00030972823,0.012643878],"study_design_scores_gemma":[0.000005861335,0.0001850874,0.9918172,0.00007709398,0.000029603198,0.00070585456,0.0034573274,0.001976007,0.00074181677,0.00012452826,0.0008618966,0.000017785384],"about_ca_topic_score_codex":0.0036966866,"about_ca_topic_score_gemma":0.0069402256,"teacher_disagreement_score":0.0061616264,"about_ca_system_score_codex":0.0008276212,"about_ca_system_score_gemma":0.000984937,"threshold_uncertainty_score":0.032586217},"labels":[],"label_agreement":null},{"id":"W4400480621","doi":"10.1145/3678168","title":"Studying the Impact of TensorFlow and PyTorch Bindings on Machine Learning Software Quality","year":2024,"lang":"en","type":"preprint","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Cloud Computing and Resource Management","field":"Computer Science","cited_by":13,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada); University of Alberta","funders":"University of Alberta","keywords":"Computer science; Artificial intelligence; Quality (philosophy); Deep learning; Software; Machine learning; Programming language; Physics","score_opus":0.10399543823728415,"score_gpt":0.3530575879943977,"score_spread":0.24906214975711355,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400480621","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.8650033,0.0030904762,0.0959753,0.0022071605,0.000601519,0.00023209053,0.0007850982,0.023498973,0.008606068],"genre_scores_gemma":[0.91157466,0.00049222045,0.07709844,0.00061709544,0.00007053553,0.00016657979,0.0013050189,0.00633248,0.0023430379],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9613212,0.011427853,0.0033618324,0.0042759534,0.01600238,0.0036107826],"domain_scores_gemma":[0.7706335,0.15090562,0.013450695,0.042552028,0.01946191,0.002996197],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.027390014,0.0021053506,0.001027574,0.0020357787,0.0015831648,0.0033779559,0.0035620774,0.0019758916,0.00414483],"category_scores_gemma":[0.20164531,0.0018595267,0.0015978614,0.0025013266,0.0030239667,0.00948123,0.0041217464,0.005102948,0.001165938],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0077394135,0.0025388876,0.10876917,0.0025549037,0.001117727,0.0013107798,0.0019794735,0.35031936,0.09182252,0.035147294,0.025844214,0.37085626],"study_design_scores_gemma":[0.0005857896,0.0027583966,0.039732818,0.00048052927,0.00069962157,0.0009911036,0.0011035613,0.7957449,0.12167524,0.018735684,0.017176481,0.0003159456],"about_ca_topic_score_codex":0.010417138,"about_ca_topic_score_gemma":0.0098874755,"teacher_disagreement_score":0.027390014,"about_ca_system_score_codex":0.0029658983,"about_ca_system_score_gemma":0.004231039,"threshold_uncertainty_score":0.14485401},"labels":[],"label_agreement":null},{"id":"W4400652308","doi":"10.1145/3678172","title":"A Disruptive Research Playbook for Studying Disruptive Innovations","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Computer science; Disruptive technology; Transformative learning; Framing (construction); Disruptive innovation; Emerging technologies; Software; Generative grammar; Data science; Empirical research; Management science; Artificial intelligence; Sociology; Engineering","score_opus":0.25035296368953364,"score_gpt":0.4320521837769804,"score_spread":0.18169922008744677,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400652308","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.032010578,0.020807981,0.5353144,0.06540525,0.008279945,0.0055271643,0.0040495098,0.0029728964,0.32563233],"genre_scores_gemma":[0.223963,0.033452153,0.559148,0.011059996,0.0025861235,0.014850823,0.0026357123,0.0013393711,0.15096486],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99417514,0.004287645,0.00023527641,0.00041565823,0.00062762195,0.00025872953],"domain_scores_gemma":[0.9786702,0.01689458,0.00064651197,0.0016307962,0.001206871,0.0009508913],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.007076492,0.0029967115,0.001107724,0.0051207766,0.0058897953,0.008435646,0.0030388837,0.004372289,0.017917126],"category_scores_gemma":[0.014425104,0.0006947469,0.0010226439,0.004755674,0.012185089,0.012465757,0.0064028017,0.007711138,0.0040725255],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000111675785,0.0002717613,0.0008803118,0.0015082475,0.00001705177,0.00092363555,0.07051655,0.001693651,0.0029445544,0.70100474,0.09927658,0.12085127],"study_design_scores_gemma":[0.000040918403,0.00025790883,0.0008660795,0.0018784575,0.000013016168,0.0007463651,0.04019135,0.0019473154,0.0012457025,0.10310239,0.8496388,0.00007161091],"about_ca_topic_score_codex":0.0056974245,"about_ca_topic_score_gemma":0.010940088,"teacher_disagreement_score":0.9929235,"about_ca_system_score_codex":0.00559663,"about_ca_system_score_gemma":0.0063396064,"threshold_uncertainty_score":0.05993879},"labels":[],"label_agreement":null},{"id":"W4400685466","doi":"10.1145/3676961","title":"SimClone: Detecting Tabular Data Clones Using Value Similarity","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Data Quality and Management","field":"Decision Sciences","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; Huawei Technologies (Canada); University of Manitoba","funders":"","keywords":"Computer science; Data mining; Header; Software; Visualization; clone (Java method); Similarity (geometry); Margin (machine learning); Artificial intelligence; Image (mathematics); Machine learning","score_opus":0.549696504884526,"score_gpt":0.4849508565192975,"score_spread":0.06474564836522856,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400685466","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3299357,0.003010132,0.5207184,0.0008367124,0.00040275865,0.00090408133,0.022390163,0.11718436,0.004617651],"genre_scores_gemma":[0.51656324,0.0004744499,0.4413514,0.00047263128,0.00012564083,0.0006241468,0.03522575,0.0025737733,0.0025889562],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9939301,0.0009319231,0.0008229218,0.0016868454,0.0023824333,0.0002458499],"domain_scores_gemma":[0.9758462,0.009042843,0.004354957,0.0050080908,0.0052155126,0.0005323804],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0043531405,0.0017555577,0.0015723298,0.008597459,0.0011067379,0.0029619262,0.0023047633,0.0018765712,0.00181507],"category_scores_gemma":[0.030685188,0.0005044657,0.001615589,0.0069058393,0.00072235777,0.004066015,0.0024766217,0.0012936408,0.0022421074],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0012138094,0.0007249763,0.22193466,0.001390245,0.000863576,0.0011566597,0.001816645,0.017426169,0.048588645,0.005800591,0.061494347,0.6375897],"study_design_scores_gemma":[0.00017664158,0.00078537746,0.05326982,0.00023602882,0.00025288435,0.002438635,0.0010891164,0.77217156,0.114143975,0.015497264,0.039686743,0.00025190492],"about_ca_topic_score_codex":0.0029698575,"about_ca_topic_score_gemma":0.004724973,"teacher_disagreement_score":0.008597459,"about_ca_system_score_codex":0.00083599874,"about_ca_system_score_gemma":0.00158952,"threshold_uncertainty_score":0.023021877},"labels":[],"label_agreement":null},{"id":"W4400934506","doi":"10.1145/3680463","title":"An Empirical Study of Testing Machine Learning in the Wild","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; York University; Polytechnique Montréal","funders":"","keywords":"Computer science; White-box testing; Regression testing; Oracle; Test strategy; Workflow; Quality assurance; Machine learning; Software reliability testing; Manual testing; Software quality; Empirical research; Non-regression testing; Software performance testing; Software engineering; Artificial intelligence; Software testing; Software; Software system; Software construction; Software development; Database; Programming language","score_opus":0.1074213531816717,"score_gpt":0.370455849240389,"score_spread":0.2630344960587173,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4400934506","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98773164,0.000347329,0.008864909,0.00063303404,0.000028208982,0.00019473981,0.00045444138,0.000112810645,0.0016329238],"genre_scores_gemma":[0.99045753,0.00009381988,0.007901549,0.0002529354,0.00002785567,0.00023731326,0.00066429446,0.0000744976,0.00029013655],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9191273,0.054223515,0.005041789,0.0072514187,0.012611536,0.0017444863],"domain_scores_gemma":[0.34449747,0.55799323,0.04166053,0.029754164,0.022524476,0.0035701802],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.04144661,0.00062401965,0.00063764903,0.0037640906,0.0013124808,0.0029287892,0.0030523066,0.0020282539,0.0017497237],"category_scores_gemma":[0.2779034,0.0005819688,0.0006022694,0.0032874031,0.0062811906,0.0060033565,0.0024106908,0.0026931122,0.00050650025],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00093102374,0.002655725,0.8257572,0.0009823916,0.00035110308,0.0012856523,0.025232852,0.006494014,0.0030069454,0.007651851,0.0062814564,0.11936978],"study_design_scores_gemma":[0.00042613223,0.0051100934,0.72774965,0.0019010243,0.00034291108,0.0037008023,0.046354137,0.13708596,0.014851499,0.025367646,0.03673925,0.0003708078],"about_ca_topic_score_codex":0.0027802798,"about_ca_topic_score_gemma":0.0035282096,"teacher_disagreement_score":0.9585534,"about_ca_system_score_codex":0.002113375,"about_ca_system_score_gemma":0.0013470409,"threshold_uncertainty_score":0.21919328},"labels":[],"label_agreement":null},{"id":"W4401021697","doi":"10.1145/3680470","title":"Enhancing Energy-Awareness in Deep Learning through Fine-Grained Energy Measurement","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Green IT and Sustainability","field":"Engineering","cited_by":10,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Energy consumption; Granularity; Energy accounting; Energy (signal processing); Efficient energy use; Instrumentation (computer programming); Deep learning; Work (physics); Artificial intelligence; Operating system; Mechanical engineering; Electrical engineering; Engineering","score_opus":0.05270298121953904,"score_gpt":0.2806003900837206,"score_spread":0.22789740886418158,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401021697","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.20885757,0.000607772,0.776474,0.00080648187,0.0000834802,0.00008158309,0.0005404809,0.0078155175,0.004733078],"genre_scores_gemma":[0.889053,0.00018101338,0.108955584,0.00021492285,0.000018480523,0.000054345357,0.00040296747,0.00022148906,0.0008983585],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99948275,0.000116963616,0.000027936321,0.00014962144,0.00014840243,0.000074338095],"domain_scores_gemma":[0.998541,0.0006762506,0.00017561877,0.00029559556,0.00024069434,0.00007094206],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011378694,0.0010014013,0.0004963694,0.0006366689,0.0003011548,0.0011784042,0.0011766421,0.000487485,0.0012988026],"category_scores_gemma":[0.005522923,0.00035216016,0.00033332102,0.0006518271,0.0005659385,0.0036533717,0.0012752187,0.0015112169,0.0002656195],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029579192,0.00051698845,0.015493415,0.00023555412,0.00013302399,0.00008308852,0.00021672405,0.71267986,0.021591999,0.008616446,0.0038227937,0.23631433],"study_design_scores_gemma":[0.000006709028,0.000036684116,0.0008320543,0.000009851284,0.000010198543,0.000009115171,0.000020794356,0.98599374,0.006145233,0.0060387966,0.000886681,0.000010091403],"about_ca_topic_score_codex":0.006267202,"about_ca_topic_score_gemma":0.013912437,"teacher_disagreement_score":0.006267202,"about_ca_system_score_codex":0.001027453,"about_ca_system_score_gemma":0.0010360768,"threshold_uncertainty_score":0.012461424},"labels":[],"label_agreement":null},{"id":"W4401047823","doi":"10.1145/3680468","title":"Reinforcement Learning Informed Evolutionary Search for Autonomous Systems Testing","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Reinforcement learning; Computer science; Artificial intelligence; Machine learning; Domain (mathematical analysis); Evolutionary algorithm; Population; Evolutionary robotics; Domain knowledge","score_opus":0.12685296961192286,"score_gpt":0.3405972594143513,"score_spread":0.21374428980242846,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401047823","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09699502,0.00054953404,0.89646214,0.00040467695,0.00004569467,0.00011058748,0.000039608585,0.0009868313,0.004405802],"genre_scores_gemma":[0.8828777,0.00013492454,0.115334,0.00012891286,0.000016634593,0.00015879645,0.00006120846,0.00006734762,0.0012204107],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999171,0.00043877316,0.000030229214,0.0000804874,0.00020471937,0.00007472796],"domain_scores_gemma":[0.99744344,0.0018932886,0.0001888534,0.00015357393,0.0002379085,0.00008300968],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0016370455,0.00071681075,0.0006920752,0.00066859374,0.00024220074,0.0004855109,0.0011193671,0.0008395492,0.0012917006],"category_scores_gemma":[0.0065676924,0.00035260696,0.0004245105,0.0003471333,0.0009840295,0.0006239649,0.000786807,0.0011269045,0.00017473234],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000021561085,0.000034111323,0.000538301,0.000021118101,0.000016288637,0.00003410996,0.000025314088,0.9785776,0.0007393286,0.0029572372,0.00016680582,0.01686815],"study_design_scores_gemma":[0.000007820371,0.000019071198,0.000061387225,0.00000340906,0.000002398907,0.000006319471,0.0000033798476,0.99793005,0.000213886,0.001605495,0.000145173,0.0000016169246],"about_ca_topic_score_codex":0.0030523369,"about_ca_topic_score_gemma":0.0025771346,"teacher_disagreement_score":0.0030523369,"about_ca_system_score_codex":0.00081350806,"about_ca_system_score_gemma":0.0010919076,"threshold_uncertainty_score":0.008657634},"labels":[],"label_agreement":null},{"id":"W4401123839","doi":"10.1145/3680472","title":"MULTICR: Predicting Merged and Abandoned Code Changes in Modern Code Review Using Multi-Objective Search","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Code review; Code (set theory); Machine learning; Eclipse; Process (computing); Artificial intelligence; Software quality; Interoperability; Genetic programming; Search-based software engineering; Data mining; Software engineering; Software; Software development; Software development process; Programming language; World Wide Web","score_opus":0.15337003444371033,"score_gpt":0.38079161845485976,"score_spread":0.22742158401114942,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401123839","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7379465,0.0044424627,0.24626245,0.0012534689,0.00016908845,0.00069640455,0.0016541074,0.0046809195,0.002894548],"genre_scores_gemma":[0.85702187,0.00048803622,0.1371533,0.00032063632,0.00007277452,0.00026496744,0.0024729182,0.00015444114,0.0020511039],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99744713,0.0008908871,0.00024423184,0.00069322495,0.00055141357,0.00017314203],"domain_scores_gemma":[0.9885583,0.0075792097,0.0015108705,0.00044560884,0.0015469706,0.00035900425],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005171971,0.0015234766,0.0013893448,0.004711947,0.0004383468,0.0013590708,0.0018752705,0.0016202342,0.0011421992],"category_scores_gemma":[0.014349556,0.0004944665,0.0011368623,0.0024626448,0.0004267539,0.0013367286,0.0011587875,0.0011103225,0.00037647886],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00064254616,0.0009202978,0.1053193,0.0012504554,0.00074710196,0.0006024982,0.0006521185,0.47775835,0.0045683547,0.0019979945,0.008971931,0.39656898],"study_design_scores_gemma":[0.000031263404,0.00020441962,0.0061316905,0.000046130084,0.000060869817,0.000070824484,0.00007928223,0.99080795,0.0010672525,0.0008202875,0.00065950595,0.000020596268],"about_ca_topic_score_codex":0.011535248,"about_ca_topic_score_gemma":0.018510027,"teacher_disagreement_score":0.011535248,"about_ca_system_score_codex":0.0012509265,"about_ca_system_score_gemma":0.0023283206,"threshold_uncertainty_score":0.027352333},"labels":[],"label_agreement":null},{"id":"W4401635770","doi":"10.1145/3688841","title":"An Exploratory Study on Machine Learning Model Management","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary; Concordia University","funders":"","keywords":"Documentation; Computer science; Software versioning; Software engineering; Automation; Code refactoring; Knowledge management; Data science; Process management; Software; Engineering","score_opus":0.10743960691064752,"score_gpt":0.35192113569131234,"score_spread":0.24448152878066481,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401635770","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9524736,0.0012133108,0.023428855,0.005712843,0.000055242497,0.00065472495,0.0012213561,0.000389176,0.014850789],"genre_scores_gemma":[0.96296406,0.00091867597,0.028738348,0.00090411067,0.00004541654,0.00052831025,0.0020722277,0.00019329003,0.0036355658],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9833264,0.009811507,0.00083204714,0.0013252577,0.0037607295,0.00094407826],"domain_scores_gemma":[0.8260937,0.14105868,0.007133661,0.009813312,0.012587848,0.0033127812],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.022900263,0.00045578228,0.0004935883,0.0024253768,0.0021581273,0.0045162383,0.0024721152,0.001486306,0.0036587443],"category_scores_gemma":[0.13001609,0.00039861814,0.00056516955,0.0044719893,0.0016081701,0.008988138,0.0023643742,0.002994867,0.001026767],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004984131,0.0051103183,0.3319247,0.0021448836,0.00013329848,0.0017470513,0.14378522,0.008932286,0.0034727387,0.047390744,0.04158285,0.41327754],"study_design_scores_gemma":[0.00017783907,0.0021294204,0.18759534,0.0021756636,0.00014493361,0.0025243086,0.19576456,0.15414676,0.007207289,0.048059158,0.39976934,0.0003053314],"about_ca_topic_score_codex":0.0046936385,"about_ca_topic_score_gemma":0.007968163,"teacher_disagreement_score":0.022900263,"about_ca_system_score_codex":0.0030871509,"about_ca_system_score_gemma":0.003634916,"threshold_uncertainty_score":0.121109664},"labels":[],"label_agreement":null},{"id":"W4401635890","doi":"10.1145/3688838","title":"History-Driven Fuzzing for Deep Learning Libraries","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Fuzz testing; Computer science; Heuristic; Artificial intelligence; Set (abstract data type); Machine learning; Natural language processing; Programming language; Software","score_opus":0.06773932218144975,"score_gpt":0.29855303110614345,"score_spread":0.23081370892469372,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401635890","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26617515,0.001482722,0.70717806,0.0011564462,0.00009203577,0.000371016,0.0013271321,0.019027524,0.0031899002],"genre_scores_gemma":[0.8413182,0.00022528347,0.15491876,0.00045153973,0.00002337388,0.00023795902,0.0012339938,0.00035285376,0.001237992],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.997214,0.0006404426,0.00025286898,0.0007310903,0.0008390958,0.00032244285],"domain_scores_gemma":[0.99003243,0.0065439506,0.00094913796,0.0013438134,0.00092983147,0.00020078816],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00307524,0.0015585833,0.0008434035,0.0023724153,0.0006095274,0.0013217838,0.0027268012,0.0011685041,0.0021691541],"category_scores_gemma":[0.016925247,0.0009518524,0.0018011567,0.00073524413,0.0019246296,0.0030283346,0.002092718,0.0019173824,0.00031941],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0004964342,0.00024465335,0.029519891,0.00049875467,0.0002455995,0.0004972302,0.0003298028,0.6994122,0.012864845,0.013675925,0.003121489,0.23909318],"study_design_scores_gemma":[0.000022023563,0.000052757972,0.0007615606,0.000038436745,0.000030460387,0.000051409774,0.000020263475,0.97757566,0.007284441,0.013484817,0.0006636434,0.000014445913],"about_ca_topic_score_codex":0.010153112,"about_ca_topic_score_gemma":0.015798865,"teacher_disagreement_score":0.010153112,"about_ca_system_score_codex":0.0031635494,"about_ca_system_score_gemma":0.0030977775,"threshold_uncertainty_score":0.022953272},"labels":[],"label_agreement":null},{"id":"W4401724691","doi":"10.1145/3688842","title":"My Fuzzers Won’t Build: An Empirical Study of Fuzzing Build Failures","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Fuzz testing; Computer science; Software engineering; Software bug; Context (archaeology); Software; Set (abstract data type); Operating system; Programming language","score_opus":0.09234755339412065,"score_gpt":0.36974165303234796,"score_spread":0.2773940996382273,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4401724691","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99750334,0.00011669926,0.0011615547,0.00028992436,0.000004360011,0.000042680156,0.0002753023,0.000037787253,0.00056843983],"genre_scores_gemma":[0.9972646,0.00010504862,0.0015830352,0.00012333883,0.000008849613,0.00005238776,0.00059053034,0.00002726416,0.0002450115],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9938351,0.0024251395,0.0006752897,0.0008147748,0.0017778564,0.00047184378],"domain_scores_gemma":[0.76007736,0.1809156,0.031525474,0.010810126,0.013503847,0.0031676819],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.011479079,0.0005325733,0.0004165731,0.0033635243,0.0012856189,0.0014842574,0.0016814588,0.0015523524,0.0013397359],"category_scores_gemma":[0.09841094,0.0005941055,0.00046864024,0.0027228585,0.0019969721,0.0044907886,0.0014015264,0.002985418,0.00066038355],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00029464392,0.001314647,0.947741,0.000187373,0.00010217747,0.0006523363,0.016649725,0.0031228059,0.0010516769,0.0014287434,0.003032471,0.02442245],"study_design_scores_gemma":[0.00006832298,0.0011390966,0.9128461,0.0003054225,0.00009765428,0.0017544753,0.026279146,0.046031892,0.0019692,0.0030164192,0.006371902,0.00012040791],"about_ca_topic_score_codex":0.008763247,"about_ca_topic_score_gemma":0.011497354,"teacher_disagreement_score":0.9885209,"about_ca_system_score_codex":0.0010457839,"about_ca_system_score_gemma":0.001102296,"threshold_uncertainty_score":0.060707927},"labels":[],"label_agreement":null},{"id":"W4402048137","doi":"10.1145/3690631","title":"T-Rec: Fine-Grained Language-Agnostic Program Reduction Guided by Lexical Syntax","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Syntax; Programming language; Reduction (mathematics); Natural language processing; Abstract syntax tree; Artificial intelligence","score_opus":0.0649629231789791,"score_gpt":0.3495486032377718,"score_spread":0.2845856800587927,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402048137","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050681308,0.00077016297,0.84098715,0.00055515964,0.00022892162,0.000576186,0.0010305545,0.09902516,0.0061454065],"genre_scores_gemma":[0.26547408,0.00050951407,0.7040958,0.0009530962,0.00009859232,0.0008686104,0.0047839363,0.014503935,0.008712461],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9972676,0.0005976339,0.00024979797,0.00058987667,0.00097317965,0.00032193292],"domain_scores_gemma":[0.99544275,0.0012029749,0.00041662186,0.0021046132,0.0007108597,0.00012226195],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00153433,0.0019264423,0.0008665344,0.0016384668,0.00072732853,0.0015043512,0.0034991216,0.0011001214,0.0038635582],"category_scores_gemma":[0.0061787255,0.0007044721,0.002363468,0.0010793338,0.0021271778,0.0034243881,0.0030798921,0.0027019985,0.0020663117],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009564023,0.0008962805,0.011975479,0.0026294168,0.00043392184,0.0009013906,0.0012340929,0.048906077,0.16017139,0.08112842,0.060101997,0.6306651],"study_design_scores_gemma":[0.0005290227,0.0013969972,0.0048921686,0.00033536804,0.00060955225,0.0017888488,0.00056896766,0.48783326,0.25283775,0.0998551,0.14897357,0.0003793331],"about_ca_topic_score_codex":0.0032149984,"about_ca_topic_score_gemma":0.005898043,"teacher_disagreement_score":0.0038635582,"about_ca_system_score_codex":0.00088142516,"about_ca_system_score_gemma":0.0039102244,"threshold_uncertainty_score":0.01292485},"labels":[],"label_agreement":null},{"id":"W4402215615","doi":"10.1145/3691628","title":"A Large-Scale Study of IoT Security Weaknesses and Vulnerabilities in the Wild","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Blockchain Technology Applications and Security","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Computer science; Internet of Things; Scale (ratio); Strengths and weaknesses; Computer security; Data science; Cartography","score_opus":0.03387503586808571,"score_gpt":0.2963998230558209,"score_spread":0.2625247871877352,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402215615","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99544156,0.0002508085,0.0019735668,0.0002466141,0.000010527205,0.0001017683,0.0005225112,0.00010027168,0.0013523407],"genre_scores_gemma":[0.98861665,0.0005791405,0.007348824,0.00028295192,0.000022105134,0.0002027519,0.0017417165,0.00014090551,0.0010649544],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.99515367,0.0012987934,0.00032757784,0.0006854252,0.0023108274,0.0002238045],"domain_scores_gemma":[0.9435351,0.035878107,0.008258276,0.0038823877,0.007412193,0.0010339259],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0040733516,0.00040995987,0.00028458278,0.004678966,0.0012707604,0.00095282384,0.0008178265,0.00079189934,0.00070979993],"category_scores_gemma":[0.030076703,0.0003383297,0.00043117255,0.0038899397,0.0019121624,0.0035777758,0.0017897913,0.0013315425,0.00033561673],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037197804,0.001739471,0.71772665,0.0014393997,0.00036970084,0.0043706074,0.049577616,0.004044727,0.015114959,0.0035610097,0.015352254,0.18633173],"study_design_scores_gemma":[0.000031214102,0.0007728191,0.9020979,0.0007681773,0.00019541325,0.003581818,0.037193697,0.017311193,0.008938585,0.0019727007,0.026945233,0.00019134652],"about_ca_topic_score_codex":0.0038083673,"about_ca_topic_score_gemma":0.007673753,"teacher_disagreement_score":0.004678966,"about_ca_system_score_codex":0.0007839077,"about_ca_system_score_gemma":0.0010170286,"threshold_uncertainty_score":0.021542192},"labels":[],"label_agreement":null},{"id":"W4402216867","doi":"10.1145/3691627","title":"Reputation Gaming in Crowd Technical Knowledge Sharing","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Mobile Crowdsensing and Crowdsourcing","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University; Polytechnique Montréal","funders":"","keywords":"Computer science; Reputation; Knowledge management","score_opus":0.06536044955787561,"score_gpt":0.3277611355513543,"score_spread":0.26240068599347866,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402216867","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.80034935,0.0011903233,0.16719493,0.0017681613,0.0001289455,0.0005582153,0.0002893388,0.0014191995,0.027101506],"genre_scores_gemma":[0.9886901,0.00007393909,0.010075239,0.00007517755,0.000019122808,0.00004682737,0.00003961647,0.00002195927,0.0009580774],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.9926742,0.0037606508,0.00031055175,0.0011560901,0.0014071984,0.00069134025],"domain_scores_gemma":[0.97716796,0.013849832,0.0034850033,0.0028403269,0.0015038439,0.0011530448],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008781587,0.000644946,0.0007404192,0.0032679944,0.0032148995,0.0036336828,0.0017224115,0.0017472776,0.0020625659],"category_scores_gemma":[0.030811688,0.0005829336,0.00072539,0.0016264096,0.0033182194,0.0042085997,0.0044227033,0.0009901272,0.00038488457],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001404866,0.0007463667,0.2453476,0.0011231486,0.0006229731,0.0041780407,0.028135857,0.15890498,0.015113111,0.13741483,0.009129082,0.39787915],"study_design_scores_gemma":[0.0001435593,0.0005033198,0.07746805,0.00036230226,0.00029092625,0.0022564842,0.012939072,0.66184664,0.01131287,0.19959018,0.032795135,0.0004914565],"about_ca_topic_score_codex":0.008016945,"about_ca_topic_score_gemma":0.0058541135,"teacher_disagreement_score":0.008781587,"about_ca_system_score_codex":0.0022117305,"about_ca_system_score_gemma":0.0019070621,"threshold_uncertainty_score":0.046441972},"labels":[],"label_agreement":null},{"id":"W4402483878","doi":"10.1145/3696002","title":"A Novel Refactoring and Semantic Aware Abstract Syntax Tree Differencing Tool and a Benchmark for Evaluating the Accuracy of Diff Tools","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":23,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Code refactoring; Computer science; Benchmark (surveying); Abstract syntax tree; Abstract syntax; Programming language; Syntax; Software engineering; Tree (set theory); Semantics (computer science); Artificial intelligence; Software","score_opus":0.18087197980816785,"score_gpt":0.3856275967855456,"score_spread":0.20475561697737774,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402483878","genre_codex":"empirical","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.39467457,0.005151602,0.36939597,0.00047049974,0.00057920825,0.0007837068,0.015764933,0.20786849,0.005311022],"genre_scores_gemma":[0.43415183,0.0006821627,0.51726115,0.00029977984,0.00007769156,0.0005980889,0.040732354,0.0034124015,0.0027845378],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9883265,0.0020626402,0.0021482331,0.0027654388,0.004214083,0.00048300647],"domain_scores_gemma":[0.96712923,0.014577638,0.003172623,0.0058631995,0.008528554,0.0007288902],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005836069,0.0025058198,0.0011613932,0.011168179,0.00071018527,0.0015548762,0.002982541,0.0022579846,0.0012778747],"category_scores_gemma":[0.037841298,0.0006724041,0.0011552662,0.005152533,0.0007406771,0.0033756099,0.002385216,0.0012322888,0.0012596853],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010166198,0.00071654044,0.0757986,0.0022120052,0.00044689298,0.0012675058,0.001148724,0.029288042,0.041278,0.0041210605,0.056444973,0.786261],"study_design_scores_gemma":[0.0006858154,0.0017298457,0.06468144,0.00054450077,0.00039491965,0.0040224376,0.0010063279,0.6727848,0.17367646,0.007020215,0.07286976,0.00058350037],"about_ca_topic_score_codex":0.006514304,"about_ca_topic_score_gemma":0.0069300844,"teacher_disagreement_score":0.011168179,"about_ca_system_score_codex":0.0010040193,"about_ca_system_score_gemma":0.0022347313,"threshold_uncertainty_score":0.030864477},"labels":[],"label_agreement":null},{"id":"W4402516053","doi":"10.1145/3695989","title":"Non-Flaky and Nearly Optimal Time-Based Treatment of Asynchronous Wait Web Tests","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Ubisoft (Canada)","funders":"","keywords":"Computer science; Asynchronous communication; Web application; World Wide Web; Computer network","score_opus":0.04587187782534556,"score_gpt":0.3016080723484287,"score_spread":0.2557361945230831,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402516053","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7590974,0.0017751998,0.22248016,0.0009440966,0.00018294896,0.00025796547,0.0036932803,0.00799569,0.0035731664],"genre_scores_gemma":[0.9498145,0.00014399066,0.044590868,0.00017941714,0.00009049286,0.0001879933,0.003395208,0.00052613346,0.0010714607],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9917149,0.0019089959,0.0009827043,0.002407829,0.0021404242,0.0008452082],"domain_scores_gemma":[0.9581916,0.022816142,0.0061232555,0.0074166195,0.004152961,0.0012993781],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004589241,0.0008567235,0.0008538467,0.0026489557,0.0007763374,0.0018704757,0.0018115428,0.001466494,0.001800057],"category_scores_gemma":[0.050830305,0.0004366964,0.00090329914,0.0019588338,0.0010650404,0.0021139183,0.0013442124,0.0015780524,0.00061208685],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003311094,0.000986592,0.2859802,0.0011563099,0.00034197408,0.0018613106,0.002250596,0.19166388,0.04629816,0.01696142,0.020213595,0.42897484],"study_design_scores_gemma":[0.00024522634,0.0008158573,0.077810444,0.00012273664,0.00018334614,0.0013091922,0.00086850126,0.8513567,0.022857375,0.0319124,0.0123982215,0.00012006645],"about_ca_topic_score_codex":0.0032956146,"about_ca_topic_score_gemma":0.0043923073,"teacher_disagreement_score":0.004589241,"about_ca_system_score_codex":0.0011106811,"about_ca_system_score_gemma":0.0021523547,"threshold_uncertainty_score":0.024270535},"labels":[],"label_agreement":null},{"id":"W4402516515","doi":"10.1145/3695995","title":"Deep API Sequence Generation via Golden Solution Samples and API Seeds","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Web Data Mining and Analysis","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"National Natural Science Foundation of China","keywords":"Computer science; Sequence (biology); Programming language; Biology","score_opus":0.13558136938960888,"score_gpt":0.31603201239977097,"score_spread":0.1804506430101621,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402516515","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24227068,0.002272755,0.71006984,0.00091445877,0.0003032501,0.000604444,0.001916925,0.035627127,0.006020565],"genre_scores_gemma":[0.4911722,0.00036541553,0.4928754,0.00050815026,0.000076800105,0.000480016,0.008389559,0.0013878595,0.0047445893],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99863476,0.00028572488,0.00009391275,0.00044388094,0.0003878509,0.00015384566],"domain_scores_gemma":[0.99681133,0.0013924259,0.00021528472,0.0006619907,0.0007361301,0.00018273381],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0013490642,0.0018349148,0.0014316583,0.002234122,0.0006996028,0.0010855767,0.0022121114,0.0015524811,0.0038875903],"category_scores_gemma":[0.009938694,0.0007640226,0.0014314232,0.0014482387,0.0007268334,0.0024087045,0.0013148254,0.0015549192,0.0022422397],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009223768,0.00092989614,0.015652258,0.0007312198,0.00017517856,0.0007519417,0.0003880062,0.15945417,0.026191311,0.008510452,0.037806846,0.74848646],"study_design_scores_gemma":[0.00010445218,0.00018489138,0.00077514973,0.000025879332,0.000057681602,0.00017152216,0.00010007748,0.9814388,0.008418509,0.0051447446,0.0035576115,0.000020578584],"about_ca_topic_score_codex":0.008114816,"about_ca_topic_score_gemma":0.01682202,"teacher_disagreement_score":0.008114816,"about_ca_system_score_codex":0.0008620215,"about_ca_system_score_gemma":0.0023316336,"threshold_uncertainty_score":0.016135156},"labels":[],"label_agreement":null},{"id":"W4402860132","doi":"10.1145/3697014","title":"ZigZagFuzz: Interleaved Fuzzing of Program Options and Files","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Nexen (Canada)","funders":"Samsung; Hanyang University","keywords":"Fuzz testing; Computer science; Programming language; Software","score_opus":0.08854984899731846,"score_gpt":0.34345537315099234,"score_spread":0.25490552415367385,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4402860132","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.26438805,0.0009776702,0.70005983,0.00032568295,0.0000812728,0.00038461722,0.0005641338,0.030590376,0.0026283367],"genre_scores_gemma":[0.68955743,0.00017512668,0.30538964,0.00038133867,0.00001783679,0.00019902145,0.0009875728,0.0009192528,0.0023727408],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.998381,0.00029576643,0.00013766583,0.0005141038,0.0005061843,0.00016521537],"domain_scores_gemma":[0.99599457,0.0018807907,0.00037908327,0.0013020472,0.00033518136,0.00010833979],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0014136719,0.0013071863,0.0006002002,0.0013288921,0.00042025844,0.0008907681,0.0022658221,0.0009941766,0.0019075369],"category_scores_gemma":[0.0060335933,0.00057669147,0.0010449721,0.0005789679,0.0014645681,0.0021382775,0.001315637,0.0010559473,0.00037847494],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0024681375,0.0003349831,0.022966437,0.0007492111,0.00032298654,0.0009766446,0.0007739832,0.10534372,0.2134185,0.016409757,0.0056126066,0.63062304],"study_design_scores_gemma":[0.00027035215,0.0009978644,0.006815648,0.00012714109,0.0002972605,0.001186538,0.00018272578,0.68393433,0.2715059,0.022285482,0.012203938,0.00019278459],"about_ca_topic_score_codex":0.004794141,"about_ca_topic_score_gemma":0.0058460324,"teacher_disagreement_score":0.004794141,"about_ca_system_score_codex":0.00082518207,"about_ca_system_score_gemma":0.0010557193,"threshold_uncertainty_score":0.009532511},"labels":[],"label_agreement":null},{"id":"W4403604523","doi":"10.1145/3672448","title":"Understanding Test Convention Consistency as a Dimension of Test Quality","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Test (biology); Consistency (knowledge bases); Dimension (graph theory); Quality (philosophy); Reliability engineering; Artificial intelligence; Mathematics; Engineering","score_opus":0.22788467896783968,"score_gpt":0.37761842431744,"score_spread":0.14973374534960035,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4403604523","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5015908,0.0015886242,0.4844935,0.0030794314,0.0000860504,0.00021323816,0.00044403205,0.0012128347,0.0072915936],"genre_scores_gemma":[0.90789664,0.00019569238,0.09062272,0.00024375385,0.000060348382,0.00013575047,0.00038326127,0.00018619838,0.00027555393],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.95267165,0.019985873,0.005214783,0.004211188,0.016191337,0.0017252881],"domain_scores_gemma":[0.61564565,0.27121887,0.039628122,0.03777708,0.032905385,0.0028248471],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.03141218,0.0009803295,0.0010024451,0.0069274693,0.00086744194,0.0070457985,0.0018721287,0.0016834674,0.0009837581],"category_scores_gemma":[0.23781149,0.0007444669,0.0011029274,0.0054456014,0.0046758913,0.013153047,0.0033192888,0.0035672027,0.0001471649],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005151683,0.00044878488,0.5132155,0.000671176,0.0004798431,0.0004797493,0.008555736,0.05938972,0.017570622,0.14265892,0.0023051433,0.25370958],"study_design_scores_gemma":[0.00018881826,0.0017980611,0.32766014,0.00081040943,0.00046213713,0.0016116454,0.005584556,0.3100977,0.0244578,0.31143975,0.015471634,0.00041734008],"about_ca_topic_score_codex":0.0048261248,"about_ca_topic_score_gemma":0.0029193927,"teacher_disagreement_score":0.9685878,"about_ca_system_score_codex":0.0024480997,"about_ca_system_score_gemma":0.0031764111,"threshold_uncertainty_score":0.16612548},"labels":[],"label_agreement":null},{"id":"W4404002909","doi":"10.1145/3702980","title":"Trained without My Consent: Detecting Code Inclusion in Language Models Trained on Code","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Fonds de recherche du Québec; Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Code (set theory); Inclusion (mineral); Programming language; Physics","score_opus":0.10487511590320443,"score_gpt":0.33703465410834965,"score_spread":0.23215953820514523,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404002909","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.3016911,0.0019015402,0.58236516,0.007719772,0.0010717622,0.0014420899,0.033166267,0.05346269,0.017179696],"genre_scores_gemma":[0.7038024,0.00032296023,0.22867717,0.002722151,0.00020807619,0.0012406096,0.05291411,0.0016638294,0.008448721],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9912145,0.0033542456,0.00062508095,0.0023789778,0.0019524347,0.00047482506],"domain_scores_gemma":[0.9630858,0.018177656,0.002245161,0.011643349,0.004089282,0.0007587372],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010788836,0.0009933998,0.00077092217,0.0012254332,0.00075715437,0.001926035,0.0018653753,0.0016700004,0.004654708],"category_scores_gemma":[0.068947665,0.0005505059,0.00096285035,0.00083342445,0.0011151519,0.00312864,0.0030705126,0.002805435,0.004878815],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.001762024,0.0006822518,0.09749233,0.000804162,0.00036208282,0.0018392222,0.0019930915,0.041946776,0.019336676,0.010288839,0.13154899,0.6919435],"study_design_scores_gemma":[0.00024005598,0.00041250812,0.018521024,0.00042203302,0.00012640569,0.0009485592,0.0006448225,0.85644585,0.026458694,0.037392415,0.058231685,0.00015592999],"about_ca_topic_score_codex":0.0066916468,"about_ca_topic_score_gemma":0.013144397,"teacher_disagreement_score":0.010788836,"about_ca_system_score_codex":0.00094107655,"about_ca_system_score_gemma":0.0036243794,"threshold_uncertainty_score":0.05705756},"labels":[],"label_agreement":null},{"id":"W4404060023","doi":"10.1145/3702983","title":"Repairs and Breaks Prediction for Deep Neural Networks","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"JST-Mirai Program","keywords":"Computer science; Artificial neural network; Artificial intelligence; Deep neural networks; Data science","score_opus":0.04038717887472058,"score_gpt":0.2922085178703729,"score_spread":0.2518213389956523,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404060023","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.64605546,0.0021328868,0.3412974,0.001191435,0.00021704887,0.00016451017,0.0016642484,0.005089524,0.0021874798],"genre_scores_gemma":[0.9469523,0.0002932753,0.049040027,0.00014934262,0.00003215396,0.00009729576,0.0018583909,0.000076977696,0.0015001926],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9989064,0.0002143539,0.00011396231,0.00031337113,0.00030699218,0.00014500784],"domain_scores_gemma":[0.99326766,0.0037451256,0.0011468338,0.00037905114,0.0012052081,0.0002560765],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002463661,0.0015342407,0.0006440231,0.0015568971,0.0003723142,0.0006189358,0.0014925865,0.0011747574,0.0010094681],"category_scores_gemma":[0.010682212,0.00042894893,0.0005559325,0.0007628549,0.00043133183,0.0013052617,0.00073103234,0.001942965,0.00030036768],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00052315264,0.00035937718,0.038843863,0.00015705073,0.000104118095,0.00019405645,0.0001434565,0.77785075,0.0029411274,0.0009846344,0.0043059755,0.17359245],"study_design_scores_gemma":[0.000004377181,0.000033268432,0.0010774754,0.000008938029,0.000007720744,0.00001034993,0.000012629235,0.9970795,0.00096626463,0.0006429538,0.00015216554,0.000004422418],"about_ca_topic_score_codex":0.021111565,"about_ca_topic_score_gemma":0.027576517,"teacher_disagreement_score":0.021111565,"about_ca_system_score_codex":0.0020200065,"about_ca_system_score_gemma":0.0010620266,"threshold_uncertainty_score":0.041977406},"labels":[],"label_agreement":null},{"id":"W4404060074","doi":"10.1145/3702979","title":"ZS4C: Zero-Shot Synthesis of Compilable Code for Incomplete Code Snippets Using LLMs","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University; Queen's University; University of Windsor; University of Manitoba","funders":"","keywords":"Computer science; Code (set theory); Zero (linguistics); Programming language; Linguistics","score_opus":0.1333233069938429,"score_gpt":0.3604581061756769,"score_spread":0.22713479918183402,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404060074","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.050171938,0.0015525544,0.48680526,0.0006575026,0.00068291195,0.0005871658,0.014560451,0.4387998,0.006182438],"genre_scores_gemma":[0.17394607,0.0006859449,0.7104835,0.000978777,0.00018924227,0.0010858624,0.057887238,0.04563713,0.009106202],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970407,0.00048426926,0.00022387633,0.0008967197,0.0011435624,0.00021086905],"domain_scores_gemma":[0.9936745,0.0031947824,0.00046267008,0.0013733234,0.0011370634,0.00015768348],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017801846,0.0038404106,0.0011744575,0.0034440102,0.00089903304,0.0019720767,0.0029975907,0.001692748,0.009031501],"category_scores_gemma":[0.012709071,0.0011835141,0.0024689506,0.0016421503,0.0015063358,0.0024345892,0.0028600988,0.0019847758,0.0072137685],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0018220335,0.00040820395,0.013673028,0.0034681996,0.0005027654,0.0017662862,0.0013692424,0.042521622,0.084329665,0.0078363605,0.17671032,0.66559225],"study_design_scores_gemma":[0.00038183815,0.0006379863,0.0064501534,0.0003716647,0.0002652653,0.0010971802,0.0007690455,0.6819559,0.17051227,0.026429106,0.11086968,0.0002599916],"about_ca_topic_score_codex":0.004637171,"about_ca_topic_score_gemma":0.012033711,"teacher_disagreement_score":0.009031501,"about_ca_system_score_codex":0.0009578751,"about_ca_system_score_gemma":0.0026531997,"threshold_uncertainty_score":0.030213416},"labels":[],"label_agreement":null},{"id":"W4404060176","doi":"10.1145/3702976","title":"Non-Linear Software Documentation with Interactive Code Examples","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McGill University","funders":"","keywords":"Computer science; Documentation; Software engineering; Software; Programming language","score_opus":0.06900117343039758,"score_gpt":0.34881224904692565,"score_spread":0.27981107561652807,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404060176","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.30464303,0.0015260592,0.61228603,0.003494464,0.00027892503,0.00092052243,0.0013146411,0.016134009,0.05940231],"genre_scores_gemma":[0.5309378,0.00054958946,0.4488563,0.0003962431,0.00009142871,0.00064687437,0.0012536722,0.0012658241,0.01600215],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9949935,0.0025557089,0.00046447603,0.00041687253,0.0013488508,0.00022058982],"domain_scores_gemma":[0.91436785,0.05893858,0.0049483925,0.013862554,0.006601756,0.0012808546],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0050445287,0.00060993154,0.00039295116,0.0017758749,0.0009424139,0.002985702,0.0018139352,0.0012623324,0.011231177],"category_scores_gemma":[0.04808603,0.00048131836,0.00037468455,0.001970725,0.0011377914,0.005872825,0.003551854,0.0014008665,0.0027777175],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010488572,0.00094212824,0.0106348675,0.0026833029,0.000043809127,0.0015827736,0.02503895,0.0072576767,0.034830756,0.04756466,0.031365458,0.83700675],"study_design_scores_gemma":[0.00064621045,0.0022588454,0.020002883,0.0033150732,0.00015295201,0.006130728,0.011993823,0.07559513,0.071392246,0.08682447,0.72124547,0.0004420865],"about_ca_topic_score_codex":0.00082590105,"about_ca_topic_score_gemma":0.0027853863,"teacher_disagreement_score":0.011231177,"about_ca_system_score_codex":0.0005792406,"about_ca_system_score_gemma":0.0016163912,"threshold_uncertainty_score":0.037572026},"labels":[],"label_agreement":null},{"id":"W4404373849","doi":"10.1145/3702985","title":"A Machine Learning Approach for Automated Filling of Categorical Fields in Data Entry Forms—RCR Report","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"BNP Paribas Cardif","keywords":"Computer science; Categorical variable; Artificial intelligence; Software engineering; Machine learning; Programming language; Data mining","score_opus":0.10919899534806334,"score_gpt":0.34966833276716797,"score_spread":0.24046933741910465,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404373849","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.004772072,0.00012999038,0.96533257,0.0010147889,0.00013139391,0.0003544817,0.0016058361,0.025459427,0.0011995587],"genre_scores_gemma":[0.032480948,0.000054028234,0.96213955,0.00024066435,0.00006647711,0.00039297255,0.002295587,0.0011904993,0.0011392637],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9565956,0.019029824,0.0047365706,0.006343305,0.012123283,0.0011713709],"domain_scores_gemma":[0.8683937,0.057347346,0.00464709,0.051872205,0.01631513,0.0014244737],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02242103,0.0014774561,0.0018052962,0.0047881342,0.0021960677,0.0059084888,0.00698022,0.0029468352,0.017069157],"category_scores_gemma":[0.14016901,0.0016307747,0.004038628,0.0042421087,0.0028241565,0.012221376,0.006364879,0.005443709,0.011589341],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0009000588,0.0007002763,0.005699503,0.00055540586,0.00013977158,0.00048000755,0.0010903078,0.02763703,0.010568619,0.0637809,0.07835115,0.8100971],"study_design_scores_gemma":[0.00026854945,0.00021538907,0.0012849013,0.00017675044,0.000058357688,0.00064065796,0.00046217773,0.792602,0.028498918,0.1159531,0.05960739,0.00023183231],"about_ca_topic_score_codex":0.007447238,"about_ca_topic_score_gemma":0.0066798003,"teacher_disagreement_score":0.02242103,"about_ca_system_score_codex":0.0027276287,"about_ca_system_score_gemma":0.0069207973,"threshold_uncertainty_score":0.118575156},"labels":[],"label_agreement":null},{"id":"W4404570261","doi":"10.1145/3743133","title":"Developer Perspectives on Licensing and Copyright Issues Arising from Generative AI for Software Development","year":2025,"lang":"en","type":"preprint","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Law, AI, and Intellectual Property","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"National Science Foundation","keywords":"Generative grammar; Coding (social sciences); Copyright law; Computer science; Artificial intelligence; Intellectual property; Sociology; Social science","score_opus":0.09700942967423726,"score_gpt":0.33212191879071784,"score_spread":0.23511248911648058,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404570261","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7860141,0.0029945492,0.048935693,0.04578356,0.00022764811,0.00015157477,0.00006847966,0.00017116625,0.11565324],"genre_scores_gemma":[0.98817796,0.0010546374,0.0043691127,0.002064315,0.000047617745,0.0000632201,0.000030358426,0.000087184846,0.004105629],"study_design_codex":"qualitative","study_design_gemma":"qualitative","domain_scores_codex":[0.9177951,0.05927286,0.0027916122,0.0021810024,0.015128454,0.0028309869],"domain_scores_gemma":[0.6100009,0.34117422,0.013526031,0.009850817,0.021715576,0.0037325071],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07740404,0.0004771013,0.0004363922,0.003693746,0.00812512,0.0120568285,0.0014505017,0.0038892268,0.003206026],"category_scores_gemma":[0.17986001,0.00079688826,0.0005347441,0.0030359367,0.014225615,0.012968457,0.008245942,0.005504939,0.00044464652],"study_design_candidate":"qualitative","study_design_consensus":"qualitative","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000072731855,0.000048542544,0.023184977,0.00021261538,0.000019626395,0.0023706825,0.84270734,0.0005972769,0.003061743,0.07423978,0.0028638984,0.05062073],"study_design_scores_gemma":[0.000037195412,0.00017381852,0.01189586,0.000937121,0.000054016655,0.0028883463,0.7544888,0.0035520813,0.0041897413,0.049863704,0.17177384,0.00014543066],"about_ca_topic_score_codex":0.005431627,"about_ca_topic_score_gemma":0.008124798,"teacher_disagreement_score":0.07740404,"about_ca_system_score_codex":0.0076327487,"about_ca_system_score_gemma":0.0054476187,"threshold_uncertainty_score":0.4093566},"labels":[],"label_agreement":null},{"id":"W4404635154","doi":"10.1145/3705309","title":"Detecting Refactoring Commits in Machine Learning Python Projects: A Machine Learning-Based Approach","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Queen's University","funders":"","keywords":"Code refactoring; Computer science; Maintainability; Python (programming language); Software engineering; Java; Programming language; Software; Software development; Artificial intelligence; Machine learning","score_opus":0.09068230133371412,"score_gpt":0.3197279064261304,"score_spread":0.22904560509241628,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4404635154","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.4966501,0.0038665126,0.439967,0.0014069127,0.000501274,0.0015082407,0.01315624,0.036248405,0.006695309],"genre_scores_gemma":[0.6836524,0.0006322105,0.28518766,0.0003896836,0.0002009567,0.00065166125,0.023776462,0.0005354983,0.004973387],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99225205,0.000958468,0.0009818639,0.0026146893,0.002451739,0.0007411957],"domain_scores_gemma":[0.9791443,0.0071102646,0.004693426,0.0024046514,0.0056228535,0.001024528],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004857549,0.001568887,0.001418849,0.01087656,0.0010773907,0.0021334868,0.002880781,0.0018624122,0.0011272583],"category_scores_gemma":[0.018672936,0.0005176634,0.0014218369,0.004731525,0.0006450117,0.0023773387,0.0018900326,0.0019053183,0.0018734427],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005643325,0.0012942326,0.29552376,0.0007988716,0.00034831036,0.00154814,0.00086148805,0.026482167,0.010463891,0.0018818207,0.021855572,0.63837737],"study_design_scores_gemma":[0.00007068383,0.00038134726,0.08840367,0.00023684306,0.00026738815,0.0012011067,0.0007741007,0.87018025,0.018058028,0.00625794,0.014020757,0.00014791946],"about_ca_topic_score_codex":0.008461891,"about_ca_topic_score_gemma":0.012582295,"teacher_disagreement_score":0.01087656,"about_ca_system_score_codex":0.0010998694,"about_ca_system_score_gemma":0.0024633294,"threshold_uncertainty_score":0.025689542},"labels":[],"label_agreement":null},{"id":"W4405396136","doi":"10.1145/3708519","title":"Automatic Programming: Large Language Models and Beyond","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":33,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Bộ Giáo dục và Ðào tạo; Ministry of Education, India","keywords":"Computer science; Programmer; Coding (social sciences); Software engineering; Popularity; Software deployment; Programming language; Key (lock); Computer security","score_opus":0.05968806315916537,"score_gpt":0.33576718525188126,"score_spread":0.2760791220927159,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405396136","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009965851,0.0014914011,0.97095245,0.0037583855,0.00009565885,0.00010418605,0.00041044346,0.0022298843,0.010991781],"genre_scores_gemma":[0.2633992,0.0031765634,0.71714664,0.0014653134,0.00047801764,0.00078109285,0.0014204689,0.0024556576,0.009677043],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.99477625,0.002345012,0.0002466421,0.0007361267,0.0016309092,0.00026500938],"domain_scores_gemma":[0.9757829,0.016439559,0.0011723186,0.004596164,0.001584777,0.0004243088],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004962765,0.00083473226,0.0009310224,0.0015368704,0.0012856335,0.005560269,0.0031098612,0.0018232376,0.005819689],"category_scores_gemma":[0.020628989,0.0011929276,0.0020523826,0.0015249613,0.004078828,0.012480675,0.0035846105,0.0049659875,0.0017139299],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000047998696,0.00005925855,0.001097725,0.00022632678,0.00004769448,0.00016624862,0.0009206434,0.0326223,0.0011590582,0.9262631,0.0048767123,0.032512926],"study_design_scores_gemma":[0.00002006936,0.000018408566,0.00022143484,0.00011852688,0.000019063198,0.0001166784,0.00010356177,0.1339834,0.0006784028,0.834369,0.030325301,0.00002618871],"about_ca_topic_score_codex":0.004570557,"about_ca_topic_score_gemma":0.0042747064,"teacher_disagreement_score":0.005819689,"about_ca_system_score_codex":0.0020745185,"about_ca_system_score_gemma":0.0027021372,"threshold_uncertainty_score":0.026245952},"labels":[],"label_agreement":null},{"id":"W4405444248","doi":"10.1145/3708473","title":"Leveraging Data Characteristics for Bug Localization in Deep Learning Programs","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research; National Science Foundation","keywords":"Computer science; Deep learning; Artificial intelligence; Software engineering; Data science; Machine learning","score_opus":0.13358464562709096,"score_gpt":0.344974517411349,"score_spread":0.21138987178425803,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405444248","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6860875,0.0025187326,0.2727672,0.001767032,0.00018376346,0.00021252547,0.006101637,0.028069226,0.0022924158],"genre_scores_gemma":[0.89873576,0.00029280153,0.08971149,0.00036426366,0.00004575873,0.00017704605,0.0093452595,0.00034815972,0.0009795935],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983701,0.000317777,0.00020395979,0.0004959175,0.0004369451,0.0001753506],"domain_scores_gemma":[0.99077576,0.0046053817,0.0012406972,0.0014312351,0.0016420877,0.00030480282],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002018475,0.0016157489,0.00081331347,0.0031129848,0.00042939754,0.0010818564,0.0017662655,0.0012845969,0.00078940956],"category_scores_gemma":[0.016091252,0.00052888226,0.0006951669,0.0017453989,0.00066012936,0.0027044206,0.0016371405,0.0020972686,0.00051037344],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008120402,0.001136326,0.1632457,0.0008243216,0.00024322832,0.00058850733,0.00044276184,0.22638814,0.017860027,0.0020383776,0.01709382,0.5693268],"study_design_scores_gemma":[0.000035758334,0.00014467037,0.0064195837,0.00003575717,0.000046184734,0.000089900896,0.00008036169,0.9788329,0.008941648,0.0031751813,0.0021758715,0.00002208156],"about_ca_topic_score_codex":0.0057045114,"about_ca_topic_score_gemma":0.009396717,"teacher_disagreement_score":0.0057045114,"about_ca_system_score_codex":0.0011020845,"about_ca_system_score_gemma":0.0012690544,"threshold_uncertainty_score":0.011342645},"labels":[],"label_agreement":null},{"id":"W4405570253","doi":"10.1145/3708523","title":"The Impact of Generative AI on Creativity in Software Development: A Research Agenda","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Creativity in Education and Neuroscience","field":"Psychology","cited_by":17,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Dalhousie University","funders":"Conselho Nacional de Desenvolvimento Científico e Tecnológico; Natural Sciences and Engineering Research Council of Canada; Coordenação de Aperfeiçoamento de Pessoal de Nível Superior; Alfred P. Sloan Foundation; National Science Foundation","keywords":"Computer science; Generative grammar; Creativity; Software development; Software engineering; Software; Engineering ethics; Artificial intelligence; Psychology; Programming language; Engineering","score_opus":0.29514797537011833,"score_gpt":0.5079426103073964,"score_spread":0.21279463493727807,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405570253","genre_codex":"other","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.24565044,0.050911468,0.07476425,0.08185963,0.00069010817,0.00023093441,0.0001734834,0.00027675036,0.5454429],"genre_scores_gemma":[0.9552971,0.0238019,0.013836523,0.0017948087,0.00029268264,0.00012685063,0.000051977637,0.0000560179,0.004742183],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9944494,0.00325552,0.00015147842,0.0004135059,0.0011418146,0.00058828184],"domain_scores_gemma":[0.9301179,0.059600987,0.002256868,0.003033606,0.0030654278,0.0019251683],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008852249,0.0007003386,0.00053474866,0.0023129906,0.002446961,0.011565948,0.0020892832,0.0020894282,0.007547517],"category_scores_gemma":[0.0254158,0.00044200206,0.0006979952,0.0025779696,0.013419786,0.01433575,0.0055983523,0.0038010385,0.00080009503],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014930294,0.00031041115,0.01636423,0.0011642395,0.000081119666,0.00053184177,0.015577708,0.0061212163,0.00093631784,0.8157818,0.0024761863,0.1405056],"study_design_scores_gemma":[0.000040052993,0.00036079937,0.015842019,0.0022429319,0.00009931455,0.0009716249,0.0339845,0.013397463,0.0024740263,0.8551659,0.07532199,0.00009938191],"about_ca_topic_score_codex":0.002497823,"about_ca_topic_score_gemma":0.002709616,"teacher_disagreement_score":0.011565948,"about_ca_system_score_codex":0.0030223415,"about_ca_system_score_gemma":0.0038169972,"threshold_uncertainty_score":0.046815693},"labels":[],"label_agreement":null},{"id":"W4405891407","doi":"10.1145/3709357","title":"Software Fairness Debt: Building a Research Agenda for Addressing Bias in AI Systems","year":2024,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":12,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Technical debt; Computer science; Transparency (behavior); Software; Software development; Fairness measure; Software system; Process management; Knowledge management; Risk analysis (engineering); Engineering management; Software engineering; Business; Computer security; Engineering; Telecommunications","score_opus":0.37015791711926477,"score_gpt":0.4458075324305976,"score_spread":0.07564961531133285,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4405891407","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.029668307,0.032914538,0.4396593,0.43617627,0.0023182048,0.0006396653,0.00018192477,0.00033455118,0.05810725],"genre_scores_gemma":[0.77469844,0.024787687,0.16712281,0.023830373,0.0022051323,0.0014283583,0.0002016475,0.0002554546,0.0054700524],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.92943394,0.052391946,0.002933367,0.0037909776,0.008618138,0.0028315866],"domain_scores_gemma":[0.67513,0.26043025,0.011982277,0.019274902,0.027677579,0.005504929],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.12604621,0.0011258541,0.001671498,0.007966105,0.009087935,0.021001449,0.005198322,0.010564625,0.0057745194],"category_scores_gemma":[0.2052666,0.0009701502,0.0017136249,0.0066333944,0.040221576,0.05144259,0.020482248,0.011998405,0.0008646927],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00003264561,0.00006583134,0.0036932789,0.0010177844,0.00004383564,0.00010891608,0.012319685,0.0021942467,0.0001980384,0.92102313,0.0038053154,0.055497207],"study_design_scores_gemma":[0.000014781243,0.00003956787,0.00074811117,0.002658025,0.00003261214,0.00006619891,0.007187496,0.0029345264,0.00037629448,0.95365906,0.03224691,0.000036431575],"about_ca_topic_score_codex":0.007521046,"about_ca_topic_score_gemma":0.006362892,"teacher_disagreement_score":0.8739538,"about_ca_system_score_codex":0.013432033,"about_ca_system_score_gemma":0.03903042,"threshold_uncertainty_score":0.66660404},"labels":[],"label_agreement":null},{"id":"W4406153408","doi":"10.1145/3711816","title":"LLM-Powered Static Binary Taint Analysis","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Web Application Security Vulnerabilities","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Ant Group","keywords":"Taint checking; Computer science; Firmware; Static analysis; Vulnerability (computing); Binary number; Scalability; State (computer science); Computer security; Software; Computer hardware; Operating system; Programming language","score_opus":0.04508240882382882,"score_gpt":0.3181407358867413,"score_spread":0.2730583270629125,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406153408","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.04539574,0.0008185274,0.8283787,0.0009367223,0.00019854984,0.00019947755,0.0011342234,0.11507565,0.007862436],"genre_scores_gemma":[0.5762294,0.00051448465,0.40066192,0.0013479622,0.0001775778,0.00029906924,0.0026831878,0.010078391,0.0080079455],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9967225,0.0008223572,0.00019922863,0.00044053092,0.0015110713,0.00030430872],"domain_scores_gemma":[0.9937748,0.0020893298,0.0006614258,0.0021850711,0.00113493,0.00015452654],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0018563883,0.0012144812,0.0007967296,0.0026323749,0.0007008471,0.0020734782,0.0018392972,0.0008635251,0.00564243],"category_scores_gemma":[0.009924416,0.0008330283,0.0014387621,0.00079668587,0.0015605096,0.0063906214,0.0038914185,0.0016820368,0.002601721],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010012002,0.00043961522,0.023514824,0.0015373675,0.00027676162,0.0011819035,0.0011655713,0.04632268,0.11394856,0.09358767,0.04852843,0.66849536],"study_design_scores_gemma":[0.00008371367,0.00031809462,0.0031011894,0.0002985411,0.0001960141,0.0011645219,0.00031197487,0.7323647,0.1024187,0.08550001,0.074013144,0.00022936155],"about_ca_topic_score_codex":0.001964323,"about_ca_topic_score_gemma":0.0038468668,"teacher_disagreement_score":0.00564243,"about_ca_system_score_codex":0.0011341879,"about_ca_system_score_gemma":0.0026215545,"threshold_uncertainty_score":0.018875837},"labels":[],"label_agreement":null},{"id":"W4406385999","doi":"10.1145/3712196","title":"System Safety Monitoring of Learned Components Using Temporal Metric Forecasting","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Fault Detection and Control Systems","field":"Engineering","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Alliance de recherche numérique du Canada; Natural Sciences and Engineering Research Council of Canada; Mitacs; Canada Research Chairs; Science Foundation Ireland","keywords":"Computer science; Metric (unit); Real-time computing; Systems engineering; Engineering; Operations management","score_opus":0.10698489540344563,"score_gpt":0.3098304973908639,"score_spread":0.20284560198741824,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406385999","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.42430404,0.0005452159,0.56911016,0.00036358193,0.00006995061,0.00007023297,0.0004419535,0.0031859444,0.001908998],"genre_scores_gemma":[0.9831671,0.000062533356,0.016221978,0.000024419756,0.000007741426,0.000017534894,0.00019695009,0.000028411365,0.0002733338],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9995782,0.00007045463,0.00003199859,0.00013301517,0.00013569272,0.000050606315],"domain_scores_gemma":[0.99820423,0.000797108,0.00028519073,0.00020427664,0.00042835396,0.00008078924],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0010550568,0.0012043976,0.0004944927,0.0006849167,0.00019707263,0.0004955917,0.0008976369,0.00035784373,0.00057867804],"category_scores_gemma":[0.004711137,0.0002390458,0.00041785373,0.00049737276,0.00028683,0.0012192044,0.000481512,0.0008967902,0.00018177895],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001369941,0.00005435796,0.006909126,0.00004374254,0.000035251778,0.000045723118,0.000050829167,0.91777945,0.0034053621,0.00080292847,0.00050497364,0.070231244],"study_design_scores_gemma":[0.0000014700024,0.00001745259,0.0004945693,0.0000018923761,0.0000045700635,0.0000059446024,0.000003811747,0.99792117,0.0009514712,0.00052188186,0.00007301865,0.0000028017803],"about_ca_topic_score_codex":0.010887687,"about_ca_topic_score_gemma":0.010979608,"teacher_disagreement_score":0.010887687,"about_ca_system_score_codex":0.0009022876,"about_ca_system_score_gemma":0.0009434746,"threshold_uncertainty_score":0.021648645},"labels":[],"label_agreement":null},{"id":"W4406465537","doi":"10.1145/3711904","title":"Making Software Development More Diverse and Inclusive: Key Themes, Challenges, and Future Directions","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"University-Industry-Government Innovation Models","field":"Business, Management and Accounting","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; University of British Columbia","funders":"Conselho Nacional de Desenvolvimento Científico e Tecnológico","keywords":"Computer science; Key (lock); Software development; Software engineering; Engineering ethics; Data science; Software; Systems engineering; Engineering; Computer security","score_opus":0.07451218141943525,"score_gpt":0.2934927424907315,"score_spread":0.21898056107129624,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406465537","genre_codex":"commentary","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.17784305,0.04852721,0.07590133,0.65641856,0.00396546,0.0011403906,0.00029358445,0.00026551235,0.035644922],"genre_scores_gemma":[0.8549911,0.034154706,0.08271774,0.018015426,0.000984509,0.0010129627,0.0002451131,0.000107589294,0.0077708582],"study_design_codex":"design_other","study_design_gemma":"qualitative","domain_scores_codex":[0.9806474,0.011102751,0.0011758696,0.0015087778,0.003234832,0.0023303211],"domain_scores_gemma":[0.957481,0.02448845,0.0027811904,0.0014850169,0.009058428,0.004705954],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.033299986,0.0008663231,0.00061789516,0.0021871575,0.008811267,0.015794097,0.0024293857,0.004809173,0.0031162922],"category_scores_gemma":[0.023418713,0.0006226235,0.00085817947,0.0035787362,0.01514432,0.022180175,0.0104567,0.007674294,0.000581768],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000143121,0.00032780066,0.017491465,0.005053329,0.000043466945,0.0012750713,0.31002,0.00077135384,0.0028883496,0.2799624,0.026576733,0.355447],"study_design_scores_gemma":[0.000028597753,0.0001623531,0.0056104576,0.0038955198,0.00003381434,0.0014140427,0.5624527,0.001914177,0.0012746961,0.23731107,0.18578349,0.00011914743],"about_ca_topic_score_codex":0.0035187604,"about_ca_topic_score_gemma":0.003866016,"teacher_disagreement_score":0.033299986,"about_ca_system_score_codex":0.010593024,"about_ca_system_score_gemma":0.029251639,"threshold_uncertainty_score":0.17610925},"labels":[],"label_agreement":null},{"id":"W4406494387","doi":"10.1145/3712008","title":"Automation in Model-Driven Engineering: A Look Back, and Ahead","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Model-Driven Software Engineering Techniques","field":"Computer Science","cited_by":11,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"Agencia Estatal de Investigación; European Commission; Natural Sciences and Engineering Research Council of Canada; Ministerio de Ciencia e Innovación; Universidad de Málaga","keywords":"Computer science; Automation; Software engineering; Systems engineering; Engineering","score_opus":0.04854884486490869,"score_gpt":0.2991379539605627,"score_spread":0.250589109095654,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406494387","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.009139972,0.4120086,0.32811978,0.21497673,0.0044340505,0.00008763877,0.00014729821,0.0007578864,0.030328065],"genre_scores_gemma":[0.30893806,0.3750341,0.27019233,0.021252647,0.00887912,0.00030685452,0.00032642786,0.00076732395,0.014303142],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9904974,0.004911231,0.00040438076,0.00079013355,0.0030085824,0.00038816832],"domain_scores_gemma":[0.9705198,0.0214758,0.0006014869,0.0033962172,0.0030504933,0.00095616357],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.019338826,0.0011883735,0.0014378293,0.0025670193,0.0019472787,0.008629524,0.002565645,0.005467483,0.003947463],"category_scores_gemma":[0.021396928,0.0013335485,0.0013498353,0.0027747431,0.01052707,0.035990197,0.00574233,0.013248755,0.0017533128],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00012379099,0.00017806429,0.00092149025,0.0013855097,0.00007354735,0.00014502366,0.0014796483,0.0063163093,0.0006109952,0.7812958,0.018204167,0.18926561],"study_design_scores_gemma":[0.00003238651,0.00013771665,0.00036131297,0.0013152632,0.00003158497,0.00025674983,0.0010922664,0.019023042,0.00083729404,0.79781264,0.17899404,0.000105774554],"about_ca_topic_score_codex":0.0027474405,"about_ca_topic_score_gemma":0.0023392132,"teacher_disagreement_score":0.019338826,"about_ca_system_score_codex":0.0039402717,"about_ca_system_score_gemma":0.0032120799,"threshold_uncertainty_score":0.102274716},"labels":[],"label_agreement":null},{"id":"W4406688098","doi":"10.1145/3714461","title":"Exploring Parameter-Efficient Fine-Tuning Techniques for Code Generation with Large Language Models","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":43,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Code generation; Code (set theory); Programming language; Software engineering; Key (lock); Operating system; Set (abstract data type)","score_opus":0.13664566024265196,"score_gpt":0.3367384357981753,"score_spread":0.20009277555552335,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406688098","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0520844,0.0018874794,0.9171027,0.00063370453,0.00014537963,0.00027946118,0.00069186505,0.023796469,0.0033785922],"genre_scores_gemma":[0.5033743,0.000656407,0.48738304,0.0008115688,0.00007192721,0.00066747743,0.0021002719,0.0029843948,0.0019506399],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9987362,0.00045680944,0.00009652996,0.0003672533,0.00023434592,0.000108886015],"domain_scores_gemma":[0.9957652,0.0028300423,0.000196191,0.00079928496,0.00029651998,0.00011273708],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017527387,0.0016641556,0.0008798523,0.0008293414,0.000416937,0.0015001369,0.0025161195,0.0013135881,0.003332289],"category_scores_gemma":[0.015830526,0.00073559175,0.0012062811,0.000756048,0.0008858412,0.002860836,0.0019076803,0.0029362917,0.0023281712],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037426336,0.00038955032,0.0055670594,0.0007574429,0.0001892885,0.0002879374,0.00058380153,0.49560973,0.02405798,0.010345452,0.010918698,0.45091888],"study_design_scores_gemma":[0.00006119341,0.00006936251,0.00024622332,0.000032410095,0.00002515733,0.00006073649,0.00007670412,0.9784138,0.005627974,0.011804465,0.0035635352,0.000018474078],"about_ca_topic_score_codex":0.005018904,"about_ca_topic_score_gemma":0.012148481,"teacher_disagreement_score":0.005018904,"about_ca_system_score_codex":0.000963216,"about_ca_system_score_gemma":0.0018172908,"threshold_uncertainty_score":0.011147618},"labels":[],"label_agreement":null},{"id":"W4406738291","doi":"10.1145/3715006","title":"The Good, the Bad, and the Monstrous: Predicting Highly Change-Prone Source Code Methods at Their Inception","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Source code; Code (set theory); Computer security; Programming language","score_opus":0.0632421140070817,"score_gpt":0.3331406067424445,"score_spread":0.2698984927353628,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406738291","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.97677386,0.00079123233,0.0199747,0.00064541376,0.000024925585,0.000029990131,0.00031816875,0.00033037964,0.0011112888],"genre_scores_gemma":[0.9859098,0.00016247043,0.012746798,0.0000850195,0.000019367848,0.000014339149,0.0005178632,0.000050199647,0.00049405],"study_design_codex":"observational","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99747616,0.00073787605,0.00016362558,0.000539363,0.00077636295,0.00030660554],"domain_scores_gemma":[0.96412593,0.024062086,0.0045519844,0.0022662014,0.0035685417,0.0014252681],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.006565736,0.0008637313,0.00041772355,0.004637455,0.00085097505,0.0018993593,0.0007289036,0.0012063852,0.0004527418],"category_scores_gemma":[0.029815756,0.0003996367,0.00044006205,0.0019107797,0.0011174685,0.0026339428,0.0012556394,0.0019458095,0.0004137377],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00032192483,0.0003346978,0.8506148,0.00011776178,0.000079904035,0.00030252337,0.001116201,0.023595445,0.0024043343,0.0013110742,0.0024951717,0.11730613],"study_design_scores_gemma":[0.000039882772,0.00034853804,0.37988216,0.00019120285,0.00014105812,0.000684537,0.00243406,0.58978695,0.0078034136,0.013159353,0.0054398444,0.000089072826],"about_ca_topic_score_codex":0.010300768,"about_ca_topic_score_gemma":0.018761799,"teacher_disagreement_score":0.010300768,"about_ca_system_score_codex":0.00088716904,"about_ca_system_score_gemma":0.0018274263,"threshold_uncertainty_score":0.03472334},"labels":[],"label_agreement":null},{"id":"W4406866174","doi":"10.1145/3715008","title":"Scalable Similarity-Aware Test Suite Minimization with Reinforcement Learning","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Computer science; Test suite; Reinforcement learning; Suite; Scalability; Minification; Similarity (geometry); Test (biology); Artificial intelligence; Software engineering; Machine learning; Test case; Programming language; Operating system","score_opus":0.04087899492526334,"score_gpt":0.2917156633049735,"score_spread":0.2508366683797102,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406866174","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.07953009,0.00064597087,0.909829,0.000604361,0.000058195175,0.00030131187,0.0002484203,0.0058884732,0.0028941503],"genre_scores_gemma":[0.7348679,0.00014360348,0.26067922,0.00041369797,0.000056868877,0.00044504934,0.0010869538,0.0005576891,0.0017490966],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99798125,0.0007610918,0.00009571987,0.00040095206,0.00053355127,0.00022738292],"domain_scores_gemma":[0.9937338,0.0044193217,0.00051197724,0.0005177365,0.0005641155,0.00025301834],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0019303793,0.0017302461,0.001463363,0.0011304257,0.0003786787,0.00076562155,0.002193855,0.0011019119,0.0026208963],"category_scores_gemma":[0.009485103,0.0006529322,0.0011574406,0.0008268983,0.0010692684,0.0013616671,0.001498418,0.0021044281,0.00053345965],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0001349174,0.00028871695,0.0021443788,0.000179478,0.00006627223,0.00012526775,0.000055400043,0.85406226,0.0044277934,0.0027284687,0.0029034482,0.13288364],"study_design_scores_gemma":[0.000024971869,0.00005504497,0.00013529489,0.000006006191,0.00000844138,0.000020342357,0.000008706248,0.99590826,0.00077976397,0.0028293168,0.00022052914,0.0000032358234],"about_ca_topic_score_codex":0.0038278026,"about_ca_topic_score_gemma":0.0049206633,"teacher_disagreement_score":0.0038278026,"about_ca_system_score_codex":0.001221734,"about_ca_system_score_gemma":0.001960493,"threshold_uncertainty_score":0.010208964},"labels":[],"label_agreement":null},{"id":"W4406866661","doi":"10.1145/3715109","title":"HumanEvalComm: Benchmarking the Communication Competence of Code Generation for LLMs and LLM Agent","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; Kelowna General Hospital; University of British Columbia","funders":"","keywords":"Benchmarking; Computer science; Competence (human resources); Knowledge management; Business; Psychology; Marketing","score_opus":0.08995787232149585,"score_gpt":0.34867155279438006,"score_spread":0.25871368047288423,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406866661","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.94323725,0.0005036569,0.035617195,0.00054686354,0.00016246836,0.0009730293,0.0016025918,0.0064333915,0.010923616],"genre_scores_gemma":[0.9118089,0.00015994349,0.07518152,0.00035409292,0.000029813818,0.0012553494,0.006818368,0.0006652083,0.003726848],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9849145,0.009136591,0.0013057032,0.0017967961,0.0021900837,0.00065632653],"domain_scores_gemma":[0.93659776,0.0412175,0.0031974225,0.009975861,0.0062996875,0.0027116719],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.010986316,0.0011106329,0.0005602631,0.0018712084,0.00068466284,0.0018754794,0.0022461088,0.0019427042,0.0024312495],"category_scores_gemma":[0.054456152,0.00043119947,0.00061232666,0.0011072088,0.0017629663,0.002683057,0.0034729508,0.0024716798,0.0012187858],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.007538795,0.016112057,0.099203765,0.004483792,0.00066604,0.0010579686,0.018573293,0.14214353,0.042805314,0.017902922,0.049624592,0.5998879],"study_design_scores_gemma":[0.0022441417,0.013761508,0.11575112,0.00058277836,0.00025043747,0.0010702792,0.008640958,0.6559467,0.075216495,0.020470269,0.105456986,0.0006084432],"about_ca_topic_score_codex":0.0041897814,"about_ca_topic_score_gemma":0.004271389,"teacher_disagreement_score":0.010986316,"about_ca_system_score_codex":0.001540262,"about_ca_system_score_gemma":0.0019257758,"threshold_uncertainty_score":0.058101892},"labels":[],"label_agreement":null},{"id":"W4406957844","doi":"10.1145/3712002","title":"Quantum Software Engineering: Roadmap and Challenges Ahead","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Quantum Computing Algorithms and Architecture","field":"Computer Science","cited_by":51,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Toronto Metropolitan University","funders":"Natural Sciences and Engineering Research Council of Canada; Ministerio de Ciencia e Innovación; European Commission","keywords":"Computer science; Software engineering; Software requirements; Software; Systems engineering; Software development; Software construction; Programming language; Engineering","score_opus":0.05628912930751191,"score_gpt":0.285834054176502,"score_spread":0.22954492486899009,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406957844","genre_codex":"commentary","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.015241951,0.28939125,0.17336366,0.4806907,0.0043722615,0.00021114474,0.00025708295,0.0008222576,0.035649795],"genre_scores_gemma":[0.27753082,0.4262945,0.25347435,0.02560243,0.006033299,0.00074412505,0.00067211926,0.0005189621,0.009129368],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.9926475,0.0034889826,0.0003728508,0.0005910538,0.0022204898,0.0006790458],"domain_scores_gemma":[0.93761563,0.04592626,0.0010350351,0.0032226415,0.009537875,0.00266255],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021810252,0.0011582705,0.0014168891,0.0035675569,0.0029180811,0.008735271,0.003467224,0.008072298,0.009625739],"category_scores_gemma":[0.03284924,0.00089240377,0.0013125157,0.0031533136,0.009664441,0.027238838,0.0073492345,0.0096238395,0.0026147682],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00010426726,0.00026324845,0.00068183517,0.0017828052,0.000036391375,0.0001360615,0.00094605866,0.0075436225,0.0008682246,0.71995664,0.030501988,0.23717886],"study_design_scores_gemma":[0.000029738789,0.00014310739,0.0003105557,0.0011796907,0.000014959436,0.0001326289,0.002030167,0.016402844,0.00066323567,0.86301655,0.116018035,0.00005851575],"about_ca_topic_score_codex":0.003275798,"about_ca_topic_score_gemma":0.0029131353,"teacher_disagreement_score":0.021810252,"about_ca_system_score_codex":0.0036753311,"about_ca_system_score_gemma":0.013658839,"threshold_uncertainty_score":0.115345},"labels":[],"label_agreement":null},{"id":"W4406957873","doi":"10.1145/3715693","title":"Assessing the Robustness of Test Selection Methods for Deep Neural Networks","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Robustness (evolution); Artificial intelligence; Artificial neural network; Machine learning; Robustness testing; Biology","score_opus":0.05595642973612091,"score_gpt":0.38734945323747966,"score_spread":0.33139302350135874,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4406957873","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.55061644,0.004046886,0.43720192,0.0014371708,0.00028398403,0.00043835456,0.00071338954,0.0022399917,0.0030218773],"genre_scores_gemma":[0.9368037,0.0002823818,0.061106656,0.00021970934,0.00007277293,0.00021280585,0.0007267164,0.00016390112,0.00041127024],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9754701,0.013873562,0.0019206365,0.002141975,0.005849483,0.0007441508],"domain_scores_gemma":[0.70974886,0.24563453,0.014235136,0.018684816,0.010218598,0.001478001],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0404065,0.0014216339,0.00077393855,0.0030639528,0.0006309539,0.0011556074,0.0019073617,0.0018994577,0.0007317656],"category_scores_gemma":[0.19915907,0.00051131076,0.0009153992,0.0013145249,0.0018986333,0.0017297618,0.0020001354,0.0019603223,0.00032748436],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025465358,0.0005330928,0.14040475,0.0007260109,0.0013422352,0.00037600237,0.0004973376,0.5728644,0.011087346,0.010903579,0.004804438,0.25391427],"study_design_scores_gemma":[0.0000992087,0.00077446515,0.01163297,0.0001247513,0.00009790899,0.0002447726,0.00012317814,0.96169966,0.015727516,0.008327031,0.0011075354,0.00004095096],"about_ca_topic_score_codex":0.0020861588,"about_ca_topic_score_gemma":0.0016963055,"teacher_disagreement_score":0.0404065,"about_ca_system_score_codex":0.0013099586,"about_ca_system_score_gemma":0.0012672583,"threshold_uncertainty_score":0.21369255},"labels":[],"label_agreement":null},{"id":"W4407084737","doi":"10.1145/3715111","title":"Software Engineering by and for Humans in an AI Era","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Innovative Human-Technology Interaction","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Generalitat Valenciana; Schweizerischer Nationalfonds zur Förderung der Wissenschaftlichen Forschung","keywords":"Computer science; Software engineering; Software; Programming language","score_opus":0.054314640578669325,"score_gpt":0.34711768674939253,"score_spread":0.2928030461707232,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407084737","genre_codex":"commentary","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.03797664,0.03842043,0.2859709,0.387015,0.0033883718,0.0001713201,0.00010384522,0.00084418204,0.24610932],"genre_scores_gemma":[0.7287244,0.02434607,0.15638411,0.037819568,0.0026975288,0.00044356933,0.0001459022,0.0006748258,0.048763987],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","domain_scores_codex":[0.97402745,0.018051792,0.00079576945,0.0019218486,0.0043326304,0.0008704897],"domain_scores_gemma":[0.96499276,0.020735696,0.0018956464,0.006365108,0.0038310878,0.0021797158],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0221631,0.0007143775,0.0006038066,0.0020865996,0.0051894593,0.014179217,0.0016661779,0.0049230875,0.006489315],"category_scores_gemma":[0.029106187,0.00040448052,0.0004455088,0.001703303,0.03342192,0.022357678,0.009073308,0.006009363,0.0019705691],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000027406912,0.000042843112,0.000985299,0.00021455926,0.000022253693,0.00008977155,0.009348507,0.0007243796,0.0006475873,0.9067692,0.010870198,0.070258],"study_design_scores_gemma":[0.000011219652,0.000040440216,0.00030223042,0.0004137194,0.000013580325,0.00014032533,0.004841667,0.0009464687,0.00057509134,0.7445676,0.24812396,0.000023658611],"about_ca_topic_score_codex":0.0021143437,"about_ca_topic_score_gemma":0.002226854,"teacher_disagreement_score":0.0221631,"about_ca_system_score_codex":0.0029821042,"about_ca_system_score_gemma":0.009668509,"threshold_uncertainty_score":0.1172111},"labels":[],"label_agreement":null},{"id":"W4407241478","doi":"10.1145/3711903","title":"Obfuscated Clone Search in JavaScript based on Reinforcement Subsequence Learning","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Blackberry (Canada); Defence Research and Development Canada; McGill University; Queen's University","funders":"","keywords":"Computer science; JavaScript; Scripting language; Codebase; Code refactoring; Programming language; Reinforcement learning; Artificial intelligence; Automatic summarization; Source code; Code (set theory); Theoretical computer science; Software","score_opus":0.0702671741606317,"score_gpt":0.32818521897802067,"score_spread":0.25791804481738895,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407241478","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.46735638,0.0013903937,0.51761335,0.00056531525,0.00011312043,0.00018664516,0.000151685,0.009867871,0.0027551649],"genre_scores_gemma":[0.89482176,0.00012598878,0.10243202,0.00023201694,0.00003048908,0.00006415599,0.0002842991,0.00017561074,0.0018335246],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9993325,0.00016419248,0.00004090408,0.00019891218,0.000175112,0.00008839183],"domain_scores_gemma":[0.9976501,0.0012595352,0.00027668182,0.0003334929,0.00034524038,0.00013500094],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011521956,0.00084620283,0.0010632003,0.0006333886,0.00036224697,0.00043723747,0.0013259231,0.0011576959,0.00077803415],"category_scores_gemma":[0.004480189,0.00030928783,0.00056102534,0.00041856695,0.00080450246,0.0012565372,0.0008156558,0.001250961,0.0003378868],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00040170387,0.00037387968,0.006999155,0.00013329119,0.00007623716,0.00028544705,0.00016413057,0.6346679,0.016220888,0.0028004686,0.0021402773,0.33573663],"study_design_scores_gemma":[0.000011780076,0.00007798993,0.00027216255,0.000004027318,0.000007020144,0.00003491526,0.000008192811,0.9961994,0.0021780543,0.0010178447,0.00018409749,0.0000043625914],"about_ca_topic_score_codex":0.006831211,"about_ca_topic_score_gemma":0.007364699,"teacher_disagreement_score":0.006831211,"about_ca_system_score_codex":0.0010222932,"about_ca_system_score_gemma":0.0012409065,"threshold_uncertainty_score":0.013582885},"labels":[],"label_agreement":null},{"id":"W4407580437","doi":"10.1145/3717061","title":"An Empirical Study of Retrieval-Augmented Code Generation: Challenges and Opportunities","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Code (set theory); Data science; Information retrieval; Programming language","score_opus":0.20914820288916802,"score_gpt":0.39155547964417226,"score_spread":0.18240727675500423,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4407580437","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.95658207,0.003781275,0.027922008,0.0013471126,0.000153639,0.00042596567,0.0021728827,0.0016086135,0.006006331],"genre_scores_gemma":[0.96812695,0.0008456453,0.022531573,0.00036852618,0.000057254514,0.00034915557,0.0057207197,0.0004035268,0.0015967109],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9808797,0.010143108,0.0013409278,0.0026569571,0.0045164954,0.00046276787],"domain_scores_gemma":[0.785157,0.17002147,0.010213731,0.019410731,0.013425196,0.0017718404],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021476127,0.0010584434,0.00060903176,0.0021253352,0.000996602,0.0022297194,0.0018870052,0.0014627165,0.0025550772],"category_scores_gemma":[0.17397694,0.00067466294,0.00083370804,0.0022852167,0.001963749,0.008249703,0.0024328157,0.0030750628,0.0012688381],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002260276,0.002491333,0.42510787,0.0037320955,0.0005470207,0.0011239925,0.008799589,0.030044839,0.009897376,0.0075148814,0.029920008,0.47856075],"study_design_scores_gemma":[0.0007413123,0.0062406515,0.4096999,0.0019310837,0.0007746839,0.004658395,0.014853172,0.4106149,0.031851828,0.01888646,0.09926211,0.0004855858],"about_ca_topic_score_codex":0.003689805,"about_ca_topic_score_gemma":0.004335057,"teacher_disagreement_score":0.021476127,"about_ca_system_score_codex":0.001069858,"about_ca_system_score_gemma":0.0014557012,"threshold_uncertainty_score":0.11357802},"labels":[],"label_agreement":null},{"id":"W4408028155","doi":"10.1145/3721127","title":"Accountability in Code Review: The Role of Intrinsic Drivers and the Impact of LLMs","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Artificial Intelligence in Law","field":"Social Sciences","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"Aalborg Universitet","keywords":"Accountability; Computer science; Code (set theory); Political science; Programming language","score_opus":0.06464805720780835,"score_gpt":0.4021598462318052,"score_spread":0.3375117890239968,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408028155","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9051833,0.0014823753,0.023354718,0.016091393,0.00019869288,0.00029197914,0.00011055202,0.00022018922,0.053066745],"genre_scores_gemma":[0.99749076,0.00013670705,0.0015373662,0.00018440593,0.00003835684,0.000057485187,0.000013612622,0.00002841349,0.00051281985],"study_design_codex":"observational","study_design_gemma":"qualitative","domain_scores_codex":[0.8380231,0.11556838,0.0054867454,0.0063969176,0.029339299,0.0051856483],"domain_scores_gemma":[0.48993075,0.32371178,0.09359964,0.024652302,0.049858157,0.018247424],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.054921597,0.00050097273,0.00059771177,0.0035440018,0.0053300867,0.009731581,0.0015670438,0.0019783915,0.002944089],"category_scores_gemma":[0.27255237,0.00086597895,0.0007156091,0.0017486296,0.009755436,0.008612778,0.007801226,0.003216822,0.00037781923],"study_design_candidate":"qualitative","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00042539864,0.0007168446,0.6241038,0.0014819705,0.00050976005,0.00075995794,0.15695357,0.0025358982,0.004461206,0.06865945,0.0030240964,0.136368],"study_design_scores_gemma":[0.000102596256,0.0010746644,0.7636067,0.0016468832,0.00043734367,0.0010184237,0.09759081,0.0154891135,0.0068597724,0.066390626,0.045239422,0.00054351037],"about_ca_topic_score_codex":0.0054005138,"about_ca_topic_score_gemma":0.0075737895,"teacher_disagreement_score":0.054921597,"about_ca_system_score_codex":0.009595331,"about_ca_system_score_gemma":0.012529235,"threshold_uncertainty_score":0.29045665},"labels":[],"label_agreement":null},{"id":"W4408054024","doi":"10.1145/3721125","title":"Unraveling Code Clone Dynamics in Deep Learning Frameworks","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University; University of Toronto; Université du Québec à Montréal","funders":"","keywords":"Computer science; Code (set theory); Software engineering; Artificial intelligence; Data science; Programming language","score_opus":0.03839091002517494,"score_gpt":0.32376943785259454,"score_spread":0.2853785278274196,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408054024","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.98729557,0.00051537843,0.010925586,0.00025589645,0.000007378844,0.000017075201,0.00008984713,0.00031401808,0.00057926914],"genre_scores_gemma":[0.993064,0.000098761324,0.006292526,0.000045632576,0.0000043997125,0.000019204734,0.00018916017,0.000058479956,0.00022791955],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9972941,0.0006618699,0.00016535439,0.0006855628,0.0008299243,0.00036305442],"domain_scores_gemma":[0.9718847,0.014176072,0.0070490926,0.0027823672,0.0032569459,0.00085070584],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0045671156,0.0004445414,0.00043392842,0.0030204481,0.0006622813,0.001526167,0.0011812065,0.0007431563,0.0004457586],"category_scores_gemma":[0.040502697,0.0003634454,0.00047388574,0.0018757641,0.0017299923,0.0039462578,0.001686953,0.001324553,0.000114804345],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00027370747,0.00025172537,0.77916527,0.00020349058,0.00017752252,0.00071214227,0.0051459707,0.044888817,0.007687228,0.006488162,0.0014714693,0.15353449],"study_design_scores_gemma":[0.000035651396,0.00027990656,0.35597098,0.00012185644,0.00013580605,0.0007790343,0.0024781493,0.6141773,0.0061908187,0.014515616,0.005236694,0.00007812979],"about_ca_topic_score_codex":0.010689818,"about_ca_topic_score_gemma":0.014433778,"teacher_disagreement_score":0.010689818,"about_ca_system_score_codex":0.0017663075,"about_ca_system_score_gemma":0.001342619,"threshold_uncertainty_score":0.02415353},"labels":[],"label_agreement":null},{"id":"W4408473356","doi":"10.1145/3723355","title":"Enhancing Log Sentiments: An Exploratory Study of Sentiments and Emotions with Software Logs","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"National Natural Science Foundation of China","keywords":"Computer science; Exploratory research; Data science; Software; World Wide Web; Programming language; Sociology","score_opus":0.03968515645815301,"score_gpt":0.30287675622678417,"score_spread":0.26319159976863116,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408473356","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99509734,0.00008369719,0.0020411855,0.00019393,0.000014992488,0.00009175503,0.0012529347,0.000100469195,0.0011237335],"genre_scores_gemma":[0.9894432,0.0001378668,0.005780547,0.00024893245,0.000038864087,0.00030052653,0.0028221833,0.000072150106,0.0011557417],"study_design_codex":"observational","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99853885,0.00071957614,0.00007862197,0.00017251259,0.0003646558,0.0001258564],"domain_scores_gemma":[0.9888208,0.0075672767,0.0014898309,0.0005059486,0.0011801616,0.0004358789],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017164806,0.0003443166,0.000326274,0.0012148353,0.00069400715,0.0013262894,0.00039357613,0.00053855637,0.0006675928],"category_scores_gemma":[0.010252211,0.00015312739,0.00024028773,0.0011347024,0.0005876592,0.0017649141,0.0008947162,0.0009895741,0.0004059608],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.000991669,0.0022831839,0.70480376,0.0010988602,0.00013212749,0.002421416,0.12001685,0.0014625216,0.03462532,0.0018255631,0.0199302,0.11040855],"study_design_scores_gemma":[0.000056029734,0.0008422988,0.8280163,0.0002708265,0.00007174286,0.0011756305,0.10077038,0.020770654,0.009388618,0.0025596092,0.035929114,0.00014876954],"about_ca_topic_score_codex":0.00144289,"about_ca_topic_score_gemma":0.0038176952,"teacher_disagreement_score":0.0017164806,"about_ca_system_score_codex":0.0004578391,"about_ca_system_score_gemma":0.00037463004,"threshold_uncertainty_score":0.009077728},"labels":[],"label_agreement":null},{"id":"W4408797770","doi":"10.1145/3716822","title":"An empirical study on vulnerability disclosure management of open source software systems","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Information and Cyber Security","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"","keywords":"Computer science; Vulnerability (computing); Vulnerability management; Open source; Software; Open source software; Empirical research; Computer security; Vulnerability assessment; Operating system","score_opus":0.07854885103263813,"score_gpt":0.37620784924822176,"score_spread":0.2976589982155836,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408797770","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9971205,0.00014660647,0.00069940137,0.00044499073,0.0000046868945,0.00004708931,0.000045005025,0.0000062272134,0.0014855023],"genre_scores_gemma":[0.99850535,0.00019909507,0.00075670844,0.000105222156,0.000008354746,0.00004992447,0.00006649093,0.000005257005,0.00030362429],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9765667,0.012028348,0.0024124898,0.0015536164,0.00590408,0.0015349218],"domain_scores_gemma":[0.6389281,0.24606323,0.07655386,0.008590672,0.023537364,0.0063267844],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.021238975,0.00026238864,0.0003001959,0.0029102322,0.0019422567,0.002712273,0.0009513494,0.001000097,0.0018048682],"category_scores_gemma":[0.14360324,0.0003430175,0.00026999993,0.0033711193,0.0022633236,0.0067420644,0.002685453,0.0021298402,0.0002978593],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00014783963,0.0008306828,0.83123416,0.0004399633,0.000042131316,0.0007450775,0.11099843,0.0004929186,0.0011915447,0.002854449,0.0012356971,0.049787242],"study_design_scores_gemma":[0.00003329927,0.0008858732,0.6985892,0.00076284906,0.00005525989,0.0013849678,0.2740823,0.005299435,0.0017241563,0.002372148,0.014714355,0.00009617608],"about_ca_topic_score_codex":0.0029470187,"about_ca_topic_score_gemma":0.0034146835,"teacher_disagreement_score":0.021238975,"about_ca_system_score_codex":0.0021098203,"about_ca_system_score_gemma":0.0031502394,"threshold_uncertainty_score":0.11232376},"labels":[],"label_agreement":null},{"id":"W4408851891","doi":"10.1145/3725212","title":"Towards On-the-Fly Code Performance Profiling","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Huawei Technologies (Canada)","funders":"National Key Research and Development Program of China","keywords":"Computer science; Profiling (computer programming); On the fly; Programming language; Software engineering; Operating system","score_opus":0.088324649823905,"score_gpt":0.3216710607829046,"score_spread":0.23334641095899958,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4408851891","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.10716114,0.00086750166,0.823279,0.0006251524,0.00008616327,0.0001082478,0.0015733875,0.062389657,0.003909704],"genre_scores_gemma":[0.6123236,0.00050785893,0.37432718,0.00037896066,0.000062738734,0.0001389792,0.005885637,0.0032513726,0.003123648],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9987457,0.00020923755,0.000050397593,0.00042968863,0.0004327357,0.00013217745],"domain_scores_gemma":[0.9966865,0.00096678507,0.00048749434,0.0008716003,0.00083991664,0.00014773906],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007441768,0.0020095492,0.00068140845,0.0023475238,0.0003306542,0.0011514265,0.0015641594,0.0009985013,0.0012639222],"category_scores_gemma":[0.0074222647,0.00071880035,0.0006798364,0.0010854339,0.00040319742,0.0027112672,0.0011485147,0.0016799659,0.002374214],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00037796778,0.0005190811,0.034469154,0.00026452844,0.00009440934,0.0002743736,0.00023676902,0.16975941,0.057150666,0.0038308455,0.017460683,0.71556216],"study_design_scores_gemma":[0.000009049394,0.00005604814,0.0027951524,0.000019863315,0.000012486175,0.00006374733,0.000029106937,0.97517395,0.014270946,0.0043638595,0.0031871663,0.000018595052],"about_ca_topic_score_codex":0.0044529755,"about_ca_topic_score_gemma":0.0072434847,"teacher_disagreement_score":0.0044529755,"about_ca_system_score_codex":0.00066220306,"about_ca_system_score_gemma":0.0013800313,"threshold_uncertainty_score":0.008854091},"labels":[],"label_agreement":null},{"id":"W4409478867","doi":"10.1145/3729533","title":"Evaluating API-Level Deep Learning Fuzzers: A Comprehensive Benchmarking Study","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Anomaly Detection Techniques and Applications","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"","keywords":"Computer science; Benchmarking; Deep learning; Artificial intelligence; Software engineering; Machine learning; Data science","score_opus":0.16246850762860587,"score_gpt":0.392204384378163,"score_spread":0.22973587674955714,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4409478867","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.88584137,0.016197816,0.048252303,0.0019424473,0.00070004474,0.0005478522,0.011902247,0.025483115,0.009132816],"genre_scores_gemma":[0.8662659,0.0022198863,0.08192795,0.0012571409,0.00012014481,0.00033785414,0.044244155,0.001033261,0.0025937387],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9917892,0.0021030472,0.00083922246,0.001982224,0.002764398,0.0005219729],"domain_scores_gemma":[0.9831149,0.008782233,0.0012487732,0.0033358466,0.0029701102,0.00054813974],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.007018362,0.00308005,0.0010529362,0.0036782154,0.0007700109,0.0014484739,0.0043864036,0.0022340722,0.001452719],"category_scores_gemma":[0.03009481,0.0006796612,0.0014439883,0.0024986418,0.0018560579,0.004100663,0.0024358043,0.0023318944,0.0008554612],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0022799226,0.002254241,0.10265206,0.0047243116,0.0016814743,0.00083658966,0.00052482716,0.36290702,0.019521717,0.006069791,0.07874124,0.41780692],"study_design_scores_gemma":[0.00047919416,0.0019764525,0.030112078,0.0005193504,0.00038805767,0.0009117238,0.00033592258,0.89469916,0.04193434,0.0059063374,0.02257733,0.00016020102],"about_ca_topic_score_codex":0.014145503,"about_ca_topic_score_gemma":0.020008087,"teacher_disagreement_score":0.014145503,"about_ca_system_score_codex":0.0031183048,"about_ca_system_score_gemma":0.002320378,"threshold_uncertainty_score":0.037117064},"labels":[],"label_agreement":null},{"id":"W4410028485","doi":"10.1145/3733717","title":"Learning-Guided Fuzzing for Testing Stateful SDN Controllers","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Smart Grid Security and Resilience","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Fuzz testing; Stateful firewall; Computer science; Computer security; Programming language; Software","score_opus":0.0684615723814985,"score_gpt":0.3068296410605721,"score_spread":0.2383680686790736,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410028485","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.14151107,0.00041825115,0.85267895,0.0004134257,0.0000435686,0.00015316122,0.00016202895,0.0036669655,0.0009525377],"genre_scores_gemma":[0.8400962,0.00009111228,0.15859176,0.00021657365,0.000018617053,0.0001301809,0.00027296576,0.00014627034,0.00043630786],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9973889,0.0009529415,0.00017070513,0.00057047105,0.00071932486,0.00019759317],"domain_scores_gemma":[0.9812122,0.015043436,0.00095567055,0.0014415291,0.001037948,0.00030916234],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0026056392,0.0013273978,0.00081575493,0.0011165491,0.00046965916,0.0008065266,0.0018519097,0.001350612,0.0012473814],"category_scores_gemma":[0.021640044,0.00059812353,0.001213091,0.00038803078,0.0018364543,0.0021537377,0.0015478772,0.0018953246,0.00014919128],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028325422,0.0001723403,0.008070155,0.00017168019,0.00009632149,0.0001284619,0.0001972427,0.88781416,0.007714885,0.0063572857,0.000570815,0.08842341],"study_design_scores_gemma":[0.000011565615,0.00003314118,0.00020939659,0.0000079422025,0.000007407947,0.000019411329,0.000010105687,0.99371666,0.0019697216,0.00390016,0.00010962064,0.0000049602822],"about_ca_topic_score_codex":0.005911865,"about_ca_topic_score_gemma":0.008197906,"teacher_disagreement_score":0.005911865,"about_ca_system_score_codex":0.0015199273,"about_ca_system_score_gemma":0.0020901135,"threshold_uncertainty_score":0.013780057},"labels":[],"label_agreement":null},{"id":"W4410089993","doi":"10.1145/3733715","title":"Stress Testing Control Loops in Cyber-Physical Systems—RCR Report","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Embedded Systems Design Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"Knut och Alice Wallenbergs Stiftelse","keywords":"Computer science; Cyber-physical system; Stress testing (software); Reliability engineering; Programming language; Operating system","score_opus":0.06220045181420397,"score_gpt":0.3230771558298215,"score_spread":0.2608767040156175,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410089993","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.056925945,0.0014462018,0.8832947,0.012261051,0.0036268784,0.0006616791,0.006570861,0.005571868,0.029640896],"genre_scores_gemma":[0.6655913,0.0017485833,0.30181524,0.0019331588,0.0013889313,0.0016663068,0.012267743,0.0033227643,0.010265975],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9794263,0.008542728,0.0013121647,0.0017275292,0.008154104,0.0008372919],"domain_scores_gemma":[0.8820337,0.054044917,0.003917204,0.036815923,0.021409463,0.0017788381],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.017476365,0.0011301452,0.0010197914,0.0014346349,0.0007312922,0.0023503297,0.0033738425,0.0013882318,0.0141103435],"category_scores_gemma":[0.08371857,0.0006029742,0.0019945654,0.0011332338,0.0020435182,0.0037183347,0.002163349,0.003464663,0.003894595],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0010599022,0.0015654126,0.011533561,0.000998616,0.00059543364,0.0006313286,0.0006312383,0.29909596,0.020660752,0.21083775,0.20207743,0.25031266],"study_design_scores_gemma":[0.00038629342,0.0012733681,0.009348709,0.00038857097,0.0001758449,0.0005784883,0.0005180328,0.7256213,0.05774686,0.12325517,0.08042236,0.0002849826],"about_ca_topic_score_codex":0.008568755,"about_ca_topic_score_gemma":0.003320636,"teacher_disagreement_score":0.017476365,"about_ca_system_score_codex":0.0024994547,"about_ca_system_score_gemma":0.0033014133,"threshold_uncertainty_score":0.09242493},"labels":[],"label_agreement":null},{"id":"W4410344404","doi":"10.1145/3735553","title":"Large Language Models for Automated Web-Form-Test Generation: An Empirical Study","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Test (biology); Software engineering; Web application; Natural language processing; Programming language; World Wide Web; Information retrieval","score_opus":0.10826609347819456,"score_gpt":0.39020246290793753,"score_spread":0.281936369429743,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410344404","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9754132,0.0009413905,0.018625537,0.00039686236,0.000042684103,0.0005138186,0.0010537676,0.001078884,0.0019338917],"genre_scores_gemma":[0.96631205,0.00035116333,0.028575733,0.00016178357,0.000034968438,0.0004462553,0.003326062,0.00024307307,0.0005489484],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.97970325,0.013672685,0.0015014908,0.0017710034,0.0028767241,0.00047484922],"domain_scores_gemma":[0.66264033,0.306694,0.007829366,0.012152107,0.008890552,0.0017936482],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02084597,0.0014197938,0.0009199445,0.0018226053,0.00074057933,0.0025519584,0.0020138877,0.001418776,0.002690488],"category_scores_gemma":[0.1549455,0.00079020363,0.0013564496,0.001959577,0.0013777647,0.005385495,0.0019468053,0.0032098119,0.0012510248],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004241603,0.012302609,0.28104794,0.0035507276,0.0006507839,0.0014012964,0.0074223545,0.15701818,0.007697104,0.0049620466,0.016910246,0.5027951],"study_design_scores_gemma":[0.00052588666,0.0021042898,0.06903405,0.0003881955,0.00051731296,0.00068277994,0.0028269272,0.90572435,0.007264513,0.0030777035,0.0076934434,0.00016052775],"about_ca_topic_score_codex":0.009740454,"about_ca_topic_score_gemma":0.0078726495,"teacher_disagreement_score":0.02084597,"about_ca_system_score_codex":0.0023118793,"about_ca_system_score_gemma":0.0021413022,"threshold_uncertainty_score":0.11024529},"labels":[],"label_agreement":null},{"id":"W4410537502","doi":"10.1145/3736407","title":"CodeUltraFeedback: An LLM-as-a-Judge Dataset for Aligning Large Language Models to Coding Preferences","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Computer science; Coding (social sciences); Natural language processing; Artificial intelligence; Programming language; Statistics; Mathematics","score_opus":0.0823360169353762,"score_gpt":0.36758500400909394,"score_spread":0.2852489870737177,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410537502","genre_codex":"empirical","genre_gemma":"dataset","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"dataset","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.407223,0.004532687,0.14539485,0.0040328857,0.0023728414,0.003202879,0.2473659,0.15245856,0.03341642],"genre_scores_gemma":[0.33622497,0.0004527857,0.12730294,0.002258604,0.00021620486,0.0033737037,0.51091605,0.006567552,0.012687123],"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9922357,0.003972553,0.0005622721,0.0014568007,0.0014009301,0.0003716477],"domain_scores_gemma":[0.9853184,0.008112648,0.0006775443,0.0024428351,0.002658139,0.0007904847],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004622301,0.0032753432,0.0008615781,0.0019060731,0.0012079363,0.0018025776,0.0024735653,0.0029161354,0.0074150073],"category_scores_gemma":[0.030730693,0.0005495879,0.0013105334,0.0013932728,0.0011109551,0.0027889672,0.002731407,0.0035098784,0.008638643],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.002893943,0.0018646268,0.026920449,0.0032144694,0.00051393756,0.0011260307,0.00279384,0.041786443,0.03892606,0.005330113,0.5992308,0.27539936],"study_design_scores_gemma":[0.001701753,0.0025344295,0.041017644,0.0005626281,0.00024969495,0.0014762162,0.0028367285,0.57031256,0.05678223,0.015271726,0.30654952,0.00070484425],"about_ca_topic_score_codex":0.010693455,"about_ca_topic_score_gemma":0.029158758,"teacher_disagreement_score":0.010693455,"about_ca_system_score_codex":0.0016142774,"about_ca_system_score_gemma":0.0020728053,"threshold_uncertainty_score":0.024805605},"labels":[],"label_agreement":null},{"id":"W4410610020","doi":"10.1145/3735635","title":"Do Current Language Models Support Code Intelligence for R Programming Language?","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Data Mining Algorithms and Applications","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Computer science; Programming language; Current (fluid); First-generation programming language; Programming paradigm","score_opus":0.09567141734564485,"score_gpt":0.37935461176167246,"score_spread":0.28368319441602763,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410610020","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.072939575,0.007815368,0.7034459,0.012629806,0.0014513546,0.00049972406,0.018248448,0.16730689,0.015662942],"genre_scores_gemma":[0.3224582,0.0040821033,0.5804629,0.006999962,0.0005595901,0.0011729944,0.057299268,0.018724952,0.008240093],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.98989457,0.004841242,0.00067374855,0.0028169404,0.0013651992,0.00040827927],"domain_scores_gemma":[0.9589766,0.024269307,0.0023166852,0.0092684645,0.004369041,0.00079986535],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011167321,0.0024287596,0.0013539528,0.0019259262,0.00061059615,0.0045305477,0.004706239,0.0019954266,0.0051963343],"category_scores_gemma":[0.07583054,0.0012017819,0.0027300592,0.0022039446,0.0013098832,0.008976677,0.0026763144,0.004779557,0.012608322],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015228937,0.00051566045,0.026586741,0.003757023,0.0010388779,0.00037206325,0.001272316,0.103347786,0.010578712,0.02433263,0.21613291,0.61054236],"study_design_scores_gemma":[0.00029948165,0.00036213046,0.0044401004,0.000599706,0.0003168162,0.00048696855,0.00032761015,0.8213017,0.009557451,0.043565497,0.11855236,0.00019006334],"about_ca_topic_score_codex":0.0067509026,"about_ca_topic_score_gemma":0.01255785,"teacher_disagreement_score":0.011167321,"about_ca_system_score_codex":0.001311418,"about_ca_system_score_gemma":0.003919497,"threshold_uncertainty_score":0.059059143},"labels":[],"label_agreement":null},{"id":"W4410644286","doi":"10.1145/3736758","title":"CI/CD Configuration Practices in Open Source Android Apps: An Empirical Study","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Advanced Malware Detection Techniques","field":"Computer Science","cited_by":9,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Trent University","funders":"","keywords":"Computer science; Android (operating system); Open source; Empirical research; Open source software; World Wide Web; Operating system; Software","score_opus":0.12759613669649247,"score_gpt":0.42595277183427765,"score_spread":0.2983566351377852,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410644286","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.99781597,0.00018512817,0.0005077751,0.00026112338,0.000005751761,0.00008879379,0.000044045955,0.000013131855,0.0010782538],"genre_scores_gemma":[0.99764,0.00034239638,0.0011764619,0.00014947211,0.000008132654,0.00012297503,0.000061188686,0.000018328443,0.00048100456],"study_design_codex":"qualitative","study_design_gemma":"observational","domain_scores_codex":[0.9869985,0.005285017,0.0012978358,0.0013348211,0.004121339,0.00096246344],"domain_scores_gemma":[0.8839195,0.07614893,0.020237852,0.0043331613,0.0117201395,0.0036404312],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.014215207,0.00037324795,0.00042939425,0.0025811226,0.0016987991,0.0036228143,0.0015147302,0.0011158866,0.0014559126],"category_scores_gemma":[0.077222,0.0007571919,0.00034064052,0.0019829227,0.0028166308,0.004905611,0.002910603,0.002021043,0.00033939272],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00013855379,0.00091525255,0.40292776,0.0008462354,0.000049870472,0.0016339608,0.5189549,0.00017939789,0.002380937,0.0014647203,0.0014160189,0.06909243],"study_design_scores_gemma":[0.000022417935,0.00054793473,0.4406226,0.0009566126,0.00005491355,0.0015470906,0.53896755,0.0019177959,0.0009248863,0.0005721605,0.013774657,0.00009129842],"about_ca_topic_score_codex":0.0053210007,"about_ca_topic_score_gemma":0.008688668,"teacher_disagreement_score":0.98578477,"about_ca_system_score_codex":0.0020535258,"about_ca_system_score_gemma":0.0030742441,"threshold_uncertainty_score":0.07517809},"labels":[],"label_agreement":null},{"id":"W4410771641","doi":"10.1145/3736405","title":"Improving Code Reviewer Recommendation: Accuracy, Latency, Workload, and Bystanders","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"","keywords":"Computer science; Workload; Code (set theory); Latency (audio); Operating system; Telecommunications; Programming language","score_opus":0.07592995547935187,"score_gpt":0.34187308048692294,"score_spread":0.26594312500757106,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4410771641","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.777763,0.020869741,0.1025967,0.012106114,0.0044838353,0.05863597,0.004866925,0.0042659203,0.014411741],"genre_scores_gemma":[0.8156888,0.0021441944,0.12221518,0.0035422407,0.0011993482,0.050485868,0.0011259139,0.00035243377,0.003246006],"study_design_codex":"design_other","study_design_gemma":"observational","domain_scores_codex":[0.8089734,0.13829252,0.02031814,0.009471051,0.020945083,0.0019998176],"domain_scores_gemma":[0.3070288,0.5809882,0.050146967,0.029349271,0.027425367,0.0050613466],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.16866674,0.0014578751,0.0023266077,0.0018592115,0.0014210037,0.0031819784,0.002646645,0.0036766604,0.0059917555],"category_scores_gemma":[0.47928455,0.0011046118,0.0034530198,0.0017129154,0.0020354302,0.004657249,0.0019024414,0.0033973372,0.0023069794],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.1477088,0.024839044,0.053415064,0.025378067,0.008585053,0.00018486273,0.0025294742,0.008092523,0.009078204,0.0025954756,0.017614447,0.699979],"study_design_scores_gemma":[0.14496918,0.5530471,0.12469719,0.0075039677,0.019370629,0.00096452556,0.0018286635,0.04792901,0.041820057,0.015069014,0.041543074,0.0012575185],"about_ca_topic_score_codex":0.0015005564,"about_ca_topic_score_gemma":0.0021938924,"teacher_disagreement_score":0.16866674,"about_ca_system_score_codex":0.0022541347,"about_ca_system_score_gemma":0.005142894,"threshold_uncertainty_score":0.8920056},"labels":[],"label_agreement":null},{"id":"W4411087888","doi":"10.1145/3742894","title":"The Havoc Paradox in Generator-Based Fuzzing","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Fuzz testing; Computer science; Generator (circuit theory); Programming language; Software; Power (physics)","score_opus":0.0603762164017855,"score_gpt":0.3174931715724063,"score_spread":0.25711695517062083,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411087888","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.43518162,0.0007122447,0.5546521,0.00071311323,0.000049828337,0.00012345928,0.0000872194,0.005007104,0.0034734039],"genre_scores_gemma":[0.9102899,0.000118204465,0.08817828,0.00028046008,0.000012161829,0.000055245197,0.00009142952,0.00036175447,0.00061259663],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9933089,0.0025712051,0.00037492512,0.00096104614,0.0024169215,0.0003669973],"domain_scores_gemma":[0.9726977,0.018838016,0.0017898956,0.0049455184,0.0014013009,0.00032762048],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0064800642,0.00070715597,0.0007472973,0.0014283498,0.0006045751,0.0013437523,0.0016957483,0.001208624,0.000853636],"category_scores_gemma":[0.03470755,0.00059378706,0.00065591856,0.0006592985,0.0031288199,0.0033140604,0.0019332258,0.0017837307,0.00017899025],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015841769,0.0005181816,0.03721983,0.00084037485,0.00035787054,0.0016587693,0.0022876614,0.35094917,0.12263222,0.09930183,0.002845074,0.3798048],"study_design_scores_gemma":[0.00012475278,0.00076244975,0.0042147837,0.0001076178,0.00014200051,0.0011857409,0.0002049688,0.8167184,0.100768715,0.071712814,0.003960565,0.00009711801],"about_ca_topic_score_codex":0.0016231107,"about_ca_topic_score_gemma":0.0024276911,"teacher_disagreement_score":0.0064800642,"about_ca_system_score_codex":0.0009357057,"about_ca_system_score_gemma":0.0014237794,"threshold_uncertainty_score":0.034270287},"labels":[],"label_agreement":null},{"id":"W4411236083","doi":"10.1145/3744644","title":"LLM-Cure: LLM-Based Competitor User Review Analysis for Feature Enhancement","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Toronto; Queen's University; Université du Québec à Montréal","funders":"","keywords":"Computer science; Feature (linguistics); Artificial intelligence","score_opus":0.046042394086885305,"score_gpt":0.3511223933018579,"score_spread":0.3050799992149726,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411236083","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.18613519,0.007776779,0.6033106,0.0019767147,0.00093055225,0.0037184746,0.027516574,0.1573546,0.011280518],"genre_scores_gemma":[0.39253986,0.0008433176,0.55871063,0.0009345908,0.00044435667,0.0014781277,0.032530155,0.002025878,0.010493102],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99326694,0.0021990328,0.00059084484,0.001392854,0.0023126993,0.0002375773],"domain_scores_gemma":[0.9838418,0.0070839706,0.0023809373,0.0014023995,0.004749075,0.00054189836],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0047212476,0.002401,0.001333626,0.009213857,0.0006525779,0.0018282013,0.0018363126,0.001332357,0.0029368722],"category_scores_gemma":[0.02209772,0.0004958334,0.0018354154,0.0025316903,0.00039140193,0.002151292,0.0015920231,0.0012073922,0.004207655],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011922916,0.0010070722,0.044489413,0.003266103,0.0006790597,0.0011964041,0.001399245,0.012186871,0.04146761,0.0030556463,0.124237806,0.76582247],"study_design_scores_gemma":[0.00027250015,0.0009761006,0.02616002,0.00019091077,0.00037807252,0.0012839524,0.00064499635,0.8766328,0.03205218,0.0043252124,0.056808356,0.0002749398],"about_ca_topic_score_codex":0.006271653,"about_ca_topic_score_gemma":0.017006157,"teacher_disagreement_score":0.009213857,"about_ca_system_score_codex":0.0011658698,"about_ca_system_score_gemma":0.0022639355,"threshold_uncertainty_score":0.024968684},"labels":[],"label_agreement":null},{"id":"W4411488587","doi":"10.1145/3744920","title":"On the Utility of Domain Modeling Assistance with Large Language Models","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"McMaster University; Université de Montréal","funders":"","keywords":"Computer science; Modeling language; Domain (mathematical analysis); Usability; Software engineering; Domain analysis; Abstraction; Domain-specific language; Process (computing); Model-driven architecture; Context (archaeology); Subject-matter expert; Domain model; Software; Domain engineering; Human–computer interaction; Software development; Data science; Artificial intelligence; Domain knowledge; Programming language; Component-based software engineering; Software construction","score_opus":0.06253040734009754,"score_gpt":0.3178245270606738,"score_spread":0.25529411972057625,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4411488587","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.45064798,0.0013483646,0.5214788,0.0023520573,0.00010345089,0.000563958,0.0005417783,0.011250888,0.011712744],"genre_scores_gemma":[0.7524036,0.00040475538,0.24437149,0.00026597327,0.000025179295,0.00013744067,0.00044805047,0.0004936112,0.0014497867],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.9906313,0.006776527,0.00031402457,0.0009797461,0.0011348985,0.0001635427],"domain_scores_gemma":[0.8425185,0.1450806,0.0014694278,0.0066917525,0.0035120754,0.0007277401],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011555801,0.0013556592,0.0005168503,0.0013824096,0.00085046847,0.0025854092,0.0018226053,0.001993002,0.0031510289],"category_scores_gemma":[0.099085785,0.00064783957,0.0005574376,0.00088639295,0.0011099064,0.007871268,0.0028971694,0.0020349913,0.0009415102],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0028924793,0.002324549,0.015331045,0.0013024153,0.00027184005,0.0006448142,0.005812002,0.16960584,0.033127345,0.014728813,0.0071411985,0.74681777],"study_design_scores_gemma":[0.00015648104,0.0010008528,0.0032573417,0.00017650609,0.00010842248,0.00031731068,0.001188253,0.9523172,0.020741228,0.012933552,0.007708208,0.000094733674],"about_ca_topic_score_codex":0.0069232886,"about_ca_topic_score_gemma":0.008747774,"teacher_disagreement_score":0.011555801,"about_ca_system_score_codex":0.0009474091,"about_ca_system_score_gemma":0.0014880926,"threshold_uncertainty_score":0.061113656},"labels":[],"label_agreement":null},{"id":"W4412033360","doi":"10.1145/3747289","title":"Towards Understanding Refactoring Engine Bugs","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Concordia University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Code refactoring; Computer science; Software engineering; Programming language; Software","score_opus":0.1393592182729809,"score_gpt":0.35114929550490953,"score_spread":0.21179007723192864,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412033360","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6945721,0.010366413,0.28270212,0.0029608104,0.0001880183,0.0007966926,0.00179542,0.0024385096,0.004179936],"genre_scores_gemma":[0.6280869,0.0044283527,0.36142808,0.00097165385,0.00012677711,0.00036299968,0.0024827542,0.00066534115,0.0014470747],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98829275,0.002196435,0.0019588892,0.0023473313,0.004401916,0.000802729],"domain_scores_gemma":[0.91173893,0.042713024,0.020508911,0.0070188614,0.01715204,0.00086828636],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.009957315,0.0020744433,0.0011150754,0.016714716,0.0012228668,0.0038483685,0.0026193922,0.0024647159,0.00097651547],"category_scores_gemma":[0.05783904,0.0011695626,0.0014947853,0.0053016106,0.0019954701,0.0108021675,0.0025798217,0.0026261278,0.00037527917],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025833596,0.0007576402,0.51327986,0.0036948689,0.00032875777,0.0035324234,0.024253055,0.0076661417,0.033870555,0.0134589,0.0032197512,0.39567968],"study_design_scores_gemma":[0.00021266192,0.0016520261,0.6461844,0.0052340925,0.0017689752,0.015066074,0.028654069,0.11480818,0.07111798,0.059866786,0.054769427,0.00066546747],"about_ca_topic_score_codex":0.0065541137,"about_ca_topic_score_gemma":0.0071457117,"teacher_disagreement_score":0.016714716,"about_ca_system_score_codex":0.0014112251,"about_ca_system_score_gemma":0.003249174,"threshold_uncertainty_score":0.05265993},"labels":[],"label_agreement":null},{"id":"W4412070868","doi":"10.1145/3747347","title":"Understanding Open Source Contributor Profiles in Popular Machine Learning Libraries","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Open Source Software Innovations","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Queen's University","funders":"","keywords":"Computer science; Open source; Data science; Artificial intelligence; World Wide Web; Machine learning; Programming language; Software","score_opus":0.13514193957706988,"score_gpt":0.3225430056727442,"score_spread":0.1874010660956743,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412070868","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9954963,0.00019758627,0.0017128614,0.00016341345,0.000008328813,0.00002284112,0.00023870986,0.00009389443,0.0020660677],"genre_scores_gemma":[0.99469984,0.00020464894,0.0015306922,0.000057479047,0.00002206605,0.000053274536,0.00069047883,0.00007657397,0.0026648804],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9971084,0.0007609985,0.00025468462,0.00056834455,0.0008998458,0.00040768486],"domain_scores_gemma":[0.9754215,0.008164405,0.0078478055,0.0015093817,0.004527007,0.0025299238],"candidate_categories":["open_science"],"consensus_categories":[],"category_scores_codex":[0.0031350458,0.0003237777,0.00036180473,0.0067707803,0.0013891727,0.0032494057,0.00065147365,0.0004660984,0.0032957303],"category_scores_gemma":[0.023729619,0.0002547102,0.00026576643,0.00509007,0.0005495465,0.0044517047,0.0032409222,0.00042213508,0.001767878],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00024821225,0.00016936181,0.8874666,0.00016605358,0.000037014266,0.00034768254,0.02593599,0.00033489056,0.002308911,0.00070632185,0.0023239604,0.07995488],"study_design_scores_gemma":[0.000013252971,0.00015347164,0.94298387,0.00016377796,0.000045001714,0.0006472808,0.033726726,0.0058617704,0.0017577321,0.0013419178,0.013230942,0.000074185235],"about_ca_topic_score_codex":0.0019318803,"about_ca_topic_score_gemma":0.0041634734,"teacher_disagreement_score":0.9993485,"about_ca_system_score_codex":0.00066339754,"about_ca_system_score_gemma":0.00071851676,"threshold_uncertainty_score":0.016579866},"labels":[],"label_agreement":null},{"id":"W4412565297","doi":"10.1145/3749839","title":"Tracing Optimization for Performance Modeling and Regression Detection","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Computer science; Tracing; Programming language","score_opus":0.04749648754092181,"score_gpt":0.30007565846496576,"score_spread":0.2525791709240439,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412565297","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.034142695,0.00022306253,0.9530251,0.00019461816,0.000027093838,0.0000840934,0.000373918,0.01079438,0.001135058],"genre_scores_gemma":[0.45177373,0.0002753866,0.54005396,0.00016561571,0.000056042114,0.0003507359,0.002225395,0.0023454383,0.0027536533],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9970067,0.0008382608,0.00021986345,0.00070364826,0.00092507584,0.00030644634],"domain_scores_gemma":[0.9903025,0.005573344,0.0013053847,0.001557601,0.001119759,0.0001413759],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0037949083,0.002719838,0.0016059246,0.002745932,0.000754428,0.0017579308,0.0020837698,0.0012908562,0.0025652447],"category_scores_gemma":[0.021501243,0.0011451413,0.0017112064,0.0020548971,0.0007478434,0.0019596908,0.0016952463,0.0023955922,0.0012066925],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00018348225,0.0002574405,0.014931654,0.00012513799,0.00015912157,0.00012416001,0.0001454489,0.84154576,0.009407169,0.0072211996,0.002875666,0.123023756],"study_design_scores_gemma":[0.000005608791,0.000019689902,0.00046567124,0.0000057716825,0.000010911589,0.000017295317,0.000007867031,0.9949515,0.0021577952,0.0017447091,0.00060501107,0.000008226499],"about_ca_topic_score_codex":0.014165833,"about_ca_topic_score_gemma":0.011214383,"teacher_disagreement_score":0.014165833,"about_ca_system_score_codex":0.0014173939,"about_ca_system_score_gemma":0.0033129754,"threshold_uncertainty_score":0.028166711},"labels":[],"label_agreement":null},{"id":"W4412602677","doi":"10.1145/3749100","title":"<scp>AntiCopyPaster</scp> 3.0: Just-in-Time Clone Refactoring","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal","funders":"","keywords":"Code refactoring; Computer science; clone (Java method); Programming language; Software engineering; Software; Biology","score_opus":0.06483042430187017,"score_gpt":0.328490238395962,"score_spread":0.2636598140940918,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412602677","genre_codex":"software","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0035675496,0.00023099309,0.15256935,0.0010198529,0.00033104362,0.0003785255,0.01191452,0.8219044,0.008083736],"genre_scores_gemma":[0.073038764,0.00066570676,0.45715266,0.0035291265,0.00046438316,0.001729306,0.08915731,0.3364901,0.037772577],"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9947476,0.0007365485,0.00045567422,0.00095342187,0.0027846268,0.00032214943],"domain_scores_gemma":[0.9749341,0.008711547,0.0019534973,0.008686873,0.0047656456,0.00094835093],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004940955,0.0025532052,0.001048839,0.0030582498,0.0010569904,0.0027049824,0.0056706094,0.002675771,0.05724053],"category_scores_gemma":[0.029602423,0.002219395,0.0016375872,0.0027460952,0.0016084342,0.005712482,0.0047054575,0.0039203316,0.03906807],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008017121,0.00016940961,0.0030236016,0.0009711004,0.00012536449,0.0007683147,0.0006957206,0.0023084648,0.017160947,0.0041002952,0.80048484,0.16939011],"study_design_scores_gemma":[0.00073982985,0.0006290491,0.012410005,0.00051358726,0.00013154813,0.002433758,0.00017413174,0.093680285,0.08209113,0.017232865,0.78944814,0.0005157567],"about_ca_topic_score_codex":0.0085230125,"about_ca_topic_score_gemma":0.014602606,"teacher_disagreement_score":0.05724053,"about_ca_system_score_codex":0.0015698599,"about_ca_system_score_gemma":0.002866233,"threshold_uncertainty_score":0.19148862},"labels":[],"label_agreement":null},{"id":"W4412827352","doi":"10.1145/3744340","title":"AcTracer: Active Testing of Large Language Model via Multi-Stage Sampling","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Topic Modeling","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Alberta","funders":"","keywords":"Computer science; Sampling (signal processing)","score_opus":0.15140591138260573,"score_gpt":0.37414789205765875,"score_spread":0.22274198067505302,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4412827352","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.064208195,0.0013472639,0.9052391,0.0005872447,0.00026543194,0.0006600775,0.0006934571,0.02486438,0.0021348344],"genre_scores_gemma":[0.57811195,0.00038033532,0.4078541,0.0014972511,0.00021184691,0.0016934989,0.003951614,0.002447169,0.0038522603],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99015146,0.005610242,0.000496982,0.0017231375,0.0015311949,0.0004870531],"domain_scores_gemma":[0.96753657,0.02500265,0.00077810895,0.0034258778,0.0024920995,0.0007646354],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.011749129,0.0035591677,0.0024953613,0.0013515328,0.00099207,0.0022773047,0.006859713,0.0025052489,0.0037476954],"category_scores_gemma":[0.034601208,0.0011206681,0.0021141658,0.00075698306,0.0016149215,0.0060859127,0.004505344,0.0044452786,0.002830525],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0025666906,0.0014441628,0.013630121,0.00090796745,0.0008449425,0.0008154339,0.0009951546,0.23938479,0.034418933,0.007808046,0.019908428,0.6772753],"study_design_scores_gemma":[0.0001527707,0.00034321164,0.0004546003,0.000021271033,0.00005514472,0.000110347464,0.00006766679,0.9849727,0.008145188,0.0042804787,0.0013565146,0.00003996337],"about_ca_topic_score_codex":0.005552365,"about_ca_topic_score_gemma":0.007547141,"teacher_disagreement_score":0.011749129,"about_ca_system_score_codex":0.0009395629,"about_ca_system_score_gemma":0.0027433205,"threshold_uncertainty_score":0.062136114},"labels":[],"label_agreement":null},{"id":"W4413279387","doi":"10.1145/3760775","title":"VulScribeR: Exploring RAG-based Vulnerability Augmentation with LLMs","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":7,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba","funders":"","keywords":"Computer science; Vulnerability (computing); Computer security","score_opus":0.12021106659626543,"score_gpt":0.33913627368043525,"score_spread":0.21892520708416982,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413279387","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.25230625,0.0050842455,0.6538383,0.0021713893,0.00048864575,0.00055261457,0.0049738633,0.07451726,0.00606738],"genre_scores_gemma":[0.6058777,0.00079185975,0.37088004,0.001686972,0.00015660973,0.0005356091,0.013440086,0.0017764384,0.0048546633],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9984824,0.00043618606,0.00008876874,0.00051470497,0.00034966107,0.00012826207],"domain_scores_gemma":[0.9964347,0.0018772655,0.0002432755,0.0010003949,0.0003359636,0.000108399036],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0020531896,0.001909813,0.0012141464,0.0025431847,0.0004973365,0.0010530425,0.0023124963,0.0015883883,0.0021620276],"category_scores_gemma":[0.008109805,0.0005541764,0.0017533213,0.0012997227,0.0012539583,0.003550238,0.0027323114,0.0020400789,0.0014466564],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0005849528,0.0006304093,0.015779292,0.00062634744,0.00027361268,0.0005864232,0.00051753933,0.23441571,0.026069375,0.005791798,0.03391153,0.68081295],"study_design_scores_gemma":[0.00006953746,0.00027496277,0.0014265575,0.000052429394,0.00009398221,0.0003332731,0.00012972613,0.9604155,0.014662179,0.012625504,0.009872499,0.000043829925],"about_ca_topic_score_codex":0.0029385933,"about_ca_topic_score_gemma":0.0054358607,"teacher_disagreement_score":0.0029385933,"about_ca_system_score_codex":0.0009055794,"about_ca_system_score_gemma":0.0014362774,"threshold_uncertainty_score":0.010858417},"labels":[],"label_agreement":null},{"id":"W4413333491","doi":"10.1145/3762183","title":"Leveraging Reviewer Experience in Code Review Comment Generation","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Code (set theory); Software engineering; Programming language","score_opus":0.17826252847388172,"score_gpt":0.3910005918240723,"score_spread":0.2127380633501906,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413333491","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.41551328,0.0029520395,0.5561791,0.0031886115,0.0010243222,0.0016817262,0.0009371546,0.010784139,0.0077395225],"genre_scores_gemma":[0.90706277,0.00028798854,0.08696821,0.0006413994,0.00034448426,0.000498991,0.0010039367,0.00045287266,0.0027393322],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.96440035,0.020896586,0.0025708082,0.004998018,0.0064085517,0.00072563434],"domain_scores_gemma":[0.7421616,0.17123803,0.02003005,0.01729522,0.045766104,0.003509014],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03487788,0.001804577,0.0014370716,0.0037426387,0.0007179347,0.00291421,0.00180178,0.0023518775,0.0016101871],"category_scores_gemma":[0.20015691,0.0007194034,0.0009334161,0.0013189581,0.00085994345,0.0039735264,0.0023457075,0.0020724514,0.0016160565],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0023605304,0.001009451,0.11234104,0.001630662,0.00061795895,0.0008017747,0.0040974882,0.071349725,0.028376993,0.0017918912,0.017685965,0.75793654],"study_design_scores_gemma":[0.00035640973,0.0016807346,0.030544937,0.00037229332,0.00037431295,0.0008566565,0.00086443854,0.9157124,0.030987099,0.0058248863,0.012094614,0.0003311419],"about_ca_topic_score_codex":0.001424646,"about_ca_topic_score_gemma":0.0025371788,"teacher_disagreement_score":0.03487788,"about_ca_system_score_codex":0.0010875621,"about_ca_system_score_gemma":0.0016837528,"threshold_uncertainty_score":0.18445408},"labels":[],"label_agreement":null},{"id":"W4413592694","doi":"10.1145/3746060","title":"Cost and Benefit of Tracing Features with Embedded Annotations","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Advanced Software Engineering Methodologies","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Waterloo","funders":"Royal Swedish Academy of Sciences","keywords":"Computer science; Tracing; Software engineering; Programming language","score_opus":0.06218509040712472,"score_gpt":0.32871344618109655,"score_spread":0.2665283557739718,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4413592694","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9157448,0.0007639466,0.06683753,0.0018493177,0.00014644669,0.00029706195,0.00032699681,0.0054699345,0.008563918],"genre_scores_gemma":[0.95693487,0.00012802686,0.03980888,0.00012558867,0.000026226711,0.000059751263,0.00022435025,0.0004397523,0.002252552],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99290097,0.002922563,0.00046407906,0.0007639945,0.0022235024,0.00072499015],"domain_scores_gemma":[0.8502306,0.112633295,0.009533378,0.020848896,0.004686657,0.0020671322],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0055385935,0.0013147902,0.00054898317,0.0023167403,0.0010628895,0.0020924788,0.0028968633,0.0027958346,0.0049728705],"category_scores_gemma":[0.08881194,0.00097096537,0.0009589631,0.0013664016,0.0016424162,0.006113224,0.002590914,0.0020133723,0.00091079826],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0061541223,0.0014996084,0.08893495,0.0009760111,0.00023081679,0.0030150341,0.0029010957,0.24821392,0.050701417,0.02156774,0.0048295786,0.57097566],"study_design_scores_gemma":[0.00046906987,0.0036566264,0.043322694,0.00035181464,0.0008985796,0.0028039643,0.0022169934,0.8649963,0.04257289,0.026345072,0.012054458,0.00031138124],"about_ca_topic_score_codex":0.006415154,"about_ca_topic_score_gemma":0.005859277,"teacher_disagreement_score":0.006415154,"about_ca_system_score_codex":0.0017257689,"about_ca_system_score_gemma":0.0022639504,"threshold_uncertainty_score":0.029291272},"labels":[],"label_agreement":null},{"id":"W4414091836","doi":"10.1145/3766890","title":"FairFLRep: Fairness-Aware Fault Localization and Repair of Deep Neural Networks","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Deep neural networks; Artificial neural network; Quality (philosophy); Baseline (sea); Fault (geology); Pattern recognition (psychology)","score_opus":0.02717373461444464,"score_gpt":0.29582057895049046,"score_spread":0.2686468443360458,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414091836","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.146862,0.0015947511,0.8353573,0.0008310262,0.0003287053,0.00024487363,0.00027164328,0.011791867,0.002717758],"genre_scores_gemma":[0.8927033,0.00019353113,0.10315473,0.00053257286,0.00006304806,0.00013559882,0.00028771497,0.00030380473,0.0026257855],"study_design_codex":"simulation_or_modeling","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.99836963,0.0003888007,0.000106243846,0.0003770526,0.00050958537,0.0002486315],"domain_scores_gemma":[0.9946407,0.002435313,0.0006663809,0.0012170807,0.0008410539,0.00019945514],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005365148,0.0013270855,0.0009370231,0.0009494206,0.00095032476,0.001178695,0.003578068,0.0015324681,0.0018668595],"category_scores_gemma":[0.017260676,0.00047133185,0.0006714557,0.00039506206,0.0016769095,0.0029698831,0.0023962976,0.0017481053,0.00036450682],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00072948367,0.0002855487,0.0070143933,0.00020562929,0.00015570698,0.00028689497,0.00024784723,0.7034642,0.009068737,0.007711978,0.0069388305,0.26389077],"study_design_scores_gemma":[0.000029717507,0.00012232213,0.0004249659,0.00001628573,0.000018086103,0.000056120727,0.000022563623,0.98296124,0.007938561,0.0075434446,0.0008511424,0.000015490947],"about_ca_topic_score_codex":0.0075141643,"about_ca_topic_score_gemma":0.010398501,"teacher_disagreement_score":0.0075141643,"about_ca_system_score_codex":0.0022878982,"about_ca_system_score_gemma":0.0023722155,"threshold_uncertainty_score":0.028373957},"labels":[],"label_agreement":null},{"id":"W4414115194","doi":"10.1145/3765735","title":"Applications and Challenges of Fairness APIs in Machine Learning Software","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Ethics and Social Impacts of AI","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Manitoba; York University; University of Calgary","funders":"","keywords":"Troubleshooting; Software; Ask price; Software development; Software peer review; Face (sociological concept)","score_opus":0.10476442243813705,"score_gpt":0.3753859940296217,"score_spread":0.27062157159148464,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414115194","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.6482558,0.0020523213,0.29816344,0.021677544,0.00018479487,0.0003698946,0.00016594726,0.0015251208,0.027605139],"genre_scores_gemma":[0.9506028,0.00041526757,0.04522567,0.0014133727,0.00008325049,0.00020692243,0.00006216749,0.00051401346,0.0014764075],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.8819441,0.08034941,0.005221126,0.0050465944,0.024406845,0.003031859],"domain_scores_gemma":[0.6910545,0.23204404,0.026932267,0.029371208,0.017923202,0.0026747615],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.07520366,0.00052486843,0.0005148398,0.0039933287,0.0051393234,0.0079388125,0.0023343018,0.0020548082,0.0019173896],"category_scores_gemma":[0.2229556,0.00091897126,0.00072636746,0.0038979184,0.012384112,0.014620072,0.0091963215,0.0033870009,0.0003648747],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0003219968,0.0003214096,0.1488625,0.0010303686,0.00010256085,0.0018077902,0.25217727,0.0038571837,0.006240107,0.23589066,0.006883363,0.34250477],"study_design_scores_gemma":[0.00011372534,0.00057250995,0.0894564,0.003144239,0.00020114458,0.004963178,0.15502836,0.047953438,0.01823126,0.42008743,0.25980264,0.00044575386],"about_ca_topic_score_codex":0.0035176624,"about_ca_topic_score_gemma":0.0034567267,"teacher_disagreement_score":0.07520366,"about_ca_system_score_codex":0.004644239,"about_ca_system_score_gemma":0.0070168204,"threshold_uncertainty_score":0.39771968},"labels":[],"label_agreement":null},{"id":"W4414153585","doi":"10.1145/3767167","title":"The Havoc Paradox in Generator-Based Fuzzing—RCR Report","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Evolutionary Algorithms and Applications","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia","funders":"","keywords":"Fuzz testing; Artifact (error); Scripting language; Exploit; Affect (linguistics)","score_opus":0.04416990580043428,"score_gpt":0.3086626128790792,"score_spread":0.2644927070786449,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414153585","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09683044,0.0015775076,0.8334335,0.009699812,0.00207708,0.00040104843,0.0010032931,0.011718565,0.043258816],"genre_scores_gemma":[0.5992621,0.0012372722,0.37836614,0.0018828484,0.00060929515,0.00040560067,0.0015717249,0.0027376416,0.013927435],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.98660266,0.0038798745,0.00076489284,0.0012026146,0.0071211094,0.0004288488],"domain_scores_gemma":[0.9410555,0.035714664,0.0019522089,0.014074937,0.00662064,0.00058205065],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.012988102,0.00066526816,0.0007177486,0.0016171841,0.0010512888,0.003371084,0.0019034105,0.0015372824,0.0071406253],"category_scores_gemma":[0.061365854,0.0005465377,0.00068196224,0.0010574965,0.0021180897,0.0044787605,0.001850485,0.00308384,0.0013245549],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0008827287,0.0008168563,0.0039350684,0.0007631406,0.00015180896,0.0008914797,0.0019454436,0.043856755,0.03729241,0.5345412,0.07382233,0.30110082],"study_design_scores_gemma":[0.00028035676,0.001195467,0.0030202137,0.00040329123,0.00016649223,0.0017039912,0.00048017234,0.37438384,0.2436597,0.257823,0.11656208,0.00032140547],"about_ca_topic_score_codex":0.0023954415,"about_ca_topic_score_gemma":0.0017049676,"teacher_disagreement_score":0.012988102,"about_ca_system_score_codex":0.0015576094,"about_ca_system_score_gemma":0.0020620457,"threshold_uncertainty_score":0.06868851},"labels":[],"label_agreement":null},{"id":"W4414913062","doi":"10.1145/3770084","title":"A Survey on LLM-based Code Generation for Low-Resource and Domain-Specific Programming Languages","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Model-Driven Software Engineering Techniques","field":"Computer Science","cited_by":24,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; Kelowna General Hospital; University of British Columbia","funders":"","keywords":"Code generation; Second-generation programming language; Leverage (statistics); Syntax; Python (programming language); Code (set theory); Domain-specific language; Semantics (computer science); Fourth-generation programming language","score_opus":0.07723071174492212,"score_gpt":0.325472582362461,"score_spread":0.24824187061753888,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414913062","genre_codex":"review","genre_gemma":"review","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"review","genre_consensus":"review","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.041717052,0.55124474,0.28532588,0.012076667,0.00148529,0.0015640628,0.027983036,0.03144324,0.047159966],"genre_scores_gemma":[0.096443325,0.36345363,0.41997692,0.0066144983,0.0008704632,0.0025030093,0.08784313,0.011720421,0.010574668],"study_design_codex":"design_other","study_design_gemma":"not_applicable","domain_scores_codex":[0.98857415,0.0036992384,0.0019092552,0.0015202811,0.0038636825,0.00043338668],"domain_scores_gemma":[0.92651576,0.054453608,0.0033035867,0.006840842,0.008308591,0.0005776597],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.008713592,0.0018430768,0.001274055,0.013280635,0.00086609815,0.0034881646,0.0036636319,0.001437411,0.007469822],"category_scores_gemma":[0.08610086,0.0012661333,0.003108444,0.012565168,0.0011148588,0.0068764957,0.0031532235,0.001813722,0.004388776],"study_design_candidate":"not_applicable","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00016220675,0.0001370516,0.0086053135,0.035766277,0.00029459907,0.00024428597,0.0011889542,0.0050166,0.0032739537,0.015997687,0.08180495,0.8475082],"study_design_scores_gemma":[0.00009932483,0.00028459742,0.009900628,0.023046937,0.0004790179,0.0009168021,0.0012547965,0.018720085,0.011578247,0.020933252,0.91257465,0.00021177439],"about_ca_topic_score_codex":0.0061716,"about_ca_topic_score_gemma":0.007162504,"teacher_disagreement_score":0.013280635,"about_ca_system_score_codex":0.0015391815,"about_ca_system_score_gemma":0.0073573203,"threshold_uncertainty_score":0.046082437},"labels":[],"label_agreement":null},{"id":"W4414993542","doi":"10.1145/3771283","title":"LLM meets ML: Data-efficient Anomaly Detection on Unstable Logs","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software System Performance and Reliability","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Anomaly detection; Inference; Anomaly (physics); Artificial neural network; Key (lock); Software; Cache; Ensemble learning","score_opus":0.07009888762071217,"score_gpt":0.3147679512299077,"score_spread":0.24466906360919552,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4414993542","genre_codex":"methods","genre_gemma":"methods","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"methods","genre_consensus":"methods","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.1745263,0.0019641197,0.6920414,0.0011190507,0.00038484173,0.00032643953,0.005786232,0.12037269,0.0034790025],"genre_scores_gemma":[0.65422326,0.00036784247,0.32726535,0.00044234627,0.00013515414,0.0002657968,0.013276692,0.0010355919,0.002988031],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.99802554,0.00043650737,0.00016748307,0.00057094765,0.0006037651,0.00019568733],"domain_scores_gemma":[0.99525696,0.0017873469,0.0003126631,0.0015572815,0.0009004718,0.00018528348],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0027082246,0.0018777918,0.0014206026,0.002436306,0.00063828955,0.0015551681,0.0032043639,0.0013611848,0.0015819978],"category_scores_gemma":[0.012409442,0.0005388622,0.00093292625,0.0014481299,0.00058374763,0.0040409015,0.0021676568,0.002052704,0.0019211593],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00079357496,0.0007744943,0.03413707,0.00036811797,0.00024440925,0.00034675305,0.00022565178,0.11890709,0.013968945,0.0023243062,0.03909512,0.78881437],"study_design_scores_gemma":[0.000029776198,0.00011551026,0.0024145844,0.00001474352,0.000016512622,0.00011618906,0.00007714598,0.98365134,0.006450299,0.0038850603,0.0032048554,0.000023929271],"about_ca_topic_score_codex":0.0076635457,"about_ca_topic_score_gemma":0.0127801355,"teacher_disagreement_score":0.0076635457,"about_ca_system_score_codex":0.0008393524,"about_ca_system_score_gemma":0.0017002295,"threshold_uncertainty_score":0.015237868},"labels":[],"label_agreement":null},{"id":"W4415312861","doi":"10.1145/3771929","title":"Continuously Learning Bug Locations","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Software bug; Software regression; Software; Code (set theory); Deep learning; Forgetting; Source code; Mean reciprocal rank","score_opus":0.04947276161402145,"score_gpt":0.3231860867865974,"score_spread":0.27371332517257596,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415312861","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.5342503,0.003954112,0.4340063,0.0015852085,0.00040619762,0.00019927065,0.0026233434,0.019321602,0.0036537189],"genre_scores_gemma":[0.9280525,0.00038265463,0.0649595,0.00028529725,0.000117498756,0.00009218863,0.0033445172,0.00022605126,0.0025398291],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9985266,0.00021729454,0.00009571145,0.0007631685,0.00028289217,0.000114296156],"domain_scores_gemma":[0.9939167,0.0030004766,0.00085401675,0.00065267703,0.0012618155,0.0003142693],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001409481,0.0017457611,0.0010077351,0.0025858225,0.00036635794,0.0009595298,0.0019975842,0.0012522157,0.0014134645],"category_scores_gemma":[0.011298339,0.0005467289,0.00092426885,0.0012958106,0.0005084722,0.0021471402,0.0014157715,0.0017289396,0.0011337595],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00051250384,0.00067851593,0.08024874,0.0004939668,0.00026349002,0.00044673614,0.00027337318,0.23556119,0.008460787,0.0010532326,0.013945974,0.6580615],"study_design_scores_gemma":[0.000026821413,0.0001727863,0.005820498,0.000036674093,0.00005357281,0.00012788399,0.000062605664,0.98693407,0.002750182,0.0023394332,0.0016570938,0.00001834895],"about_ca_topic_score_codex":0.0066604395,"about_ca_topic_score_gemma":0.008770238,"teacher_disagreement_score":0.0066604395,"about_ca_system_score_codex":0.0007993166,"about_ca_system_score_gemma":0.0014096233,"threshold_uncertainty_score":0.013243318},"labels":[],"label_agreement":null},{"id":"W4415330783","doi":"10.1145/3744902","title":"Do Current Language Models Support Code Intelligence for R Programming Language? RCR Report","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Data Analysis with R","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":true,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of British Columbia, Okanagan Campus; University of British Columbia","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Automatic summarization; Parsing; Code (set theory); Natural language; Matching (statistics); Replicate; Source code; Compiler","score_opus":0.09264298462753738,"score_gpt":0.38361259729604125,"score_spread":0.29096961266850385,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415330783","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.025303436,0.00427743,0.56510055,0.0199729,0.0016186632,0.0007487432,0.11999469,0.24458377,0.018399762],"genre_scores_gemma":[0.097416416,0.0021340419,0.6260093,0.0070267306,0.0006054498,0.0025315986,0.2154036,0.042982362,0.0058905086],"study_design_codex":"not_applicable","study_design_gemma":"observational","domain_scores_codex":[0.96213216,0.01976579,0.002641917,0.008043393,0.006483815,0.0009329791],"domain_scores_gemma":[0.79309374,0.1198373,0.006734134,0.06453875,0.013689257,0.0021068386],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.041549385,0.0024142563,0.0016435237,0.0028980924,0.0010221933,0.007047577,0.005035244,0.0020218436,0.012178101],"category_scores_gemma":[0.2572184,0.0017037472,0.0030103405,0.004189463,0.0023226663,0.010627243,0.0041770977,0.0060216547,0.026448421],"study_design_candidate":"observational","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0011990081,0.00032098827,0.018108968,0.0028786885,0.0008704778,0.0001681567,0.00085557695,0.021930417,0.0049706297,0.042734984,0.6512815,0.25468072],"study_design_scores_gemma":[0.0006193682,0.00043980055,0.009015825,0.0013020753,0.0003832032,0.00046374137,0.00045980065,0.34138033,0.015998855,0.122912236,0.506694,0.0003308592],"about_ca_topic_score_codex":0.008191884,"about_ca_topic_score_gemma":0.0153630255,"teacher_disagreement_score":0.041549385,"about_ca_system_score_codex":0.0021838092,"about_ca_system_score_gemma":0.0070878426,"threshold_uncertainty_score":0.21973675},"labels":[],"label_agreement":null},{"id":"W4415589112","doi":"10.1145/3773034","title":"Synthesizing Efficient and Permissive Programmatic Runtime Shields for Neural Policies","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Neural Networks and Applications","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Alberta Medical Association","funders":"","keywords":"Shields; Overhead (engineering); Reduction (mathematics); Software; Control (management); Reliability (semiconductor); Safety standards; Runtime system","score_opus":0.0516818758031873,"score_gpt":0.3185185259233981,"score_spread":0.2668366501202108,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415589112","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.062096994,0.0004412193,0.92267805,0.00029908965,0.00008241509,0.00018637023,0.00022193286,0.007832948,0.0061608697],"genre_scores_gemma":[0.59534514,0.00035345374,0.39941275,0.00025622887,0.000029897965,0.00032125966,0.00050693826,0.0012393415,0.0025350857],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9990644,0.0002198848,0.000058842794,0.00016997798,0.00033027164,0.0001566272],"domain_scores_gemma":[0.9978811,0.0013033128,0.00018976422,0.00032505218,0.00022472459,0.00007605667],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0011373542,0.0011240686,0.00058753655,0.000617406,0.0004250744,0.0008052084,0.0009090802,0.000806218,0.0040913112],"category_scores_gemma":[0.0054763732,0.0005283346,0.0010640961,0.00024061922,0.0013629015,0.0011828538,0.0014101537,0.0013995343,0.00068057806],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00025686985,0.0001230542,0.0030804537,0.00055428257,0.00006441872,0.000284458,0.00034283922,0.7486378,0.039152667,0.0331742,0.003239276,0.17108968],"study_design_scores_gemma":[0.00003236884,0.000095225725,0.00019720769,0.00003418319,0.000024860527,0.000056106877,0.000044281078,0.97191364,0.014017847,0.010189685,0.0033808379,0.000013715244],"about_ca_topic_score_codex":0.002903355,"about_ca_topic_score_gemma":0.004991282,"teacher_disagreement_score":0.0040913112,"about_ca_system_score_codex":0.0009396314,"about_ca_system_score_gemma":0.0025515968,"threshold_uncertainty_score":0.013686836},"labels":[],"label_agreement":null},{"id":"W4415622293","doi":"10.1145/3773285","title":"TaskEval: Assessing Difficulty of Code Generation Tasks for Large Language Models","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Huawei Technologies (Canada)","funders":"","keywords":"Benchmarking; Benchmark (surveying); Task (project management); Code (set theory); Program comprehension; Code generation; Source code","score_opus":0.11296114902424609,"score_gpt":0.36077962888675685,"score_spread":0.24781847986251077,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4415622293","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.7226199,0.0011954497,0.23179214,0.00048511967,0.0002981583,0.002275503,0.010164919,0.021549778,0.009619069],"genre_scores_gemma":[0.78418124,0.0002577919,0.18926722,0.00022270023,0.00007169814,0.0030236393,0.018550184,0.0018983628,0.0025271126],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.9846506,0.008590813,0.0018254286,0.0017252542,0.00280406,0.00040372048],"domain_scores_gemma":[0.8590377,0.11224894,0.00992821,0.00926369,0.007584082,0.0019374862],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013125193,0.0020994502,0.000896756,0.0041150795,0.00059396564,0.002476002,0.0017139455,0.0016497934,0.003131385],"category_scores_gemma":[0.114503175,0.00047938357,0.0012398509,0.002312798,0.0007499501,0.0038779061,0.0036891461,0.0016134037,0.0016225877],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.004863173,0.0037824006,0.16912688,0.006162766,0.0011281398,0.0004591838,0.01101167,0.05393648,0.02888154,0.0077656386,0.058356963,0.65452516],"study_design_scores_gemma":[0.001221494,0.009038024,0.30812398,0.0008278027,0.00051157543,0.00140002,0.0065490142,0.5203095,0.06526015,0.029308772,0.056605384,0.0008443152],"about_ca_topic_score_codex":0.0024732936,"about_ca_topic_score_gemma":0.0033085255,"teacher_disagreement_score":0.013125193,"about_ca_system_score_codex":0.0010064418,"about_ca_system_score_gemma":0.0013353772,"threshold_uncertainty_score":0.06941348},"labels":[],"label_agreement":null},{"id":"W4416229447","doi":"10.1145/3776739","title":"An Empirical Analysis of Machine Learning Model and Dataset Documentation, Supply Chain, and Licensing Challenges on Hugging Face","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Victoria","funders":"","keywords":"Documentation; License; Software; Face (sociological concept); Artificial neural network; Facial recognition system; Work (physics); Convolutional neural network","score_opus":0.06670334465444491,"score_gpt":0.3676025793805503,"score_spread":0.30089923472610536,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416229447","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"reproducibility","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"reproducibility","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9438133,0.0031587621,0.020369228,0.0033061963,0.00025663356,0.00017843288,0.019217039,0.0016756153,0.008024722],"genre_scores_gemma":[0.9308812,0.00061818823,0.013051281,0.0005018708,0.0001085292,0.00016016557,0.051992495,0.00034171776,0.0023445627],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.98833734,0.0062164925,0.00077719206,0.0016657653,0.002502224,0.00050105766],"domain_scores_gemma":[0.87235785,0.090268105,0.006236547,0.022727834,0.0072016716,0.0012080243],"candidate_categories":["metaresearch","open_science"],"consensus_categories":[],"category_scores_codex":[0.022000683,0.000926572,0.0008632117,0.0029285622,0.0016381457,0.0027123413,0.0022206753,0.0020683096,0.0039019752],"category_scores_gemma":[0.099203065,0.00039984495,0.001004014,0.0043028737,0.0019622413,0.0061177867,0.0023502212,0.003338382,0.0018958197],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0016220867,0.0017927383,0.44949323,0.0014535598,0.00069288444,0.0013419086,0.0020839719,0.1727332,0.0016625657,0.023490654,0.15383878,0.1897945],"study_design_scores_gemma":[0.0002914403,0.00093665725,0.17789447,0.0009054098,0.000307321,0.0026081188,0.004312743,0.67455226,0.006165662,0.03982922,0.09191566,0.0002809862],"about_ca_topic_score_codex":0.0071057477,"about_ca_topic_score_gemma":0.009163543,"teacher_disagreement_score":0.9977793,"about_ca_system_score_codex":0.0016966455,"about_ca_system_score_gemma":0.0014472764,"threshold_uncertainty_score":0.11635214},"labels":[],"label_agreement":null},{"id":"W4416767288","doi":"10.1145/3778031","title":"Understanding and Estimating the Execution Time of Quantum Circuits","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Quantum Computing Algorithms and Architecture","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Quantum computer; Quantum circuit; Quantum; Quantum algorithm; Qubit; Quantum information; Provisioning; Benchmark (surveying); Quantum network","score_opus":0.08260533931163938,"score_gpt":0.2945200368030502,"score_spread":0.21191469749141084,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4416767288","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.79400826,0.00077325787,0.19917443,0.00040177826,0.0000428937,0.00007168279,0.0012923803,0.0018115403,0.0024238615],"genre_scores_gemma":[0.9482323,0.00026887425,0.048566725,0.000042070926,0.000017677901,0.000056527428,0.0019347033,0.00016383716,0.00071737624],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.999493,0.00010651545,0.000027112073,0.00019241215,0.00011543964,0.00006542668],"domain_scores_gemma":[0.9965468,0.0022494607,0.00036719142,0.00038183798,0.00036007594,0.00009466225],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001008751,0.00061846577,0.00040748183,0.0010877014,0.00031777105,0.00082375377,0.000993184,0.00078172656,0.0010750486],"category_scores_gemma":[0.009805019,0.00034853362,0.00047630403,0.0008494483,0.0005524268,0.00195745,0.00041115689,0.0010178462,0.0002863636],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00015621667,0.00010249875,0.024906658,0.0001612398,0.000050098315,0.00006644725,0.000088888024,0.9109034,0.008347778,0.0057420866,0.0016759116,0.047798738],"study_design_scores_gemma":[0.0000043966797,0.000021220792,0.002533729,0.0000032675061,0.0000056757704,0.00001482284,0.000013286296,0.9926132,0.0021457972,0.002309863,0.00032994652,0.000004740204],"about_ca_topic_score_codex":0.009008863,"about_ca_topic_score_gemma":0.011006764,"teacher_disagreement_score":0.009008863,"about_ca_system_score_codex":0.0012446536,"about_ca_system_score_gemma":0.0014552827,"threshold_uncertainty_score":0.017912805},"labels":[],"label_agreement":null},{"id":"W4417339614","doi":"10.1145/3785001","title":"An Empirical Study of Self-Admitted Technical Debt in Machine Learning Software","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Polytechnique Montréal; Queen's University","funders":"","keywords":"Technical debt; Context (archaeology); Empirical research; Code smell; Pipeline (software); Source code; Code (set theory)","score_opus":0.05521184893937985,"score_gpt":0.3650327064973448,"score_spread":0.30982085755796496,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417339614","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.9979715,0.00012990068,0.0007750765,0.00017932366,0.00000386875,0.000021605234,0.00020548268,0.000032324202,0.0006808942],"genre_scores_gemma":[0.9983571,0.00009638168,0.00073605415,0.00004500346,0.000009989183,0.00002461752,0.00037897422,0.000023264916,0.00032856775],"study_design_codex":"observational","study_design_gemma":"observational","domain_scores_codex":[0.9905129,0.002843995,0.0012801408,0.0014412809,0.0032547133,0.0006670812],"domain_scores_gemma":[0.6949841,0.14977771,0.103417456,0.013817937,0.031289227,0.006713547],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.009979682,0.00029113403,0.00025353802,0.0043814005,0.0009941591,0.0016892145,0.0008779667,0.0007960689,0.00129067],"category_scores_gemma":[0.14200312,0.00042643427,0.00025092278,0.004446433,0.0019446217,0.004505402,0.0023748598,0.0017356375,0.00045924287],"study_design_candidate":"observational","study_design_consensus":"observational","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0000978913,0.00010403912,0.97339284,0.00010649709,0.000030166608,0.00028605125,0.006301539,0.00041266106,0.0006705216,0.00049023714,0.0007070336,0.017400509],"study_design_scores_gemma":[0.000007431724,0.00022722855,0.9825613,0.00011244317,0.000015395675,0.0006947956,0.0074303816,0.005000664,0.0007635559,0.0006499116,0.0025036486,0.00003318113],"about_ca_topic_score_codex":0.0029382005,"about_ca_topic_score_gemma":0.0036806588,"teacher_disagreement_score":0.99002033,"about_ca_system_score_codex":0.0011733745,"about_ca_system_score_gemma":0.0009962147,"threshold_uncertainty_score":0.052778244},"labels":[],"label_agreement":null},{"id":"W4417436230","doi":"10.1145/3785363","title":"Galápagos: Automated N-Version Programming with LLMs","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"Université de Montréal","funders":"","keywords":"Correctness; Redundancy (engineering); Software; Symbolic execution; Program analysis; Automatic programming; Software quality; Programming paradigm; Software testing","score_opus":0.04061881690512797,"score_gpt":0.3080365130385907,"score_spread":0.2674176961334627,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W4417436230","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.09482519,0.000117183285,0.8746911,0.00013430468,0.000029432622,0.00014223276,0.00018754628,0.027203469,0.0026695246],"genre_scores_gemma":[0.5407792,0.00009174553,0.4516693,0.00017210525,0.000013607708,0.00023227352,0.0005250438,0.004131954,0.0023848142],"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9983918,0.00049220084,0.000105852894,0.0003137095,0.0005590653,0.0001374139],"domain_scores_gemma":[0.99545527,0.0018339498,0.00038780548,0.0019222928,0.00030963458,0.000091042995],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0017044256,0.00070435327,0.0003685252,0.0006781533,0.00040687277,0.0010464037,0.002145301,0.0007361535,0.0023771136],"category_scores_gemma":[0.0060685817,0.00071616174,0.0011118911,0.00033015021,0.0017450234,0.0019254384,0.0024363734,0.0011708032,0.0006497688],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0013746363,0.00043019143,0.024629148,0.00094312715,0.00023312731,0.0015778892,0.0022374557,0.28150395,0.18662642,0.10457791,0.00835447,0.38751167],"study_design_scores_gemma":[0.00014773395,0.00040792127,0.0017081473,0.00010457739,0.00007248149,0.0007491267,0.0001248392,0.77795863,0.14649852,0.053163916,0.018972997,0.00009109605],"about_ca_topic_score_codex":0.0009956084,"about_ca_topic_score_gemma":0.0013415407,"teacher_disagreement_score":0.0023771136,"about_ca_system_score_codex":0.0006688661,"about_ca_system_score_gemma":0.0010594856,"threshold_uncertainty_score":0.009013951},"labels":[],"label_agreement":null},{"id":"W7116688751","doi":"10.1145/3785472","title":"A Multi-Agent RAG Framework for Regulatory Compliance Checking of Software Requirements","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Multi-Agent Systems and Negotiation","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Calgary","funders":"","keywords":"Requirements engineering; Software requirements; Requirements analysis; Software requirements specification; Requirements elicitation; Compliance (psychology); Non-functional requirement; Requirements management; Functional requirement","score_opus":0.19879113178629848,"score_gpt":0.38080256214929037,"score_spread":0.1820114303629919,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7116688751","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":null,"model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":null,"domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.0017937529,0.00013011412,0.9896004,0.00021418727,0.00003106002,0.00027405107,0.00006855982,0.004720387,0.0031675634],"genre_scores_gemma":[0.06611857,0.00011831865,0.93001276,0.00011935597,0.000022870367,0.0003104169,0.00020757492,0.0002870616,0.002803173],"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9952512,0.002134831,0.0003653117,0.00076857914,0.0012117564,0.0002684327],"domain_scores_gemma":[0.99593866,0.0017510272,0.0005006444,0.0011228259,0.00050033606,0.0001864701],"candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0070242733,0.0009580677,0.0006850486,0.0019544482,0.0013438662,0.0030713405,0.0033652417,0.0019229889,0.0067074555],"category_scores_gemma":[0.008264283,0.0006912424,0.0018130323,0.0008995724,0.0021449681,0.003046753,0.003133739,0.002023319,0.0019512632],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.00028443043,0.00040521554,0.0014543401,0.0006792513,0.00015695894,0.0014426231,0.0017982695,0.18785274,0.019042036,0.5338636,0.009739968,0.24328063],"study_design_scores_gemma":[0.00010970279,0.00020250278,0.00028499632,0.0001346877,0.00007777364,0.00047679085,0.00022639,0.80217975,0.011096815,0.11249304,0.07263309,0.00008453745],"about_ca_topic_score_codex":0.005957267,"about_ca_topic_score_gemma":0.0071847243,"teacher_disagreement_score":0.0070242733,"about_ca_system_score_codex":0.001432291,"about_ca_system_score_gemma":0.0039372994,"threshold_uncertainty_score":0.037148356},"labels":[],"label_agreement":null},{"id":"W7117477189","doi":"10.1145/3786771","title":"Assessing and Advancing Benchmarks for Evaluating Large Language Models in Software Engineering Tasks","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Software Engineering Research","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"University of Ottawa","funders":"","keywords":"Popularity; Software quality assurance; Model-driven architecture; Software development; Coding (social sciences); Quality (philosophy); Software; Benchmark (surveying); Social software engineering","score_opus":0.07511031672640531,"score_gpt":0.3920066529526886,"score_spread":0.3168963362262833,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7117477189","genre_codex":"methods","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"evaluation","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":null,"domain_candidate":"evaluation","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.35236132,0.017329024,0.5757956,0.0029476874,0.0010179993,0.0021646393,0.0054942477,0.011877487,0.031011993],"genre_scores_gemma":[0.50213325,0.003706838,0.4746594,0.00044481418,0.00015906994,0.0019688192,0.012947919,0.0018905946,0.0020893642],"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","domain_scores_codex":[0.96129864,0.021006433,0.004160275,0.0017950061,0.010619751,0.001119984],"domain_scores_gemma":[0.84954315,0.10046197,0.008795676,0.014303503,0.024599086,0.0022966324],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.028474636,0.0021027406,0.0011808324,0.0070931436,0.0010978103,0.0044698333,0.0033217552,0.0015706029,0.0021096505],"category_scores_gemma":[0.15091833,0.00064067804,0.0010085861,0.0071850573,0.001262063,0.00547589,0.0036110946,0.0024662684,0.000944953],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.0015155736,0.0021974293,0.04428716,0.0054569943,0.00057861686,0.0003161597,0.0022337695,0.21614581,0.015310949,0.058141638,0.03178678,0.6220291],"study_design_scores_gemma":[0.00036959958,0.002782533,0.02373221,0.0027037559,0.00033167773,0.00042917213,0.002290535,0.8092759,0.040398087,0.06564186,0.05172943,0.00031513412],"about_ca_topic_score_codex":0.0067143594,"about_ca_topic_score_gemma":0.009007523,"teacher_disagreement_score":0.9715254,"about_ca_system_score_codex":0.0031144547,"about_ca_system_score_gemma":0.004158453,"threshold_uncertainty_score":0.15059012},"labels":[],"label_agreement":null},{"id":"W7128191878","doi":"10.1145/3745764","title":"NLPerturbator: Studying the Robustness of Code LLMs to Natural Language Variations","year":2025,"lang":"en","type":"article","venue":"ACM Transactions on Software Engineering and Methodology","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"route_ca_aff":true,"route_ca_fund":false,"route_ca_venue":false,"route_about_ca":false,"ca_institutions":"York University","funders":"National Key Research and Development Program of China","keywords":"Robustness (evolution); Natural language; Coding (social sciences); Code (set theory); Natural language generation","score_opus":0.056644086971520985,"score_gpt":0.31641526681795956,"score_spread":0.2597711798464386,"validation_status":"score_only:v0-immature-baseline","prediction":{"id":"W7128191878","genre_codex":"empirical","genre_gemma":"empirical","domain_codex":null,"domain_gemma":"methods","model_version":"metacan-v3-hybrid-931329e0061c","genre_candidate":"empirical","genre_consensus":"empirical","domain_candidate":"methods","domain_consensus":null,"prediction_status":"machine_predicted_unvalidated","genre_scores_codex":[0.630388,0.0017352307,0.34624124,0.0006959021,0.00019628258,0.0008765706,0.0014894708,0.01426695,0.0041104085],"genre_scores_gemma":[0.8875883,0.00031626556,0.1068349,0.00028931157,0.000052544543,0.00058630196,0.002169569,0.0013015383,0.00086123013],"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","domain_scores_codex":[0.9795815,0.009436825,0.0013828434,0.0037204975,0.0052235667,0.00065472384],"domain_scores_gemma":[0.8282169,0.12024109,0.015274989,0.02738224,0.0076454272,0.0012393644],"candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.013466648,0.0014664375,0.00082554814,0.0027865772,0.00081969576,0.0020695482,0.0022735104,0.0015791247,0.0012186419],"category_scores_gemma":[0.1662024,0.0008184973,0.0010063843,0.0016611954,0.0027639002,0.0047285375,0.003065491,0.0025697716,0.00078540656],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_system_candidate":false,"about_ca_system_consensus":false,"study_design_scores_codex":[0.003193958,0.0012084165,0.08667507,0.0026194712,0.0009984118,0.0007628776,0.004749591,0.44120932,0.08713088,0.012605271,0.008939668,0.34990695],"study_design_scores_gemma":[0.00014943589,0.0015572661,0.023015615,0.00012308448,0.00022952682,0.0005570541,0.0008071381,0.8969845,0.053193145,0.015900504,0.0072917505,0.00019100458],"about_ca_topic_score_codex":0.0050553796,"about_ca_topic_score_gemma":0.0033921795,"teacher_disagreement_score":0.98653334,"about_ca_system_score_codex":0.0019368596,"about_ca_system_score_gemma":0.0018936384,"threshold_uncertainty_score":0.071219265},"labels":[],"label_agreement":null}]}