{"id":"W4400978763","doi":"10.1109/tse.2024.3433463","title":"Assessing Evaluation Metrics for Neural Test Oracle Generation","year":2024,"lang":"en","type":"article","venue":"IEEE Transactions on Software Engineering","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"Natural Sciences and Engineering Research Council of Canada; Alberta Innovates","keywords":"Computer science; Oracle; Test (biology); Artificial neural network; Machine learning; Artificial intelligence; Data mining; Software engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0163487,0.002059615,0.001118467,0.00547723,0.0004289473,0.001725622,0.002671717,0.001831361,0.001092662],"category_scores_gemma":[0.07321007,0.0004850247,0.0008813231,0.002708521,0.0009812658,0.003163437,0.00152791,0.001560842,0.0004443592],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.004012553,"about_ca_system_score_gemma":0.001555283,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008149184,"about_ca_topic_score_gemma":0.009440368,"domain_scores_codex":[0.9812439,0.00940529,0.001910076,0.002242146,0.004620781,0.000577816],"domain_scores_gemma":[0.9242682,0.05349312,0.005338976,0.005940189,0.009645438,0.001314187],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001117059,0.0006978048,0.05391304,0.0006224934,0.0007740631,0.0001518261,0.0001986065,0.5863883,0.00411688,0.002536953,0.006851782,0.3426311],"study_design_scores_gemma":[0.00004626533,0.0005470799,0.004594519,0.0000372788,0.00005515253,0.0000606277,0.00004131041,0.988273,0.004734738,0.001086953,0.0004993793,0.00002373563],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7561023,0.007062945,0.2138563,0.0009375013,0.0002648613,0.0004984291,0.002018926,0.01212141,0.007137274],"genre_scores_gemma":[0.9271518,0.0004110581,0.06611537,0.0001981758,0.000051913,0.0002288729,0.004573859,0.0003113769,0.0009575615],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.0163487,"threshold_uncertainty_score":0.08646125,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0715888736640746,"score_gpt":0.3253112017937551,"score_spread":0.2537223281296804,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}