{"id":"W4387634923","doi":"10.48550/arxiv.2310.07856","title":"Assessing Evaluation Metrics for Neural Test Oracle Generation","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Adversarial Robustness in Machine Learning","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"York University","funders":"","keywords":"Oracle; Computer science; Test (biology); Metric (unit); Assertion; Machine learning; Test case; Artificial intelligence; Data mining; Programming language; Regression analysis; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001506673,0.0002665201,0.0002661567,0.000530124,0.0003950903,0.000592091,0.001292286,0.0002758879,0.000009333297],"category_scores_gemma":[0.001799816,0.0003301374,0.0001753145,0.001308996,0.00004221598,0.001219907,0.001470736,0.0005436326,0.00003222368],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004839582,"about_ca_system_score_gemma":0.0003452211,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00007606079,"about_ca_topic_score_gemma":0.00002903262,"domain_scores_codex":[0.9977583,0.0002838863,0.0002417047,0.001151656,0.0002365952,0.0003278503],"domain_scores_gemma":[0.9972743,0.0007636679,0.000426117,0.0008831202,0.0005595871,0.00009319784],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000002916709,0.00002930989,0.00209634,0.00003832132,0.0000248535,0.0000218841,0.00007647912,0.9797141,0.00007168964,0.0108131,0.0001997188,0.006911288],"study_design_scores_gemma":[0.0004390365,0.00003475099,0.001683896,0.00003072621,0.0001178779,0.000001239785,0.00004718228,0.9824588,0.00007176118,0.0147082,0.00007630072,0.0003302472],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.06892924,0.00003387572,0.9274237,0.0002440528,0.00212346,0.0005982345,0.000007499596,0.0004474908,0.0001924762],"genre_scores_gemma":[0.9653221,0.00001615717,0.0335834,0.00004630566,0.000400586,0.000005718528,0.0001115074,0.00003525897,0.0004789715],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8963929,"threshold_uncertainty_score":0.9999151,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.334270648900014,"score_gpt":0.2987627042028986,"score_spread":0.03550794469711538,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}