{"id":"W4404514810","doi":"10.1145/3689944.3696162","title":"BinEq - A Benchmark of Compiled Java Programs to Assess Alternative Builds","year":2023,"lang":"en","type":"article","venue":"","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":5,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo","funders":"","keywords":"Computer science; Java; Benchmark (surveying); Programming language; Software engineering; Operating system; Geography; Cartography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004026597,0.001346585,0.0004608995,0.002442044,0.0005927648,0.0007942276,0.001810803,0.0009514395,0.002020979],"category_scores_gemma":[0.01720587,0.0004330994,0.0009271539,0.00234811,0.001181702,0.00194608,0.001694614,0.001445912,0.0006978519],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008838405,"about_ca_system_score_gemma":0.001160421,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003382401,"about_ca_topic_score_gemma":0.003278845,"domain_scores_codex":[0.9953849,0.001358105,0.0006162119,0.0009381085,0.001284892,0.0004178905],"domain_scores_gemma":[0.9817074,0.009412682,0.001034174,0.004321207,0.002799325,0.0007251835],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.006017976,0.00500251,0.192807,0.005946917,0.001387264,0.00179535,0.002458855,0.2829666,0.1117118,0.02826711,0.06333616,0.2983024],"study_design_scores_gemma":[0.0007686406,0.005758543,0.1365466,0.0003641839,0.0003537947,0.001350309,0.001141969,0.6528737,0.1288892,0.0228812,0.04882333,0.0002485862],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9223362,0.000937582,0.0427234,0.0002629088,0.0002176904,0.0004039233,0.009218661,0.01669589,0.007203621],"genre_scores_gemma":[0.8566682,0.0003523776,0.09664621,0.00019769,0.00005088331,0.000470362,0.0396446,0.003484196,0.002485506],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004026597,"threshold_uncertainty_score":0.02129489,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1024521892520282,"score_gpt":0.3377391067043839,"score_spread":0.2352869174523557,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}