{"id":"W4311273067","doi":"10.29007/fhnk","title":"ARCH-COMP 2022 Category Report: Falsification with Ubounded Resources","year":2022,"lang":"en","type":"article","venue":"EPiC series in computing","topic":"Formal Methods in Verification","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University","funders":"JST-Mirai Program; Core Research for Evolutional Science and Technology; Exploratory Research for Advanced Technology; National Science Foundation; Division of Civil, Mechanical and Manufacturing Innovation; ACT-X; Japan Society for the Promotion of Science; Natural Sciences and Engineering Research Council of Canada; Division of Industrial Innovation and Partnerships; Defense Advanced Research Projects Agency","keywords":"Comparability; Competition (biology); Arch; Computer science; Work (physics); Data science; Software engineering; Engineering; Civil engineering; Mathematics; Mechanical engineering","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02638431,0.00268562,0.0017627,0.005952384,0.003740003,0.01232013,0.004936703,0.004163505,0.06295916],"category_scores_gemma":[0.04131593,0.0009156656,0.002367787,0.003637003,0.002555484,0.00609484,0.008365132,0.005161822,0.04791172],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005271978,"about_ca_system_score_gemma":0.006726699,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009884736,"about_ca_topic_score_gemma":0.01126902,"domain_scores_codex":[0.9670776,0.007140775,0.001338368,0.003069419,0.01789447,0.003479375],"domain_scores_gemma":[0.9327696,0.01614767,0.00179145,0.01833673,0.02564713,0.005307423],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.002523264,0.0007633349,0.003428265,0.000886653,0.0001381156,0.0003302461,0.000321315,0.006499574,0.004299442,0.0322593,0.8742803,0.07427027],"study_design_scores_gemma":[0.0008442702,0.001518632,0.005089943,0.0004934194,0.0001350694,0.0008569903,0.001038629,0.03149623,0.03903696,0.03695839,0.8822759,0.0002555512],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.09424745,0.006164929,0.1715367,0.01744293,0.02623155,0.002675632,0.1648649,0.1236234,0.3932126],"genre_scores_gemma":[0.2583525,0.001548302,0.1476632,0.004934893,0.001997433,0.002304505,0.3804469,0.03632264,0.1664295],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.06295916,"threshold_uncertainty_score":0.2106194,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02085790349343988,"score_gpt":0.2712023298882709,"score_spread":0.2503444263948311,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}