{"id":"W4280619266","doi":"10.4230/lipics.itp.2022.18","title":"Automatic Test-Case Reduction in Proof Assistants: A Case Study in Coq","year":2022,"lang":"en","type":"article","venue":"DROPS (Schloss Dagstuhl – Leibniz Center for Informatics)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"Prevention of Organ Failure","funders":"","keywords":"Computer science; Reduction (mathematics); Test (biology); Proof assistant; Proof of concept; Programming language; Operating system; Mathematics; Mathematical proof","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.013411,0.00104727,0.0007071313,0.001344898,0.001104198,0.0014673,0.003066057,0.001995211,0.001814604],"category_scores_gemma":[0.08403687,0.0008621739,0.0006981056,0.001413846,0.002234774,0.002527255,0.001977053,0.002162303,0.000651298],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001289634,"about_ca_system_score_gemma":0.001534885,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005994147,"about_ca_topic_score_gemma":0.005009412,"domain_scores_codex":[0.9827209,0.01005945,0.0008825361,0.001520969,0.003980021,0.0008361079],"domain_scores_gemma":[0.8032766,0.164755,0.0045119,0.01759674,0.008369414,0.001490284],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.003148677,0.005424189,0.04792395,0.002499287,0.0002808149,0.01015649,0.02353666,0.1512298,0.07847746,0.02781737,0.01524862,0.6342567],"study_design_scores_gemma":[0.001359443,0.00605028,0.02931236,0.0005606035,0.0003272688,0.01151737,0.004581789,0.6708084,0.1788408,0.01718609,0.07912508,0.0003305766],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.7600886,0.0007399402,0.2238403,0.0007820198,0.00004994213,0.0005429865,0.000315123,0.009326984,0.004314043],"genre_scores_gemma":[0.7809086,0.0002265254,0.2139248,0.0002148175,0.00002095974,0.0002066363,0.0004702118,0.002034023,0.001993525],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.013411,"threshold_uncertainty_score":0.070925,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02613850071469267,"score_gpt":0.2937174769024335,"score_spread":0.2675789761877409,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}