{"id":"W7138891804","doi":"10.1109/adacis65663.2025.11436933","title":"HOL4PSG: AI-Driven Proof Sequence Generation for the HOL4 Theorem Prover","year":2025,"lang":"","type":"article","venue":"","topic":"Logic, programming, and type systems","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Concordia University","funders":"","keywords":"Mathematical proof; Automated theorem proving; Proof assistant; Proof complexity; Gas meter prover; Computer-assisted proof; Formal proof; Proof theory","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.001391216,0.0003745711,0.0003391534,0.0001098654,0.0009336745,0.001518401,0.001740584,0.0002305003,0.00004053185],"category_scores_gemma":[0.0001822447,0.0002263165,0.0002606985,0.0007492804,0.0003000149,0.0007130454,0.0004351828,0.0002736218,0.00006574473],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001363561,"about_ca_system_score_gemma":0.0006890835,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000115042,"about_ca_topic_score_gemma":0.0001474155,"domain_scores_codex":[0.9970022,0.0002407021,0.0006268305,0.001013155,0.0004357143,0.0006813256],"domain_scores_gemma":[0.9974343,0.0002850811,0.0002494259,0.001369552,0.000566519,0.00009515575],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001489284,0.00008864913,0.00021795,0.0001029,0.00009049339,0.000002497773,0.0009604593,0.0002724897,0.002670153,0.7923344,0.006709139,0.1965359],"study_design_scores_gemma":[0.0006319819,0.0003162574,0.00005862223,0.00001922761,0.0000756242,0.000009370164,0.0001288431,0.8347093,0.0155798,0.05634225,0.09178241,0.0003462333],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0005418569,0.001120715,0.9710078,0.008565821,0.004301139,0.004300762,0.000004113371,0.0001458536,0.01001193],"genre_scores_gemma":[0.9665667,0.00004096368,0.00365333,0.002719579,0.0007652988,0.0006774466,0.000008010208,0.00001736113,0.02555136],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9673545,"threshold_uncertainty_score":0.9995181,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04919463481431056,"score_gpt":0.2964831204573316,"score_spread":0.247288485643021,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}