{"id":"W2965349778","doi":"10.4230/lipics.icalp.2019.84","title":"Short Proofs Are Hard to Find","year":2019,"lang":"en","type":"article","venue":"DROPS (Schloss Dagstuhl – Leibniz Center for Informatics)","topic":"Educational Assessment and Pedagogy","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Natural Sciences and Engineering Research Council of Canada; University of Toronto","keywords":"Mathematical proof; Computer science; Mathematics","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.000796892,0.0002430963,0.0003267573,0.0001928264,0.0005368263,0.0003380748,0.0006557031,0.0001977476,0.00064364],"category_scores_gemma":[0.0001046898,0.0002301041,0.0001756933,0.0003134352,0.00009152349,0.0009765734,0.0001289751,0.0002565103,0.001603064],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002197587,"about_ca_system_score_gemma":0.0003096047,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00010533,"about_ca_topic_score_gemma":0.0003797221,"domain_scores_codex":[0.9976856,0.00004573925,0.0006206886,0.0002271158,0.0006940174,0.0007268512],"domain_scores_gemma":[0.9985693,0.0001649733,0.0001923711,0.0003902458,0.0003637706,0.0003193074],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001453894,0.0005075616,0.7684717,0.0004358542,0.0001767562,0.000001612476,0.08793794,0.00008375957,0.0000846387,0.04874042,0.07671916,0.01669527],"study_design_scores_gemma":[0.0006158781,0.0001347269,0.07985502,0.0001045397,0.00002403099,0.000002495008,0.02281687,0.000213726,0.00008476905,0.0008088103,0.8948835,0.0004556078],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9134016,0.00002029594,0.0006983661,0.004301107,0.002130845,0.002179852,0.000299844,0.000125392,0.07684267],"genre_scores_gemma":[0.9712078,0.00001775036,0.003146509,0.002245334,0.0005946481,0.0002016656,0.0003062719,0.00003168015,0.02224833],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.8181643,"threshold_uncertainty_score":0.9991743,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04533790389707443,"score_gpt":0.3725974693966653,"score_spread":0.3272595654995908,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}