{"id":"W4226000139","doi":"10.1093/imaiai/iaae002","title":"PACMAN: PAC-style bounds accounting for the Mismatch between Accuracy and Negative log-loss","year":2024,"lang":"en","type":"preprint","venue":"Information and Inference A Journal of the IMA","topic":"Machine Learning and Data Classification","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"École de Technologie Supérieure; McGill University","funders":"Universidad de Buenos Aires; Consejo Nacional de Investigaciones Científicas y Técnicas","keywords":"Generalization; Computer science; Algorithm; Function (biology); Metric (unit); Generalization error; Cross entropy; Mathematics; Artificial intelligence; Artificial neural network; Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02484842,0.004331615,0.003342829,0.00383767,0.001699534,0.006832949,0.006726512,0.005011387,0.007431084],"category_scores_gemma":[0.1478761,0.001424662,0.002173406,0.003392784,0.007198132,0.01537319,0.009457663,0.01165477,0.002236621],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.005193604,"about_ca_system_score_gemma":0.003756423,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001823916,"about_ca_topic_score_gemma":0.001235518,"domain_scores_codex":[0.9832877,0.007312021,0.0006203622,0.00264647,0.004962434,0.001171133],"domain_scores_gemma":[0.8838221,0.09070762,0.00547465,0.009789803,0.007827314,0.002378505],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0006182335,0.0002454533,0.002927596,0.0006724859,0.0002265415,0.0004503034,0.0004238554,0.2804542,0.002663747,0.5941796,0.01646507,0.1006731],"study_design_scores_gemma":[0.00001523635,0.0001470865,0.0004273937,0.0001645735,0.00004381213,0.0002199785,0.00003724203,0.7505313,0.002061056,0.2434851,0.002823692,0.00004362988],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":"theoretical_or_conceptual","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.009195805,0.003957774,0.9745256,0.00185159,0.0003528356,0.00009274374,0.0002019161,0.0006074649,0.009214236],"genre_scores_gemma":[0.5282773,0.005416223,0.4416634,0.003994341,0.002855089,0.001079036,0.001145238,0.002304432,0.01326492],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02484842,"threshold_uncertainty_score":0.1314126,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02670544377808226,"score_gpt":0.3117600080436237,"score_spread":0.2850545642655414,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}