{"id":"W6891643041","doi":"10.48448/se5p-7p38","title":"Long-form evaluation of model editing","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Software portability; Generative grammar; Protocol (science); Classifier (UML); Language model; Set (abstract data type); Natural language; Protocol analysis","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.006201288,0.0003083686,0.0003267008,0.001778936,0.00007735994,0.0001186768,0.0008778158,0.0002008369,0.001328621],"category_scores_gemma":[0.0005892387,0.0002750272,0.00007649261,0.001865364,0.001204775,0.0002939903,0.0002944413,0.0003231426,0.005540917],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0006344571,"about_ca_system_score_gemma":0.002236871,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001569642,"about_ca_topic_score_gemma":0.00080236,"domain_scores_codex":[0.9942555,0.00004417279,0.0004400646,0.0008138465,0.003942025,0.0005043387],"domain_scores_gemma":[0.9979612,0.0000212203,0.0004885072,0.0007095468,0.0006903645,0.0001291441],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000006235669,0.0001830767,0.00006642264,0.0004772682,0.00009060754,0.000006154014,0.0007390797,0.02981669,0.005995882,0.01266839,0.9107773,0.03917293],"study_design_scores_gemma":[0.0002266018,0.00002649798,0.00000889694,0.0006258562,0.0002271004,0.000005938343,0.0001091174,0.9821886,0.0004985675,0.01156762,0.004233558,0.000281653],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.0003873102,0.001061757,0.004637348,0.00007868409,0.001318724,0.0006990791,0.000288154,0.0005063095,0.9910226],"genre_scores_gemma":[0.326019,0.0000289566,0.02678351,0.00009945172,0.002499139,0.00008531995,0.0002289713,0.002434755,0.6418209],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.9523719,"threshold_uncertainty_score":0.9999702,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09369404228987084,"score_gpt":0.3853238918115702,"score_spread":0.2916298495216994,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}