{"id":"W4389010521","doi":"10.18653/v1/2023.sigdial-1.14","title":"The Road to Quality is Paved with Good Revisions: A Detailed Evaluation Methodology for Revision Policies in Incremental Sequence Labelling","year":2023,"lang":"en","type":"article","venue":"","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Atomic Energy of Canada Limited","keywords":"Computer science; Prefix; Sequence (biology); Encoder; Labelling; Transformer; Quality (philosophy); Engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03360151,0.001403566,0.001273365,0.003304218,0.001104103,0.004557143,0.002782372,0.002463807,0.00177907],"category_scores_gemma":[0.1662657,0.0008978825,0.0009581927,0.002106002,0.002939669,0.007310363,0.003344412,0.002996353,0.0005947126],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003745971,"about_ca_system_score_gemma":0.004253815,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007485673,"about_ca_topic_score_gemma":0.008112233,"domain_scores_codex":[0.9591587,0.02080063,0.004315857,0.003874623,0.01058299,0.001267126],"domain_scores_gemma":[0.8208361,0.1057309,0.01494058,0.02889192,0.02557605,0.004024393],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001986356,0.0008480175,0.03257353,0.001352515,0.0004612929,0.0002508148,0.003568367,0.2718223,0.02925773,0.05771204,0.005007612,0.5951595],"study_design_scores_gemma":[0.000105198,0.001429852,0.005613454,0.0001761381,0.0001410028,0.0002859535,0.0003971099,0.9169579,0.03265239,0.03719946,0.004875144,0.0001663696],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.06844691,0.0005552587,0.9224662,0.0005315355,0.00008228639,0.0005517618,0.0004312621,0.004484517,0.002450296],"genre_scores_gemma":[0.5422049,0.0002166566,0.4547195,0.0001659139,0.00004233978,0.0003303032,0.0006544428,0.0007233115,0.0009425927],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.03360151,"threshold_uncertainty_score":0.1777039,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2383758045381053,"score_gpt":0.4762702837497301,"score_spread":0.2378944792116249,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}