{"id":"W4386566852","doi":"10.18653/v1/2023.eacl-demo.23","title":"CoTEVer: Chain of Thought Prompting Annotation Toolkit for Explanation Verification","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Ministry of Science and ICT, South Korea; Institute for Information and Communications Technology Promotion; Yonsei University","keywords":"Computer science; Annotation; Chain (unit); Natural language processing; Human–computer interaction; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003645443,0.002134558,0.001096407,0.00330289,0.001379425,0.003240872,0.003091384,0.002167793,0.1061162],"category_scores_gemma":[0.01819154,0.001365176,0.001916844,0.0013964,0.0008786057,0.005445237,0.005344525,0.002927422,0.03453802],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009453119,"about_ca_system_score_gemma":0.002940238,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005946417,"about_ca_topic_score_gemma":0.01141674,"domain_scores_codex":[0.9981374,0.0006196816,0.0002346194,0.0004226933,0.0004745109,0.0001110774],"domain_scores_gemma":[0.9897015,0.005818675,0.0003385343,0.001740579,0.002038914,0.0003618308],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001093622,0.0002564281,0.002250048,0.003097512,0.0002108242,0.0009791241,0.002691314,0.002917835,0.0166151,0.04435731,0.5989329,0.3265979],"study_design_scores_gemma":[0.0005502677,0.0001564304,0.002183296,0.0009114437,0.0001839292,0.0009131869,0.001324179,0.1450109,0.04337836,0.09197219,0.7131012,0.0003145917],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"methods","genre_gemma":"software","genre_scores_codex":[0.002414865,0.0004253395,0.5463779,0.000539523,0.0005523527,0.0004506959,0.01631859,0.4248283,0.008092487],"genre_scores_gemma":[0.06623081,0.000534386,0.8005642,0.0005814125,0.0002007586,0.001369118,0.06095804,0.05167356,0.01788772],"genre_candidate":"software","genre_consensus":null,"teacher_disagreement_score":0.1061162,"threshold_uncertainty_score":0.3549939,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05112360901474697,"score_gpt":0.2895949163161962,"score_spread":0.2384713073014492,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}