{"id":"W4386566852","doi":"10.18653/v1/2023.eacl-demo.23","title":"CoTEVer: Chain of Thought Prompting Annotation Toolkit for Explanation Verification","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"Ministry of Science and ICT, South Korea; Institute for Information and Communications Technology Promotion; Yonsei University","keywords":"Computer science; Annotation; Chain (unit); Natural language processing; Human–computer interaction; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.000411235,0.00004209824,0.00005901298,0.00009277781,0.00004842574,0.00002936053,0.0001959331,0.00002733647,0.000002653264],"category_scores_gemma":[0.00008497742,0.00004103336,0.00002176667,0.0003065964,0.000004789375,0.0003285967,0.00003250736,0.00002062767,0.00001153675],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0000177223,"about_ca_system_score_gemma":0.00002331969,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001157953,"about_ca_topic_score_gemma":0.00000433625,"domain_scores_codex":[0.999391,0.00001734681,0.0001794454,0.0001756635,0.0001372918,0.00009924889],"domain_scores_gemma":[0.9995231,0.00007970146,0.00008363376,0.0002021422,0.00009687497,0.00001453832],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000007658491,0.00002493511,0.0005192666,0.0001031379,0.000009812823,4.475483e-7,0.002470368,0.009362525,0.02887891,0.6453196,0.001284881,0.3120184],"study_design_scores_gemma":[0.0001498502,0.00001943898,0.002516042,0.00001045211,0.000001373341,3.936978e-7,0.00007774819,0.9769745,0.01540573,0.003990494,0.0007997919,0.00005419071],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07412942,0.000005424208,0.9238445,0.0008355367,0.0001547361,0.0002380896,0.000001485284,0.0001914695,0.0005994034],"genre_scores_gemma":[0.8311483,0.000003498751,0.1677818,0.00004450345,0.00004964869,0.00005260449,0.00002451042,0.000004213663,0.0008908968],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.967612,"threshold_uncertainty_score":0.1673292,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05112360901474697,"score_gpt":0.2895949163161962,"score_spread":0.2384713073014492,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}