{"id":"W3174432697","doi":"10.18653/v1/2021.acl-short.51","title":"Exploring Listwise Evidence Reasoning with T5 for Fact Verification","year":2021,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":44,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canada First Research Excellence Fund","keywords":"Computer science; Computational linguistics; Volume (thermodynamics); Natural language processing; Joint (building); Artificial intelligence; Cognitive science; Psychology; Engineering","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01051296,0.001434927,0.001658654,0.003471915,0.00253487,0.007129556,0.005875396,0.002871864,0.01865436],"category_scores_gemma":[0.05411778,0.001819193,0.006147896,0.002690617,0.00323686,0.01684134,0.009023941,0.005201864,0.004507706],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002682261,"about_ca_system_score_gemma":0.005660434,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01194027,"about_ca_topic_score_gemma":0.02336268,"domain_scores_codex":[0.9895573,0.003998149,0.0009145586,0.002127846,0.002540017,0.0008621179],"domain_scores_gemma":[0.9448861,0.04353536,0.001651875,0.005965524,0.003146935,0.000814282],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001882411,0.0006927579,0.006025154,0.002014621,0.0009463456,0.001876313,0.002006476,0.07676356,0.009465312,0.4893731,0.04195459,0.3669993],"study_design_scores_gemma":[0.0001659347,0.00009016846,0.0002545831,0.0001700025,0.0002517465,0.0002782092,0.0003674008,0.3766852,0.008934543,0.5981706,0.01456623,0.00006540692],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01213997,0.000903358,0.9703174,0.002642698,0.0001799386,0.0002875339,0.001053248,0.006925678,0.005550244],"genre_scores_gemma":[0.2445008,0.0004648227,0.7475821,0.0008092443,0.0002024557,0.0002015599,0.002757542,0.0008762899,0.002605221],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01865436,"threshold_uncertainty_score":0.06240499,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2909852846843825,"score_gpt":0.3030645874900807,"score_spread":0.01207930280569819,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}