{"id":"W6966782472","doi":"10.48448/bnyy-ne17","title":"Logical Fallacy Detection","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Fallacy; Logical consequence; Logical conjunction; Truth table; Task (project management); Logical reasoning; Set (abstract data type); Language model","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.001157892,0.0003435262,0.0003027812,0.001460941,0.0004858061,0.0001621267,0.001389727,0.0001845464,0.04927604],"category_scores_gemma":[0.0003351922,0.0003081506,0.00007853829,0.002614227,0.001446172,0.0001595277,0.0006732048,0.0006324024,0.01088513],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009256231,"about_ca_system_score_gemma":0.0005226198,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000427625,"about_ca_topic_score_gemma":0.001468348,"domain_scores_codex":[0.9961603,0.0001017783,0.0002640692,0.001067628,0.00171742,0.0006887659],"domain_scores_gemma":[0.9985619,0.00004465145,0.000296579,0.0007983894,0.00007360046,0.0002248911],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00004270986,0.0005562768,0.0004859926,0.00005164284,0.00006086474,0.0001260616,0.0001592022,0.0003065892,0.03435028,0.008525608,0.8727707,0.08256403],"study_design_scores_gemma":[0.0002615633,0.0001513896,0.0001353342,0.00001461986,0.00002384267,0.00004810185,0.0001210118,0.002979532,0.0002300082,0.0006900381,0.9948866,0.0004579924],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.000266673,0.0002478027,0.001637555,0.00009976323,0.001269886,0.0004935573,0.0001207695,0.001577763,0.9942862],"genre_scores_gemma":[0.105151,0.00003780588,0.005812682,0.0005505775,0.001229859,0.0001010438,0.0001405725,0.00152858,0.8854479],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.1221158,"threshold_uncertainty_score":0.9999371,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0282615285090122,"score_gpt":0.298867070414143,"score_spread":0.2706055419051308,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}