{"id":"W7128477045","doi":"10.26615/978-954-452-101-1-004","title":"FLARE: An Error Analysis Framework for Diagnosing LLM Classification Failures","year":2025,"lang":"","type":"article","venue":"","topic":"VLSI and Analog Circuit Testing","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"","keywords":"Error analysis; Error detection and correction; Feature (linguistics); Statistical analysis; Reliability (semiconductor)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.0009337132,0.0003811742,0.0005890341,0.0008709153,0.001097131,0.001597077,0.001539025,0.0003590095,0.0000980237],"category_scores_gemma":[0.001483154,0.0003833998,0.000521356,0.005547518,0.0001217556,0.001103175,0.000212247,0.0003679916,0.00002191472],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000137778,"about_ca_system_score_gemma":0.0003560944,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0002206054,"about_ca_topic_score_gemma":0.000355448,"domain_scores_codex":[0.9964268,0.0002121206,0.0008251411,0.001412286,0.0003699653,0.0007536425],"domain_scores_gemma":[0.995767,0.001748522,0.0003335442,0.001445817,0.000475596,0.0002294958],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[9.857333e-7,0.0003198355,0.02335274,0.0000811838,0.0004867866,0.000003090293,0.0006988444,0.001425592,0.0003154968,0.4793529,0.0004758631,0.4934867],"study_design_scores_gemma":[0.0002143068,0.000139724,0.04300459,0.0002183532,0.00104322,9.722296e-7,0.0006028638,0.9071313,0.0005972887,0.04507859,0.001509579,0.0004592576],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.006739066,0.0006146028,0.98053,0.008765001,0.0008189128,0.0004887315,0.00001175286,0.0003051328,0.001726818],"genre_scores_gemma":[0.9068471,0.00002748957,0.09052012,0.001532831,0.0002638081,0.00008478622,0.00003073755,0.00001534287,0.0006777635],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9057057,"threshold_uncertainty_score":0.9998618,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06904225745636498,"score_gpt":0.3627234665714213,"score_spread":0.2936812091150564,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}