{"id":"W7128477045","doi":"10.26615/978-954-452-101-1-004","title":"FLARE: An Error Analysis Framework for Diagnosing LLM Classification Failures","year":2025,"lang":"","type":"article","venue":"","topic":"VLSI and Analog Circuit Testing","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Guelph","funders":"","keywords":"Error analysis; Error detection and correction; Feature (linguistics); Statistical analysis; Reliability (semiconductor)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01648049,0.003172527,0.001460266,0.007499283,0.001683578,0.004484515,0.005079866,0.003061702,0.006899099],"category_scores_gemma":[0.07618391,0.0009202005,0.002500708,0.001818863,0.002830199,0.00689328,0.005469926,0.003944245,0.002463444],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003130802,"about_ca_system_score_gemma":0.006172211,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01425032,"about_ca_topic_score_gemma":0.02222701,"domain_scores_codex":[0.9847128,0.005198849,0.001284393,0.002208211,0.005811173,0.0007846028],"domain_scores_gemma":[0.9464278,0.03349874,0.004989202,0.005764446,0.008516232,0.0008034943],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.001159133,0.0006858804,0.04674487,0.001419846,0.0007571697,0.001086336,0.002011653,0.1483238,0.01247316,0.06573056,0.100032,0.6195756],"study_design_scores_gemma":[0.0000749008,0.0002509441,0.002305503,0.0002776675,0.0001363524,0.0004864311,0.0003350787,0.8937966,0.0154501,0.07086934,0.0158853,0.0001317606],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01069333,0.0005585456,0.9337851,0.001908127,0.000144567,0.0003460576,0.002144663,0.04828716,0.002132498],"genre_scores_gemma":[0.2465594,0.0003128772,0.7411644,0.001401859,0.0002273899,0.0003782983,0.004244791,0.002889378,0.002821527],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01648049,"threshold_uncertainty_score":0.0871582,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06904225745636498,"score_gpt":0.3627234665714213,"score_spread":0.2936812091150564,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}