{"id":"W7106844164","doi":"10.48448/4f24-2a60","title":"BTC-SAM: Leveraging LLMs for Generation of Bias Test Cases for Sentiment Analysis Models","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Minnow Environmental (Canada)","funders":"","keywords":"Test (biology); Variation (astronomy); Domain (mathematical analysis); Sentiment analysis; Language model; Identity (music)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004904298,0.00150421,0.0006043262,0.001585191,0.00057463,0.001474121,0.001749864,0.001640591,0.00399515],"category_scores_gemma":[0.04199522,0.0005363357,0.001114126,0.0004888703,0.001096888,0.002134175,0.002594263,0.001611275,0.001894184],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0009283124,"about_ca_system_score_gemma":0.001990848,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002871325,"about_ca_topic_score_gemma":0.004441216,"domain_scores_codex":[0.9937637,0.003441473,0.0004003897,0.0009122636,0.001205852,0.0002762411],"domain_scores_gemma":[0.9744971,0.0182529,0.001298691,0.002889156,0.002647254,0.0004148854],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001535139,0.00108977,0.05657663,0.001406672,0.0004304033,0.002234924,0.002867139,0.2191764,0.1328605,0.02087573,0.04237213,0.5185745],"study_design_scores_gemma":[0.0001096018,0.0001781386,0.001284454,0.00005914801,0.00004306225,0.0001871979,0.0001917227,0.954317,0.02805166,0.01041529,0.005125127,0.00003759895],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.1698355,0.0003791041,0.7826504,0.001221742,0.0002146406,0.0007450316,0.002780565,0.03794378,0.004229178],"genre_scores_gemma":[0.599842,0.00009121618,0.3898349,0.0007049524,0.00007341536,0.0007019658,0.004861296,0.002692132,0.001198074],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004904298,"threshold_uncertainty_score":0.02593672,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2991158485759924,"score_gpt":0.3881714189117396,"score_spread":0.08905557033574713,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}