{"id":"W7077908071","doi":"10.48448/48ar-ye13","title":"Single Ground Truth Is Not Enough: Adding Flexibility to Aspect-Based Sentiment Analysis Evaluation","year":2025,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Pipeline (software); Flexibility (engineering); Sentiment analysis; Task (project management); Ground truth; Set (abstract data type); Code (set theory); Annotation","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.002723525,0.0003684011,0.0004709527,0.001280776,0.0003576232,0.0005032332,0.002120236,0.0001948562,0.001845111],"category_scores_gemma":[0.0007885116,0.0003518947,0.0001841453,0.005388113,0.0002902473,0.0002079381,0.0006645122,0.0002220622,0.0001103231],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005627667,"about_ca_system_score_gemma":0.001234636,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004486121,"about_ca_topic_score_gemma":0.0001411093,"domain_scores_codex":[0.9953524,0.0001240821,0.0004496772,0.001811574,0.001640687,0.0006215467],"domain_scores_gemma":[0.9969737,0.0001807502,0.0003252913,0.001738668,0.0005601613,0.0002214381],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00009884105,0.004284945,0.001541078,0.001049646,0.002756539,0.00009490643,0.005696246,0.1364222,0.03372869,0.1238393,0.2204621,0.4700256],"study_design_scores_gemma":[0.000492203,0.0001400059,0.0004146484,0.0002069845,0.0004991033,0.00000194661,0.0001593552,0.852178,0.02468611,0.003614755,0.1167015,0.0009054432],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0004436148,0.0001047998,0.5228899,0.006299285,0.001000267,0.0007619095,0.00005999905,0.0004693748,0.4679709],"genre_scores_gemma":[0.7229163,0.00000255976,0.1144766,0.002559062,0.0002724298,0.00006735227,0.00009301363,0.00002812409,0.1595846],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.7224727,"threshold_uncertainty_score":0.9998933,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04954350575419218,"score_gpt":0.3179577340840264,"score_spread":0.2684142283298342,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}