{"id":"W4412128547","doi":"10.1109/cism64958.2025.11060864","title":"Defining and Benchmarking a Symmetric Kappa+ Prequential Error Metric on Shifting Imbalanced Streaming Data with Label Budgets","year":2025,"lang":"en","type":"article","venue":"","topic":"Data Stream Mining Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Benchmarking; Kappa; Metric (unit); Computer science; Streaming data; Data mining; Mathematics; Economics; Operations management","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007616131,0.000287411,0.0003170709,0.001186799,0.0002696707,0.0005750052,0.002170037,0.00008808836,0.000004881997],"category_scores_gemma":[0.0005172837,0.000239952,0.00001989276,0.003290521,0.00006185569,0.001095589,0.002560946,0.0002753299,0.000004827995],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007559649,"about_ca_system_score_gemma":0.0001368474,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000268935,"about_ca_topic_score_gemma":0.00006368401,"domain_scores_codex":[0.9975383,0.00008110629,0.0003109895,0.001171334,0.0004069239,0.0004913732],"domain_scores_gemma":[0.9971874,0.0007047802,0.0001821718,0.001764884,0.00006955748,0.00009123016],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002211588,0.0001425468,0.01262752,0.00007171096,0.000100842,0.00004546432,0.0001459836,0.000009394116,0.0001110586,0.1484741,0.001807142,0.8364422],"study_design_scores_gemma":[0.003076377,0.001330133,0.04759607,0.002040785,0.0001978942,0.00008760563,0.0002386742,0.9310796,0.007528092,0.003235048,0.001980641,0.001609134],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07572647,0.0006279221,0.8826619,0.0005085984,0.0002894242,0.0004965133,0.0000642174,0.001461356,0.03816362],"genre_scores_gemma":[0.529449,0.00001602131,0.4701842,0.0001829949,0.00001856279,0.00001471338,0.0000496377,0.00001116329,0.0000736685],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.9310701,"threshold_uncertainty_score":0.9784961,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02296950921326816,"score_gpt":0.2983363388782921,"score_spread":0.2753668296650239,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}