{"id":"W4382052297","doi":"10.4050/f-0079-2023-18094","title":"Towards an Evaluation Process for Regime Recognition Approaches: Addressing Variability in Labeling Training Data","year":2023,"lang":"en","type":"article","venue":"","topic":"Target Tracking and Data Fusion in Sensor Networks","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada","funders":"","keywords":"Computer science; USable; Process (computing); Task (project management); Test data; Machine learning; Training set; Artificial intelligence; Flight test; Flight training; Flight simulator; Path (computing); Test (biology); Data mining; Simulation; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.06043152,0.001839995,0.001987626,0.003472662,0.001770361,0.005957688,0.002981242,0.0038704,0.001330849],"category_scores_gemma":[0.1561627,0.0007227696,0.001155382,0.002084822,0.00273389,0.005829978,0.00466993,0.004580203,0.0008580625],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002578683,"about_ca_system_score_gemma":0.003832494,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004081298,"about_ca_topic_score_gemma":0.003919261,"domain_scores_codex":[0.956719,0.02616455,0.00260229,0.004254732,0.009323577,0.0009358729],"domain_scores_gemma":[0.8766314,0.07710448,0.00845851,0.01084357,0.02534564,0.001616318],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001040908,0.00103528,0.02012268,0.0007394438,0.0004230077,0.0001983319,0.001616091,0.2002611,0.0221113,0.03312646,0.005240927,0.7140844],"study_design_scores_gemma":[0.00006602427,0.0005068786,0.002775849,0.0002104548,0.00006210989,0.00007238241,0.0004387413,0.946412,0.01885413,0.02780086,0.002736558,0.00006386983],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.02048442,0.0002628669,0.9761342,0.0004622997,0.00004132454,0.0004666075,0.0001534869,0.0008673379,0.001127581],"genre_scores_gemma":[0.2350284,0.0001739252,0.7616922,0.0003597315,0.00007379545,0.0008074414,0.0007769937,0.0003765284,0.0007110671],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.06043152,"threshold_uncertainty_score":0.3195963,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.5960908505467749,"score_gpt":0.4150876066118485,"score_spread":0.1810032439349264,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}