{"id":"W6891809573","doi":"10.48448/fjt5-8369","title":"Rethinking Label Refurbishment: Model Robustness under Label Noise","year":2023,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McMaster University","funders":"","keywords":"Robustness (evolution); Generalization; Artificial neural network; Regularization (linguistics); Deep neural networks; Noise (video); Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.003458213,0.001130805,0.0009791474,0.002948517,0.0008292622,0.000627176,0.003534363,0.0008659275,0.0006144877],"category_scores_gemma":[0.0005078911,0.001086993,0.0001253221,0.005686534,0.002944099,0.0008562267,0.001381673,0.00137535,0.007872987],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001699393,"about_ca_system_score_gemma":0.002434659,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0007219072,"about_ca_topic_score_gemma":0.005167751,"domain_scores_codex":[0.9904739,0.0001321307,0.0008845507,0.002614325,0.003748023,0.002147019],"domain_scores_gemma":[0.9953637,0.0001694127,0.0009193057,0.002413856,0.0005041691,0.0006296019],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00006625246,0.001203838,0.00009296909,0.0003570094,0.0003097804,0.0001930944,0.00117074,0.1982123,0.009496262,0.04772024,0.7287254,0.01245206],"study_design_scores_gemma":[0.00187751,0.00008305973,0.00001671702,0.001186406,0.0001808018,0.00002526029,0.0001753434,0.9703466,0.0002695978,0.01430825,0.009664224,0.001866264],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.002928764,0.001780221,0.03172528,0.002902967,0.006591785,0.003742976,0.002068084,0.02614661,0.9221133],"genre_scores_gemma":[0.01450003,0.0002640757,0.08704264,0.001125629,0.001053917,0.0001415872,0.0002231158,0.008555548,0.8870934],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.7721342,"threshold_uncertainty_score":0.9997693,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08127156378226208,"score_gpt":0.3313697929390731,"score_spread":0.250098229156811,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}