{"id":"W4319300892","doi":"10.1109/wacv56688.2023.00278","title":"TeST: Test-time Self-Training under Distribution Shift","year":2023,"lang":"en","type":"article","venue":"2023 IEEE/CVF Winter Conference on Applications of Computer Vision (WACV)","topic":"Domain Adaptation and Few-Shot Learning","field":"Computer Science","cited_by":26,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Test data; Artificial intelligence; Test (biology); Machine learning; Inference; Adaptation (eye); Data mining; Pattern recognition (psychology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002735503,0.001845796,0.001062705,0.0007035228,0.0005117767,0.0008802094,0.003551486,0.001916949,0.004324446],"category_scores_gemma":[0.01097512,0.0005877138,0.0009554469,0.0006698301,0.0008839226,0.002706233,0.002501287,0.003340214,0.002558999],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000844117,"about_ca_system_score_gemma":0.00139174,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005789763,"about_ca_topic_score_gemma":0.007130035,"domain_scores_codex":[0.9987223,0.0003336756,0.0000692882,0.0004620901,0.000273072,0.0001394912],"domain_scores_gemma":[0.9963048,0.001370698,0.0002009982,0.001339028,0.0005978643,0.0001867341],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001198574,0.000864257,0.009296238,0.0002743246,0.0004726443,0.0002887093,0.0001786725,0.277906,0.02204043,0.002794749,0.03792687,0.6467584],"study_design_scores_gemma":[0.00006489519,0.0002240032,0.001197798,0.00001819925,0.00003038216,0.0001665379,0.00004655278,0.9809327,0.01172669,0.002824276,0.002743436,0.00002449019],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.179598,0.001666775,0.7558697,0.0007852769,0.0007251613,0.0004621161,0.001451846,0.05337493,0.006066104],"genre_scores_gemma":[0.7022183,0.0002986307,0.2775957,0.001385729,0.000150105,0.0004871163,0.007940505,0.002461703,0.007462228],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005789763,"threshold_uncertainty_score":0.01446688,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02988191807964998,"score_gpt":0.2922269165930714,"score_spread":0.2623449985134214,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}