{"id":"W4310419110","doi":"10.48550/arxiv.2211.14912","title":"Impact of Strategic Sampling and Supervision Policies on Semi-supervised Learning","year":2022,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Domain Adaptation and Few-Shot Learning","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Mitacs","keywords":"Representativeness heuristic; Computer science; Context (archaeology); Sample (material); Artificial intelligence; Machine learning; Labelling; Set (abstract data type); Selection (genetic algorithm); Training set; Labeled data; Representation (politics); Process (computing); Quality (philosophy); Supervised learning; Ask price; Data set; Variety (cybernetics); Statistics; Mathematics; Artificial neural network; Psychology","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01719171,0.001781506,0.001737666,0.0005946967,0.001352808,0.001653367,0.003003883,0.002708585,0.001274441],"category_scores_gemma":[0.06954376,0.0006909822,0.0005783915,0.0005479841,0.003649501,0.005008896,0.003111049,0.003552592,0.0006078157],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001817932,"about_ca_system_score_gemma":0.003495617,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00431752,"about_ca_topic_score_gemma":0.005553065,"domain_scores_codex":[0.9902408,0.006285154,0.0003288563,0.001517504,0.001163768,0.0004638322],"domain_scores_gemma":[0.9401158,0.04521955,0.002397147,0.007310563,0.003262406,0.001694523],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002509383,0.001105852,0.0161242,0.0005220667,0.0002751299,0.0002428607,0.0008484013,0.7175513,0.006288553,0.0384749,0.007075964,0.2089814],"study_design_scores_gemma":[0.00009812893,0.0002600419,0.0006516278,0.00005091896,0.00002666964,0.0000796207,0.00009377424,0.9740608,0.003086216,0.02091181,0.0006575697,0.00002264753],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2581552,0.004772053,0.7216972,0.003278441,0.0002830874,0.0004522007,0.0002702981,0.00318451,0.007906971],"genre_scores_gemma":[0.9107813,0.0005644322,0.08533264,0.001128848,0.0001354216,0.0002387595,0.0003608573,0.0002108939,0.001246972],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01719171,"threshold_uncertainty_score":0.09091955,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1721840926498918,"score_gpt":0.2534048193886386,"score_spread":0.08122072673874686,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}