{"id":"W4404970287","doi":"10.1007/978-3-031-78495-8_28","title":"Improving Sampling Methods for Fine-Tuning SentenceBERT in Text Streams","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal","funders":"","keywords":"Computer science; STREAMS; Sampling (signal processing); Operating system; Computer vision","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.002262778,0.0005092027,0.0005213931,0.001291541,0.0002385248,0.001058904,0.002271749,0.0003537619,0.000005717031],"category_scores_gemma":[0.0003112454,0.0004771578,0.0001817655,0.0009676848,0.0002402884,0.000685656,0.001113033,0.001054274,0.00001443124],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004468164,"about_ca_system_score_gemma":0.0003975103,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001960674,"about_ca_topic_score_gemma":0.0003636794,"domain_scores_codex":[0.9963195,0.00004023828,0.0005922791,0.00183087,0.0004822233,0.0007348444],"domain_scores_gemma":[0.9970431,0.001468934,0.0002297518,0.0009655746,0.000163217,0.000129463],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00000458272,0.000008628515,0.00001574831,0.00008846888,0.000006364613,0.00001704152,0.0006892512,0.01997268,0.0009702821,0.007789745,0.000003785174,0.9704334],"study_design_scores_gemma":[0.0001520846,0.0001174296,0.00001536625,0.000643579,0.000008058078,0.00004363627,4.095141e-7,0.8060587,0.00193438,0.1889593,0.001601876,0.0004651941],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00005474615,0.0009484996,0.9921684,0.0005534026,0.004761252,0.000497018,0.00000383002,0.0002347892,0.0007780815],"genre_scores_gemma":[0.02181276,0.00001812263,0.9765483,0.0004122795,0.0006730431,0.00003061714,0.000004483875,0.00005006845,0.000450336],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.9699683,"threshold_uncertainty_score":0.9999781,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03485092582465372,"score_gpt":0.3180603402263149,"score_spread":0.2832094144016611,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}