{"id":"W4404970287","doi":"10.1007/978-3-031-78495-8_28","title":"Improving Sampling Methods for Fine-Tuning SentenceBERT in Text Streams","year":2024,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Spam and Phishing Detection","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"École de Technologie Supérieure; Université du Québec à Montréal","funders":"","keywords":"Computer science; STREAMS; Sampling (signal processing); Operating system; Computer vision","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003586023,0.001871909,0.002646737,0.001889854,0.0007790334,0.001603162,0.002636174,0.002410472,0.003777629],"category_scores_gemma":[0.01338039,0.000944135,0.001107644,0.00157995,0.0006548807,0.002688452,0.001879662,0.002769041,0.002946486],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008962678,"about_ca_system_score_gemma":0.001202873,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.006927063,"about_ca_topic_score_gemma":0.01113348,"domain_scores_codex":[0.9982949,0.0004662875,0.0001688403,0.0005272891,0.0003785457,0.0001641261],"domain_scores_gemma":[0.9888498,0.008711596,0.000322994,0.0005844702,0.001224907,0.0003061986],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001244304,0.0005600547,0.003485949,0.0002962052,0.0002144587,0.000161696,0.0002267457,0.1180445,0.02895518,0.002324636,0.01109863,0.8333876],"study_design_scores_gemma":[0.00004078275,0.00005831405,0.0003326143,0.000007952072,0.00002348756,0.0000311005,0.00002022824,0.9945891,0.002685988,0.001778557,0.0004225016,0.000009375426],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.030367,0.001389799,0.960995,0.000227357,0.0002905057,0.0001451884,0.0003569502,0.005625103,0.0006029829],"genre_scores_gemma":[0.3589129,0.000772415,0.6275964,0.0007845166,0.001136057,0.0004915199,0.003728007,0.00129429,0.005283938],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.006927063,"threshold_uncertainty_score":0.01896495,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03485092582465372,"score_gpt":0.3180603402263149,"score_spread":0.2832094144016611,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}