{"id":"W4389519126","doi":"10.18653/v1/2023.emnlp-main.448","title":"OssCSE: Overcoming Surface Structure Bias in Contrastive Learning for Unsupervised Sentence Embedding","year":2023,"lang":"en","type":"article","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Queen's University","funders":"","keywords":"Natural language processing; Embedding; Sentence; Computer science; Artificial intelligence; Linguistics; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0004461679,0.0001410339,0.0001957705,0.0001206653,0.0001300385,0.0001444503,0.0004759924,0.00007088571,0.00001647721],"category_scores_gemma":[0.0003183262,0.000132707,0.00005193369,0.000595918,0.00001626855,0.0004231027,0.0002266901,0.0002161853,0.0000130686],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006297475,"about_ca_system_score_gemma":0.00005585007,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001186822,"about_ca_topic_score_gemma":0.00006830128,"domain_scores_codex":[0.9985666,0.00006667314,0.0002483056,0.0004683156,0.0002100079,0.0004400969],"domain_scores_gemma":[0.9989971,0.0005687371,0.00005696893,0.0002455376,0.00006883253,0.00006282932],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001539191,0.00001261968,0.03383797,0.0000608195,0.00001949484,0.00003394088,0.00493117,0.8992588,0.02299974,0.01914531,0.000195715,0.01948902],"study_design_scores_gemma":[0.0005224766,0.00001984492,0.003437857,0.0000493576,0.000001762032,0.000003472511,0.0006159345,0.9911824,0.002043444,0.001737127,0.0002230309,0.000163257],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5007262,0.00001758335,0.4982271,0.0003156109,0.0001875469,0.0001668629,0.000001306579,0.000231614,0.00012624],"genre_scores_gemma":[0.9352838,0.000005677,0.06394134,0.0001274114,0.00004180954,0.000006431851,0.000003705423,0.00001185713,0.000577994],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4345576,"threshold_uncertainty_score":0.5411633,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.05164863126802949,"score_gpt":0.3009311128956684,"score_spread":0.2492824816276389,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}