{"id":"W3206282310","doi":"10.18653/v1/2023.eacl-main.55","title":"What Makes Sentences Semantically Related? A Textual Relatedness Dataset and Empirical Study","year":2023,"lang":"en","type":"preprint","venue":"","topic":"Topic Modeling","field":"Computer Science","cited_by":14,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto; National Research Council Canada; Institute for Work & Health","funders":"","keywords":"Computer science; Natural language processing; Semantic similarity; Annotation; Artificial intelligence; Automatic summarization; Intuition; Sentence; Similarity (geometry); Information retrieval; Psychology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002993324,0.0005800289,0.0003842496,0.003838906,0.001418435,0.0007647754,0.00101989,0.001314382,0.002125209],"category_scores_gemma":[0.0166787,0.000176291,0.0005785834,0.003950391,0.0008317493,0.001619021,0.001387365,0.001042262,0.001717779],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008155287,"about_ca_system_score_gemma":0.0007169348,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004352865,"about_ca_topic_score_gemma":0.008346739,"domain_scores_codex":[0.9962143,0.00180492,0.0004012175,0.0006760893,0.0007452677,0.0001581331],"domain_scores_gemma":[0.9846021,0.008051129,0.002003869,0.001964839,0.002179244,0.001198756],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.00321748,0.006246567,0.22439,0.006329851,0.0006964377,0.003482846,0.01136625,0.01108311,0.03661695,0.0148686,0.4464487,0.2352533],"study_design_scores_gemma":[0.00106198,0.002114992,0.6150546,0.0005063589,0.000351567,0.007218818,0.01010894,0.05664534,0.01800179,0.01053758,0.278093,0.0003049987],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.8578566,0.002071379,0.008533373,0.001313669,0.0001364041,0.0005747357,0.1181359,0.00110753,0.0102704],"genre_scores_gemma":[0.5383066,0.0005146842,0.0245895,0.0004894673,0.0002131829,0.0009221053,0.431438,0.0001793424,0.003347179],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.004352865,"threshold_uncertainty_score":0.01583046,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09582439625974269,"score_gpt":0.3546847947762765,"score_spread":0.2588603985165338,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}