{"id":"W4393584161","doi":"10.5281/zenodo.7599666","title":"What Makes Sentences Semantically Related? A Textual Relatedness Dataset and Empirical Study","year":2021,"lang":"en","type":"dataset","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"National Research Council Canada; University of Toronto","funders":"","keywords":"Natural language processing; Computer science; Artificial intelligence; Linguistics; Information retrieval; Philosophy","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00202817,0.0005543607,0.000389701,0.003540006,0.001307408,0.000933437,0.0008125153,0.00110667,0.005850604],"category_scores_gemma":[0.01602876,0.0001756312,0.0006510178,0.003361647,0.0005663848,0.001928076,0.001709068,0.001064067,0.004189051],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0008126709,"about_ca_system_score_gemma":0.0006947218,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004179252,"about_ca_topic_score_gemma":0.008336642,"domain_scores_codex":[0.9971004,0.001301658,0.0004313691,0.0004511359,0.0005811338,0.0001342388],"domain_scores_gemma":[0.9883254,0.006011854,0.001383627,0.001380493,0.002002865,0.0008957961],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"observational","study_design_scores_codex":[0.00215927,0.002939398,0.1267164,0.005426712,0.0004452357,0.001284895,0.007264659,0.002618229,0.012848,0.007899148,0.6892896,0.1411085],"study_design_scores_gemma":[0.00106722,0.001530522,0.597734,0.001010841,0.0004491238,0.003066849,0.01446404,0.02644187,0.009896577,0.009396808,0.3345853,0.0003568393],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5698457,0.002295644,0.009452406,0.002730529,0.0002626784,0.001669735,0.3915558,0.001535695,0.02065177],"genre_scores_gemma":[0.2706737,0.0004034724,0.01805498,0.0006048484,0.0001849806,0.001983913,0.7033794,0.0001436939,0.004570979],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.005850604,"threshold_uncertainty_score":0.01957226,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03649961458743942,"score_gpt":0.3093099756010593,"score_spread":0.2728103610136199,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}