{"id":"W3091895957","doi":"10.1007/978-3-030-60887-3_19","title":"Evaluation of Similarity Measures in a Benchmark for Spanish Paraphrasing Detection","year":2020,"lang":"en","type":"book-chapter","venue":"Lecture notes in computer science","topic":"Topic Modeling","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"Polytechnique Montréal","funders":"","keywords":"Paraphrase; Similarity (geometry); Computer science; Natural language processing; Artificial intelligence; Vocabulary; Benchmark (surveying); Semantic similarity; Boundary (topology); Computation; Information retrieval; Linguistics; Algorithm; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.004217324,0.0003120604,0.00045946,0.0006299003,0.0001153298,0.0001836777,0.001530768,0.0002660638,0.000003579177],"category_scores_gemma":[0.0006551286,0.0003234041,0.0001237385,0.0005684181,0.0001932728,0.0004587608,0.0004344788,0.0004892574,0.000001164554],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004513298,"about_ca_system_score_gemma":0.0008012546,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004842431,"about_ca_topic_score_gemma":0.0006159669,"domain_scores_codex":[0.9958974,0.00009910121,0.0006126137,0.001245641,0.001775772,0.0003695161],"domain_scores_gemma":[0.9977446,0.000394236,0.0003312904,0.000774283,0.0006717293,0.00008381849],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000006660939,0.00001385639,0.00005885466,0.00004303392,0.000007131027,0.000003140595,0.0007459558,0.1359545,0.0006151099,0.002660142,8.855147e-7,0.8598908],"study_design_scores_gemma":[0.0002914257,0.0000837528,0.000198514,0.000202773,0.00001623528,0.000005609098,2.209142e-7,0.7494134,0.003829629,0.2456802,0.00004847541,0.0002296923],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.0009431799,0.0003730677,0.9958911,0.0003722222,0.001000792,0.0008358472,0.000003200496,0.00005409983,0.0005264894],"genre_scores_gemma":[0.749346,0.000008207972,0.2502004,0.0002332057,0.0001762436,0.00001868697,0.000001626725,0.00001302274,0.000002593266],"genre_candidate":"methods","genre_consensus":null,"teacher_disagreement_score":0.8596611,"threshold_uncertainty_score":0.9999218,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.0693253929892796,"score_gpt":0.2915309333346245,"score_spread":0.222205540345345,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}