{"id":"W6891776892","doi":"10.48448/66tx-q829","title":"UAlberta at SemEval-2024 Task 1: A Potpourri of Methods for Quantifying Multilingual Semantic Textual Relatedness and Similarity","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Alberta","funders":"","keywords":"Semantic similarity; Task (project management); Variety (cybernetics); Similarity (geometry); Semantic computing; Semantics (computer science); Semantic memory; Point (geometry); Test (biology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts"],"consensus_categories":[],"category_scores_codex":[0.006478879,0.0009143366,0.001318395,0.002361038,0.0003799326,0.0002687017,0.001224759,0.0007545263,0.0004880872],"category_scores_gemma":[0.003155205,0.0008092426,0.0002595207,0.00252194,0.003411817,0.0002585489,0.001368008,0.0008745734,0.0006266999],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004514132,"about_ca_system_score_gemma":0.001096573,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001253686,"about_ca_topic_score_gemma":0.003639268,"domain_scores_codex":[0.9939437,0.0003243332,0.001112429,0.002357864,0.001026706,0.001234991],"domain_scores_gemma":[0.996006,0.001103751,0.00083611,0.001199446,0.0004372784,0.0004174195],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006341696,0.0014647,0.00177235,0.014904,0.002009891,0.0001545039,0.01220617,0.0004387962,0.4581625,0.01825613,0.4192263,0.07077046],"study_design_scores_gemma":[0.003452641,0.0005952055,0.00009739018,0.005911699,0.002584777,0.0003492973,0.003888052,0.524289,0.01526371,0.004513531,0.4354675,0.003587205],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.08534814,0.1122714,0.1425457,0.002924008,0.03711791,0.03100664,0.02046827,0.009636685,0.5586812],"genre_scores_gemma":[0.07499242,0.0002406901,0.3176014,0.0001602656,0.0007023177,0.000112411,0.0002517087,0.0031944,0.6027444],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.5238501,"threshold_uncertainty_score":0.9994358,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.090822860149071,"score_gpt":0.4431904847437317,"score_spread":0.3523676245946608,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}