{"id":"W7024784001","doi":"","title":"A study of similarity measures for natural language processing as applied to candidate-project matching","year":2020,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Expert finding and Q&A systems","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Matching (statistics); Similarity (geometry); Pattern recognition (psychology); Natural language; Measure (data warehouse); Similarity measure","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow"],"consensus_categories":[],"category_scores_codex":[0.001142849,0.0007574097,0.001141184,0.0005403899,0.001095925,0.0003512652,0.002016652,0.0004186145,0.000002689986],"category_scores_gemma":[0.0005605743,0.000730717,0.0002677568,0.001107909,0.00001690427,0.0005934088,0.0003131627,0.001108416,0.00002706156],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0003652478,"about_ca_system_score_gemma":0.000193482,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002123087,"about_ca_topic_score_gemma":0.003790423,"domain_scores_codex":[0.9951172,0.0002853011,0.001024617,0.001588471,0.001273336,0.000711012],"domain_scores_gemma":[0.9974587,0.0002100252,0.0007550387,0.0009067335,0.0003958739,0.000273652],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001635484,0.001591108,0.00006080902,0.004422341,0.001024147,0.0003184753,0.02834246,0.0001783965,0.2928287,0.05659389,0.00005559311,0.6129485],"study_design_scores_gemma":[0.01562819,0.006614928,0.001484057,0.007064886,0.001333742,0.0002173034,0.1849102,0.003873996,0.7117354,0.01522906,0.03938805,0.01252018],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9834244,0.0004418329,0.0000470082,0.00002458369,0.001270219,0.003301288,0.0002102125,0.0005274807,0.01075293],"genre_scores_gemma":[0.9936451,0.000004611119,0.004313968,0.0002510996,0.00008546934,0.0006149978,0.0002093461,0.0001318289,0.0007436053],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6004283,"threshold_uncertainty_score":0.9995144,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02633078822008934,"score_gpt":0.2985178834296658,"score_spread":0.2721870952095765,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}