{"id":"W4396531244","doi":"10.22215/etd/2024-15938","title":"The Fusion of Multilingual Semantic Search and Large Language Models: A New Paradigm for Enhanced Topic Exploration and Contextual Search","year":2024,"lang":"en","type":"dissertation","venue":"","topic":"Advanced Text Analysis Techniques","field":"Computer Science","cited_by":20,"is_retracted":false,"has_abstract":true,"ca_institutions":"Carleton University","funders":"","keywords":"Computer science; Artificial intelligence; Natural language processing; Semantic search; Linguistics; Semantic Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001690053,0.000841691,0.00126799,0.00182137,0.0004343927,0.002222006,0.0007598436,0.0008444336,0.001697177],"category_scores_gemma":[0.006394734,0.0003482636,0.001266539,0.002172279,0.0005449033,0.006089199,0.00213762,0.0014259,0.0009226763],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0007088492,"about_ca_system_score_gemma":0.0009463144,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003806156,"about_ca_topic_score_gemma":0.005954255,"domain_scores_codex":[0.9986953,0.0005680555,0.00009896388,0.0002743107,0.0002950301,0.00006835997],"domain_scores_gemma":[0.9979647,0.001155693,0.0001478744,0.0003592859,0.0003021399,0.00007023399],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0007685986,0.0007968449,0.004445598,0.0005824008,0.0004273577,0.0002962122,0.001096835,0.1130576,0.02791965,0.04086887,0.007931902,0.8018081],"study_design_scores_gemma":[0.00002636486,0.0001456294,0.0008009763,0.00002917948,0.00007109713,0.0001053215,0.0002110055,0.9535558,0.004146446,0.03640488,0.0044627,0.00004059586],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"other","genre_scores_codex":[0.04421541,0.001908214,0.9468844,0.0009905226,0.0001312699,0.00008212038,0.0004454309,0.002173412,0.003169219],"genre_scores_gemma":[0.5607996,0.00160646,0.4314952,0.0004357474,0.0003175413,0.0001488175,0.001606733,0.0003732555,0.003216613],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.003806156,"threshold_uncertainty_score":0.008938015,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03589096521382491,"score_gpt":0.3618835984187028,"score_spread":0.3259926332048779,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}