{"id":"W7037049414","doi":"","title":"Designing accurate retrieval systems using language models","year":2024,"lang":"en","type":"dissertation","venue":"eScholarship@McGill (McGill)","topic":"Fish biology, ecology, and behavior","field":"Environmental Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"McGill University","keywords":"Language model; Language identification; Natural language; Modeling language; Feature (linguistics)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002857268,0.001590757,0.001896183,0.001509571,0.0006987706,0.002809249,0.002885775,0.002136218,0.004560025],"category_scores_gemma":[0.01122178,0.001055516,0.001589332,0.001059685,0.0009639325,0.006177321,0.002433252,0.002585363,0.009073836],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001135178,"about_ca_system_score_gemma":0.00191869,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00414112,"about_ca_topic_score_gemma":0.005898968,"domain_scores_codex":[0.9978551,0.000671203,0.0001826811,0.0007152591,0.0003850426,0.0001906745],"domain_scores_gemma":[0.9961206,0.001858199,0.0002607886,0.0006986837,0.0009385061,0.0001232729],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004162984,0.0004431135,0.002804166,0.001252202,0.0003269761,0.0004568539,0.000729508,0.2608304,0.07087505,0.014124,0.02087684,0.6268646],"study_design_scores_gemma":[0.00005503682,0.0001822605,0.0003889527,0.00004328256,0.00009792818,0.0001568489,0.0001792485,0.9629446,0.01804671,0.01114827,0.006704234,0.00005270696],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01630477,0.0009712253,0.9686529,0.0004107766,0.0000928417,0.0003238115,0.000404952,0.01105484,0.001783896],"genre_scores_gemma":[0.2428267,0.001142269,0.7417332,0.0008569095,0.0002706265,0.0007440147,0.003127548,0.001076029,0.008222665],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.004560025,"threshold_uncertainty_score":0.0152548,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03595437304549604,"score_gpt":0.2765850545532778,"score_spread":0.2406306815077817,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}