{"id":"W4316041138","doi":"10.20944/preprints202301.0219.v1","title":"When to Use Large Language Model: Upper Bound Analysis of BM25 Algorithms in Reading Comprehension Task","year":2023,"lang":"en","type":"preprint","venue":"Preprints.org","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Geomechanica (Canada)","funders":"","keywords":"Computer science; Task (project management); Representation (politics); Reading (process); Comprehension; Natural language processing; Artificial intelligence; Language model; Language understanding; Machine learning; Upper and lower bounds; Linguistics; Engineering; Mathematics; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0171065,0.0016689,0.001900463,0.002649401,0.001469605,0.004428118,0.002511418,0.00362439,0.003659301],"category_scores_gemma":[0.0697031,0.0006069448,0.001088392,0.002045172,0.001093254,0.007646081,0.0019443,0.004187285,0.001909433],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002480426,"about_ca_system_score_gemma":0.002243298,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01156271,"about_ca_topic_score_gemma":0.01530477,"domain_scores_codex":[0.9928539,0.003842217,0.00037192,0.0013385,0.001059865,0.0005335766],"domain_scores_gemma":[0.9576409,0.0354618,0.0009464374,0.002916217,0.002143353,0.0008912809],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002744706,0.001147711,0.01970112,0.0009397688,0.0007634119,0.0002796017,0.0007143298,0.168644,0.007348109,0.01930157,0.06765167,0.7107641],"study_design_scores_gemma":[0.0001147559,0.0002878794,0.008542614,0.0001161208,0.0001528394,0.0001086955,0.0003078071,0.9560197,0.004136437,0.02699953,0.003131514,0.00008203677],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.4334425,0.02865491,0.4859417,0.01236814,0.001109968,0.0005975156,0.003220797,0.007139536,0.02752504],"genre_scores_gemma":[0.843441,0.001450482,0.1419375,0.001462084,0.0005454908,0.0004242675,0.004841711,0.00101579,0.004881647],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0171065,"threshold_uncertainty_score":0.09046882,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1543660094793869,"score_gpt":0.3663622990574539,"score_spread":0.211996289578067,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}