{"id":"W4316041138","doi":"10.20944/preprints202301.0219.v1","title":"When to Use Large Language Model: Upper Bound Analysis of BM25 Algorithms in Reading Comprehension Task","year":2023,"lang":"en","type":"preprint","venue":"Preprints.org","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Geomechanica (Canada)","funders":"","keywords":"Computer science; Task (project management); Representation (politics); Reading (process); Comprehension; Natural language processing; Artificial intelligence; Language model; Language understanding; Machine learning; Upper and lower bounds; Linguistics; Engineering; Mathematics; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","open_science"],"consensus_categories":[],"category_scores_codex":[0.001486074,0.0004132988,0.0009995961,0.001635598,0.00007658104,0.0001202719,0.002230777,0.0003493991,0.0000548617],"category_scores_gemma":[0.0003234566,0.0004561161,0.0004004472,0.001185617,0.00003180431,0.0003580386,0.00937057,0.0007934831,0.0002191356],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002688369,"about_ca_system_score_gemma":0.00015683,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003713773,"about_ca_topic_score_gemma":0.0006086845,"domain_scores_codex":[0.9954854,0.0002281786,0.001003504,0.001915915,0.0007338944,0.000633078],"domain_scores_gemma":[0.995373,0.0002109979,0.0003511278,0.00366481,0.0001960525,0.0002040669],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001423511,0.000136385,0.3477034,0.0001157115,0.0004621504,0.00005486423,0.01906729,0.6237473,0.005846303,0.001910616,0.00003724291,0.0009044941],"study_design_scores_gemma":[0.0001926929,0.000005311101,0.1391234,0.0002149508,0.0001342533,9.282097e-7,0.0001010314,0.8561648,0.001417411,0.002171978,0.00009838009,0.0003748207],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5875104,0.00002499376,0.4111066,0.000266415,0.0002926207,0.000375194,0.00004820122,0.0002373854,0.0001381758],"genre_scores_gemma":[0.9357382,0.00003011997,0.06261723,0.0002124713,0.00004480232,0.0000835212,0.00006553213,0.0000456958,0.001162449],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3484894,"threshold_uncertainty_score":0.9997891,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1543660094793869,"score_gpt":0.3663622990574539,"score_spread":0.211996289578067,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}