{"id":"W6929258176","doi":"10.48448/zxex-ff58","title":"Carpe diem: On the Evaluation of World Knowledge in Lifelong Language Models","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Pipeline (software); Language model; Lifelong learning; Code (set theory); Benchmark (surveying); Work (physics); Dynamics (music); On Language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01371531,0.00359229,0.001448788,0.004193629,0.001545703,0.003948318,0.00618452,0.004719545,0.01219867],"category_scores_gemma":[0.06149939,0.0008497657,0.001833853,0.003282681,0.001590257,0.006827465,0.005528051,0.004247595,0.009525031],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002747795,"about_ca_system_score_gemma":0.002193761,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.02573687,"about_ca_topic_score_gemma":0.03533181,"domain_scores_codex":[0.9888352,0.006142677,0.000792791,0.002114601,0.001677846,0.0004367995],"domain_scores_gemma":[0.9740825,0.01477343,0.0005895098,0.006024047,0.003438802,0.001091646],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.001937013,0.001850453,0.01178743,0.002334059,0.0009984846,0.0005630478,0.0006122654,0.1485702,0.003268321,0.009056326,0.4829699,0.3360524],"study_design_scores_gemma":[0.000791718,0.0009352173,0.006340586,0.0004520921,0.0001734695,0.0003885233,0.0005788967,0.8692517,0.009926862,0.02926603,0.08172192,0.0001729831],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.3379239,0.0255747,0.220935,0.01197334,0.006650299,0.002350483,0.1838384,0.1345084,0.07624546],"genre_scores_gemma":[0.4168832,0.001985897,0.1935871,0.00333737,0.0005642233,0.001661522,0.3557102,0.00679188,0.0194786],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02573687,"threshold_uncertainty_score":0.07253438,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1211401595697338,"score_gpt":0.3990946823107435,"score_spread":0.2779545227410096,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}