{"id":"W6910119171","doi":"10.48448/g98m-z359","title":"Do Large Language Models Know How Much They Know?","year":2024,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal; Polytechnique Montréal","funders":"","keywords":"Scope (computer science); Software deployment; Benchmark (surveying); Language model; Property (philosophy); Key (lock)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","scholarly_communication","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.002345508,0.0008682012,0.000706665,0.00250181,0.0002510556,0.001143556,0.002682009,0.0005061896,0.00276351],"category_scores_gemma":[0.0002267662,0.0007093135,0.0002058831,0.002485682,0.00116374,0.0006192453,0.001091387,0.001013017,0.04091911],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0005393819,"about_ca_system_score_gemma":0.001110245,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000494885,"about_ca_topic_score_gemma":0.003155551,"domain_scores_codex":[0.9934961,0.0001223862,0.0003748199,0.002088268,0.002217476,0.001700929],"domain_scores_gemma":[0.9968446,0.00006056454,0.0003438135,0.002058043,0.0002175586,0.0004754259],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.000006030525,0.0002059108,0.000007115054,0.0001617523,0.00007326448,0.0001538517,0.001751225,0.00005243042,0.002783393,0.06594595,0.924086,0.004773132],"study_design_scores_gemma":[0.0005234447,0.00007799872,0.00000137029,0.0009833323,0.0001298091,0.00004945386,0.002607683,0.05347323,0.0002861706,0.01572296,0.9249014,0.001243141],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.0001182637,0.01522043,0.002129463,0.0008447898,0.001828867,0.0008871736,0.002330463,0.002650301,0.9739903],"genre_scores_gemma":[0.07101975,0.0002812179,0.005404847,0.0002213382,0.001299217,0.00005181806,0.0001646202,0.002295949,0.9192613],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.07090148,"threshold_uncertainty_score":0.9998934,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02435099445772429,"score_gpt":0.3133688716432343,"score_spread":0.28901787718551,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}