{"id":"W6947633094","doi":"10.48448/06fw-yp11","title":"Do Language Models Understand Measurements?","year":2022,"lang":"en","type":"other","venue":"Underline Science Inc.","topic":"Botanical Research and Applications","field":"Agricultural and Biological Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Kootenay Association for Science & Technology","funders":"","keywords":"Language model; Language understanding; Simple (philosophy); Embedding; Work (physics); Natural language; Modeling language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0007055606,0.0001793592,0.0001760115,0.00005137823,0.0004893898,0.0001891064,0.001192306,0.00009520829,0.02684471],"category_scores_gemma":[0.00008960002,0.00007209768,0.00006334552,0.001118336,0.000481053,0.0001099289,0.0003936066,0.0002984387,0.0001633571],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001845731,"about_ca_system_score_gemma":0.0001024041,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0006526372,"about_ca_topic_score_gemma":0.002588261,"domain_scores_codex":[0.9973237,0.00005131153,0.0001652255,0.0006133188,0.001325116,0.0005213465],"domain_scores_gemma":[0.9992838,0.00008432802,0.0001023198,0.0001940613,0.00005626394,0.0002792514],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.00001998857,0.0005646716,0.0002283864,0.0000254292,0.00006053799,0.0000176301,0.0001724577,0.0000596803,0.09037142,0.03866193,0.5991409,0.270677],"study_design_scores_gemma":[0.0003006343,0.0003292109,0.0002751122,0.00005422984,0.00002889864,0.000007763983,0.00324861,0.001330465,0.0003010085,0.01756617,0.9757296,0.0008283325],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.001684372,0.001367565,0.000178572,0.002263261,0.0001307483,0.0006463351,0.0003449139,0.0003207432,0.9930635],"genre_scores_gemma":[0.5378503,0.0004952071,0.001922944,0.001279703,0.001033689,0.0001621208,0.0004581307,0.0000426079,0.4567553],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.5363082,"threshold_uncertainty_score":0.9740449,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1099746847172679,"score_gpt":0.3209974188198076,"score_spread":0.2110227341025397,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}