{"id":"W4389518977","doi":"10.18653/v1/2023.findings-emnlp.380","title":"HoneyBee: Progressive Instruction Finetuning of Large Language Models for Materials Science","year":2023,"lang":"en","type":"article","venue":"","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":16,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Waterloo; Université de Montréal","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Benchmark (surveying); Construct (python library); Trustworthiness; Language model; Process (computing); Soundness; Code (set theory); Quality (philosophy); Artificial intelligence; Machine learning; Data science; Programming language","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005686465,0.002330325,0.001221075,0.002122615,0.000924091,0.003292781,0.005353958,0.00213657,0.005914059],"category_scores_gemma":[0.03063462,0.001489943,0.004160797,0.001278069,0.001886811,0.006547861,0.005770644,0.005357383,0.005580608],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002001446,"about_ca_system_score_gemma":0.006119155,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007304133,"about_ca_topic_score_gemma":0.01495216,"domain_scores_codex":[0.9955854,0.001561635,0.0004920569,0.001070985,0.001100303,0.0001895168],"domain_scores_gemma":[0.9817754,0.01006866,0.0007782033,0.00540135,0.001562527,0.0004138643],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.001184179,0.0009692254,0.01462683,0.003638379,0.0006381061,0.0007712143,0.002639498,0.1885846,0.04094044,0.04028569,0.1325161,0.5732056],"study_design_scores_gemma":[0.0002754257,0.0002319914,0.0009839603,0.0001678031,0.0001146273,0.000225974,0.0002524317,0.8328754,0.04133474,0.0427289,0.08067656,0.0001323153],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.03544325,0.001205852,0.69695,0.001985167,0.0004328085,0.001034297,0.009798961,0.2488025,0.004347203],"genre_scores_gemma":[0.1103749,0.0004704281,0.8466314,0.001141271,0.00008364342,0.00131547,0.02565298,0.01169597,0.002633961],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.007304133,"threshold_uncertainty_score":0.03007323,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01707591023755806,"score_gpt":0.319318363640433,"score_spread":0.302242453402875,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}