{"id":"W4376653681","doi":"10.48550/arxiv.2305.08264","title":"MatSci-NLP: Evaluating Scientific Language Models on Materials Science Language Tasks Using Text-to-Schema Modeling","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Canadian Institute for Advanced Research","keywords":"Computer science; Artificial intelligence; Natural language processing; Schema (genetic algorithms); Benchmark (surveying); Question answering; Language model; Natural language understanding; Information retrieval; Natural language","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.00660261,0.004016519,0.001285265,0.002832917,0.001180111,0.00273225,0.004526096,0.003974281,0.0110418],"category_scores_gemma":[0.02449891,0.0008509199,0.002446086,0.002677896,0.001524086,0.005769175,0.003039085,0.005186369,0.006661424],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002842177,"about_ca_system_score_gemma":0.004045913,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.01752574,"about_ca_topic_score_gemma":0.01907203,"domain_scores_codex":[0.9959054,0.0016922,0.0003748596,0.001115452,0.0006539879,0.0002580567],"domain_scores_gemma":[0.9859738,0.009709422,0.0005474017,0.001858418,0.001351763,0.0005591352],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002472468,0.003062163,0.01083729,0.004834387,0.001184157,0.0008235753,0.0006298336,0.400288,0.01404871,0.00682723,0.203299,0.3516931],"study_design_scores_gemma":[0.0006065504,0.000953666,0.004094751,0.0001707103,0.0001640738,0.0002432156,0.0004280622,0.9424859,0.01696169,0.01082005,0.02295045,0.0001208374],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.4405885,0.01244911,0.2346783,0.007368932,0.003559158,0.003536276,0.1143561,0.1414176,0.04204603],"genre_scores_gemma":[0.4903323,0.001658357,0.1945361,0.002801143,0.0004307763,0.00344777,0.2936891,0.003024539,0.01007994],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.01752574,"threshold_uncertainty_score":0.03693849,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2573202568645652,"score_gpt":0.2837385523110563,"score_spread":0.02641829544649105,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}