{"id":"W4388657357","doi":"10.26434/chemrxiv-2023-wd5cr","title":"Quantifying the distribution of materials data types in scientific literature across text, tables, and figures","year":2023,"lang":"en","type":"preprint","venue":"ChemRxiv","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Data science; Computer science; Presentation (obstetrics); Field (mathematics); Data extraction; Information retrieval","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":[],"category_scores_codex":[0.00666639,0.0003079259,0.0005192165,0.00008073947,0.0003478929,0.002054946,0.002341273,0.0002764511,0.0001639878],"category_scores_gemma":[0.001799011,0.0002170464,0.00003391998,0.000513431,0.0008545787,0.0003230545,0.004880639,0.000393032,0.00007002406],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004476667,"about_ca_system_score_gemma":0.0001393438,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.000663854,"about_ca_topic_score_gemma":0.0002331726,"domain_scores_codex":[0.9967737,0.000328751,0.0006616557,0.001233346,0.0004926936,0.0005099211],"domain_scores_gemma":[0.9969892,0.0002680388,0.0005248347,0.002020373,0.0001405698,0.00005696565],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"bench_or_experimental","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.00003252671,0.00003024845,0.002478715,0.00141535,0.000006472564,0.0000115637,0.001283924,0.001399451,0.9901721,0.0003713284,0.002711414,0.00008683358],"study_design_scores_gemma":[0.0003380324,0.00002530403,0.05273477,0.003369836,0.0000445125,0.0000198116,0.0002473741,0.005009229,0.9300102,0.003123754,0.004372694,0.0007044953],"study_design_candidate":"bench_or_experimental","study_design_consensus":"bench_or_experimental","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9921594,0.001439292,0.0001130959,0.0004313423,0.003074254,0.0003822352,0.002283598,0.0001004072,0.00001632947],"genre_scores_gemma":[0.9959601,0.0001683721,0.0006714998,0.00001690983,0.0001778775,0.00003763728,0.00251242,0.00003139359,0.0004238448],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.06016199,"threshold_uncertainty_score":0.998981,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07577923707857756,"score_gpt":0.3529841997853836,"score_spread":0.2772049627068061,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}