{"id":"W4390041668","doi":"10.48550/arxiv.2312.11690","title":"Agent-based Learning of Materials Datasets from Scientific Literature","year":2023,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Machine Learning in Materials Science","field":"Materials Science","cited_by":12,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; University of Toronto","keywords":"Computer science; Scalability; Information extraction; Data science; Bottleneck; Consistency (knowledge bases); Artificial intelligence; Resource (disambiguation); Workflow; Natural language; The Internet; Machine learning; World Wide Web; Database","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.002921635,0.001292738,0.0008929048,0.00619135,0.0008058993,0.001748233,0.002682253,0.001718673,0.003511318],"category_scores_gemma":[0.01114595,0.0005422884,0.001698749,0.003471049,0.0006942952,0.002894112,0.001932654,0.00166518,0.001845672],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001380591,"about_ca_system_score_gemma":0.00277291,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.003372606,"about_ca_topic_score_gemma":0.01053181,"domain_scores_codex":[0.9984781,0.0005339442,0.0001391521,0.0004359145,0.0003494824,0.00006339368],"domain_scores_gemma":[0.9932603,0.004674469,0.0004817828,0.0007147753,0.0006437255,0.0002248595],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00119491,0.002301577,0.01446716,0.00471317,0.0007537667,0.001535065,0.0007538947,0.2681633,0.01184442,0.02041546,0.1256261,0.5482312],"study_design_scores_gemma":[0.0002774887,0.0002213377,0.002202259,0.0001647304,0.0001481052,0.0001632998,0.0003370481,0.904732,0.01143947,0.02857182,0.05167227,0.00007020403],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3000764,0.008609485,0.4785201,0.01144017,0.001736766,0.003233892,0.1182796,0.05344787,0.02465579],"genre_scores_gemma":[0.2825173,0.00134717,0.5652353,0.0009867249,0.0003282626,0.001483376,0.1423807,0.0004783582,0.005242748],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.00619135,"threshold_uncertainty_score":0.01545131,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06760741861473786,"score_gpt":0.218393963363026,"score_spread":0.1507865447482881,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}