{"id":"W4206409932","doi":"10.4095/329265","title":"Datasets to support geoscience language models","year":2021,"lang":"en","type":"report","venue":"","topic":"Geological Modeling and Analysis","field":"Earth and Planetary Sciences","cited_by":1,"is_retracted":false,"has_abstract":false,"ca_institutions":"Natural Resources Canada","funders":"","keywords":"Earth science; Computer science; Geology; Data science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005209496,0.001742418,0.0007864652,0.002647643,0.00123662,0.002713037,0.003981052,0.001617884,0.04542961],"category_scores_gemma":[0.02817826,0.0008184033,0.001805387,0.004706199,0.0006286729,0.005564959,0.00300923,0.003018725,0.03222932],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0021169,"about_ca_system_score_gemma":0.006863619,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.06787221,"about_ca_topic_score_gemma":0.06180828,"domain_scores_codex":[0.9964622,0.0007365835,0.0003790392,0.0005242177,0.001581948,0.0003160477],"domain_scores_gemma":[0.9767304,0.005148366,0.0007628006,0.009010907,0.007124321,0.001223176],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0004622057,0.000238361,0.002661036,0.0002811253,0.00008705817,0.00005998681,0.0001088759,0.009901545,0.001537685,0.01956804,0.9430522,0.02204194],"study_design_scores_gemma":[0.0006917719,0.00007367578,0.004766977,0.0001636894,0.00007881939,0.00009610627,0.0002995239,0.09236247,0.01183553,0.02441636,0.8651121,0.00010299],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.00280664,0.00004288558,0.02035386,0.001201796,0.0002380028,0.0004186132,0.9444988,0.01832217,0.01211712],"genre_scores_gemma":[0.004997931,0.00004272697,0.02180615,0.0001427814,0.00002448389,0.000375471,0.9674835,0.001615588,0.003511462],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.06787221,"threshold_uncertainty_score":0.1519772,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07081916935282905,"score_gpt":0.2945242522094975,"score_spread":0.2237050828566684,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}