{"id":"W4405400545","doi":"10.1007/s11222-024-10546-x","title":"Optimal subsampling for generalized additive models on large-scale datasets","year":2024,"lang":"en","type":"article","venue":"Statistics and Computing","topic":"Geochemistry and Geologic Mapping","field":"Computer Science","cited_by":2,"is_retracted":false,"has_abstract":false,"ca_institutions":"Simon Fraser University","funders":"Natural Sciences and Engineering Research Council of Canada","keywords":"Scale (ratio); Generalized additive model; Mathematics; Computer science; Applied mathematics; Artificial intelligence; Statistics; Geography; Cartography","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0213131,0.002506767,0.006326412,0.002241322,0.001669146,0.002767094,0.006000157,0.003610856,0.003235637],"category_scores_gemma":[0.0559952,0.003016584,0.003973157,0.002944454,0.003310194,0.003823435,0.003744755,0.00492317,0.0008449833],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002442637,"about_ca_system_score_gemma":0.004420359,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0201068,"about_ca_topic_score_gemma":0.02708666,"domain_scores_codex":[0.988592,0.007546279,0.0007088732,0.00171746,0.0009401882,0.000495198],"domain_scores_gemma":[0.9298318,0.06019371,0.001558785,0.005399919,0.001960453,0.001055205],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00117451,0.0006287465,0.004452245,0.0007262519,0.001000967,0.0003004575,0.0003821523,0.7448678,0.002625085,0.04820598,0.008940992,0.1866949],"study_design_scores_gemma":[0.00006758332,0.00006083476,0.0003309906,0.00001801448,0.00005512293,0.00002649146,0.00003234214,0.9666234,0.0002951924,0.03201482,0.0004605131,0.00001473041],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01225465,0.0007292039,0.9851937,0.0004834852,0.00009088119,0.000133518,0.0002857265,0.0005742438,0.0002546413],"genre_scores_gemma":[0.2754216,0.00121686,0.7127222,0.0008386917,0.0008626446,0.001080671,0.004630181,0.0005503614,0.002676961],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.0213131,"threshold_uncertainty_score":0.1127158,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02585760065877823,"score_gpt":0.2810727950670061,"score_spread":0.2552151944082279,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}