{"id":"W2039140614","doi":"10.1139/x03-230","title":"An evaluation of diagnostic tests and their roles in validating forest biometric models","year":2004,"lang":"en","type":"article","venue":"Canadian Journal of Forest Research","topic":"Forest ecology and management","field":"Environmental Science","cited_by":114,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Sign test; Wilcoxon signed-rank test; Test (biology); Statistics; Nonparametric statistics; Biometrics; Statistical hypothesis testing; Parametric statistics; Benchmark (surveying); Credibility; Computer science; Mathematics; Econometrics; Artificial intelligence; Mann–Whitney U test","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.005196461,0.00007882767,0.0001391564,0.001062202,0.0001383151,0.00003934187,0.0003255418,0.00006478913,0.000131001],"category_scores_gemma":[0.001741734,0.00006536826,0.00002622239,0.001040419,0.0004041068,0.000444288,0.00004918707,0.0002483547,0.000007168529],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.000597202,"about_ca_system_score_gemma":0.0004515726,"about_ca_topic_candidate":true,"about_ca_topic_consensus":true,"about_ca_topic_score_codex":0.02340591,"about_ca_topic_score_gemma":0.6942955,"domain_scores_codex":[0.9984515,0.0002318153,0.0002945702,0.0001456319,0.0004835643,0.0003928846],"domain_scores_gemma":[0.9988759,0.0003688569,0.0001018314,0.000161676,0.000119231,0.000372491],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.000006691867,0.00004853719,0.6816652,0.000008109876,0.000007648759,0.00003840296,0.0007070003,0.3073648,0.00008836998,0.00291538,0.00005974095,0.007090171],"study_design_scores_gemma":[0.0006066426,0.0003825409,0.8535046,0.00006093235,0.000007714203,0.00002114173,0.0004223623,0.006599073,0.0001541311,0.1381424,0.00003806843,0.00006043865],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9975795,0.0003840254,0.0003373903,0.0002654073,0.00004034817,0.0002666734,0.000005157069,0.000001285468,0.001120284],"genre_scores_gemma":[0.9996476,0.00004755682,0.0002392511,0.00001629967,0.0000210156,0.00001026266,0.000002672137,0.000008140474,0.000007243111],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.6708896,"threshold_uncertainty_score":0.9830973,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.074468337547369,"score_gpt":0.336717009409145,"score_spread":0.2622486718617759,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}