{"id":"W4415303363","doi":"10.1080/00036846.2025.2564464","title":"Predictable by construction: assessing forecast directional accuracy of temporal aggregates*","year":2025,"lang":"en","type":"article","venue":"Applied Economics","topic":"Forecasting Techniques and Applications","field":"Decision Sciences","cited_by":3,"is_retracted":false,"has_abstract":true,"ca_institutions":"Wilfrid Laurier University","funders":"Social Sciences and Humanities Research Council of Canada","keywords":"Forecast error; Forecast verification; Economic forecasting; Consensus forecast","routes":{"ca_aff":true,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0007328079,0.0001222501,0.0002717732,0.0001658658,0.000217513,0.0002086597,0.0004519106,0.00009917634,0.0001874104],"category_scores_gemma":[0.0002289574,0.0001156585,0.00008083198,0.0004516388,0.0002290614,0.0002588807,0.0001454724,0.0001169731,0.00002704223],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00006822412,"about_ca_system_score_gemma":0.0001603574,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00004649887,"about_ca_topic_score_gemma":0.00001252388,"domain_scores_codex":[0.9985612,0.00001510466,0.0007208302,0.0003948269,0.0001347812,0.0001732324],"domain_scores_gemma":[0.9981754,0.000704577,0.0004992923,0.000454849,0.0001142407,0.00005164047],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.00004393966,0.0001357015,0.03059599,0.00001704406,0.00006386993,1.392307e-7,0.00006309669,0.001342093,0.0014235,0.4118254,0.1169528,0.4375365],"study_design_scores_gemma":[0.0005379244,0.00002547732,0.002372189,0.00003670847,0.00002585795,0.00001399019,0.0005311934,0.01232012,0.03856542,0.5440671,0.4012158,0.0002881905],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.6716367,0.0001441967,0.1083767,0.001250321,0.0005265849,0.0005699152,0.0002070285,0.0002159294,0.2170726],"genre_scores_gemma":[0.9528546,0.00003207123,0.04585954,0.0001299824,0.0000531155,0.00007546719,0.00003813262,0.000009885954,0.0009472215],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4372483,"threshold_uncertainty_score":0.4716415,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.04749146483030359,"score_gpt":0.3441147451625829,"score_spread":0.2966232803322793,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}