{"id":"W2923483049","doi":"","title":"Toward a Dialogue: Following Professional Standards on Education Achievement Testing","year":2018,"lang":"en","type":"article","venue":"Journal of research practice","topic":"Educational Assessment and Improvement","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Interpretation (philosophy); Test (biology); Context (archaeology); Action (physics); Sociocultural evolution; Process (computing); Psychology; Pedagogy; Engineering ethics; Computer science; Sociology; Engineering","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":true,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":[],"category_scores_codex":[0.3135305,0.001163098,0.002279517,0.004902562,0.02500193,0.03313791,0.008001942,0.04105937,0.003330986],"category_scores_gemma":[0.3402176,0.001975154,0.001729251,0.003108158,0.05785807,0.03886406,0.03639263,0.05377011,0.001761915],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01789481,"about_ca_system_score_gemma":0.06470524,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005285544,"about_ca_topic_score_gemma":0.007171033,"domain_scores_codex":[0.6124234,0.31111,0.020115,0.01037979,0.03436847,0.01160344],"domain_scores_gemma":[0.4263319,0.4549792,0.0188282,0.01422244,0.05022483,0.03541328],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"qualitative","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0000854972,0.0005089635,0.004152849,0.0009581047,0.00005895738,0.003227112,0.5443264,0.0006592277,0.001015577,0.1997514,0.1136897,0.1315662],"study_design_scores_gemma":[0.0000443694,0.0002503465,0.001207371,0.003033801,0.00002397363,0.001139982,0.3684843,0.0009078591,0.00070109,0.1646558,0.4593573,0.0001937946],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"empirical","genre_scores_codex":[0.008673177,0.003064535,0.02063413,0.9533603,0.003691484,0.0002014506,0.00001501477,0.0001157664,0.01024418],"genre_scores_gemma":[0.3923538,0.008153245,0.1511214,0.4269462,0.005187181,0.001548021,0.00007806855,0.0003368028,0.01427525],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.3135305,"threshold_uncertainty_score":0.8465391,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.4493990704713637,"score_gpt":0.6295861663902028,"score_spread":0.1801870959188391,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}