{"id":"W2901415948","doi":"10.1017/s0033291718003380","title":"Equivalence testing: reversed hypotheses, margins, and the need for controlling researcher allegiance","year":2018,"lang":"en","type":"letter","venue":"Psychological Medicine","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Allegiance; Equivalence (formal languages); Content (measure theory); Information retrieval; Psychology; Action (physics); Computer science; Statistics; Econometrics; Social psychology; Mathematics; Discrete mathematics; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":["metaresearch"],"domain":"methods","study_design":"not_applicable","genre":"commentary","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":["metaresearch"],"domain":"methods","study_design":"not_applicable","genre":"commentary","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.5599111,0.001544197,0.007124157,0.006613177,0.004551119,0.01137079,0.01262678,0.03860088,0.01162517],"category_scores_gemma":[0.9125633,0.002171983,0.009796926,0.006882385,0.02684235,0.01575378,0.008936219,0.03729757,0.002580986],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.008316454,"about_ca_system_score_gemma":0.01764642,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004165975,"about_ca_topic_score_gemma":0.005146992,"domain_scores_codex":[0.3685442,0.4491648,0.0964725,0.02881761,0.05368577,0.003315093],"domain_scores_gemma":[0.04183556,0.8978197,0.01259492,0.02552142,0.01993789,0.002290396],"domain_codex":"methods","domain_gemma":"methods","domain_candidate":"methods","domain_consensus":"methods","study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.008292764,0.0003944448,0.02066095,0.01819455,0.01412906,0.003184832,0.006103149,0.001243606,0.001244609,0.241186,0.2835124,0.4018537],"study_design_scores_gemma":[0.00950966,0.001290946,0.01240082,0.01643883,0.008679264,0.002999583,0.001512678,0.01207664,0.002599317,0.7717003,0.160194,0.0005980376],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.006735121,0.01384146,0.03015794,0.9072415,0.03416162,0.0007866577,0.0008158622,0.0002007822,0.006059043],"genre_scores_gemma":[0.1990841,0.006449092,0.1051341,0.6015069,0.07862918,0.005011735,0.0006655788,0.0003208774,0.00319842],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.4400889,"threshold_uncertainty_score":0.542708,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9152274739782401,"score_gpt":0.5964649393488219,"score_spread":0.3187625346294182,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}