{"id":"W2901415948","doi":"10.1017/s0033291718003380","title":"Equivalence testing: reversed hypotheses, margins, and the need for controlling researcher allegiance","year":2018,"lang":"en","type":"letter","venue":"Psychological Medicine","topic":"Meta-analysis and systematic reviews","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Allegiance; Equivalence (formal languages); Content (measure theory); Information retrieval; Psychology; Action (physics); Computer science; Statistics; Econometrics; Social psychology; Mathematics; Discrete mathematics; Political science","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":["metaresearch"],"domain":"methods","study_design":"not_applicable","genre":"commentary","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":["metaresearch"],"domain":"methods","study_design":"not_applicable","genre":"commentary","about_ca_system":false,"about_ca_topic":false,"confidence":"high","status":"direct model label, unvalidated"}],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","insufficient_payload"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.1867228,0.0005979714,0.006026557,0.0002888588,0.0003178283,0.0004475764,0.003793597,0.0006289507,0.008670478],"category_scores_gemma":[0.4216589,0.0001858375,0.001020358,0.001166855,0.002407748,0.00007710792,0.0001982071,0.001477716,0.0007339016],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00002716822,"about_ca_system_score_gemma":0.0000183389,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00001534137,"about_ca_topic_score_gemma":0.00000211399,"domain_scores_codex":[0.969431,0.01448953,0.007096646,0.002120463,0.00614735,0.0007149598],"domain_scores_gemma":[0.886426,0.1012974,0.004977739,0.00480039,0.002283525,0.0002150043],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0001405638,0.00001362196,0.00008254793,0.0001007432,0.0001366894,0.00003139909,0.00009751168,3.307208e-7,0.00002729208,0.0003102838,0.9905619,0.008497107],"study_design_scores_gemma":[0.002567562,0.0004033833,0.001271305,0.0004011995,0.0003210259,0.00004074301,0.0001496183,0.001495069,2.169721e-7,0.03377959,0.9593473,0.0002229668],"study_design_candidate":"not_applicable","study_design_consensus":"not_applicable","genre_codex":"commentary","genre_gemma":"commentary","genre_scores_codex":[0.0004050259,0.00651926,0.01537815,0.9489537,0.0009151949,0.003419838,0.00003838322,0.00001607697,0.02435441],"genre_scores_gemma":[0.008589708,0.0005334816,0.01277037,0.891602,0.01148063,0.0006218366,0.00003147962,0.00005069152,0.07431974],"genre_candidate":"commentary","genre_consensus":"commentary","teacher_disagreement_score":0.2349361,"threshold_uncertainty_score":0.9922357,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9152274739782401,"score_gpt":0.5964649393488219,"score_spread":0.3187625346294182,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}