{"id":"W2799978916","doi":"10.1017/s0033291718001289","title":"Equivalence and non-inferiority testing in psychotherapy research","year":2018,"lang":"en","type":"letter","venue":"Psychological Medicine","topic":"Statistical Methods in Clinical Trials","field":"Mathematics","cited_by":15,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Equivalence (formal languages); Content (measure theory); Psychology; Psychotherapist; Information retrieval; Computer science; Mathematics; Discrete mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[{"model":"gemma","categories":[],"domain":null,"study_design":"not_applicable","genre":"commentary","about_ca_system":false,"about_ca_topic":false,"confidence":"low","status":"direct model label, unvalidated"},{"model":"gpt","categories":[],"domain":null,"study_design":"theoretical_or_conceptual","genre":"commentary","about_ca_system":false,"about_ca_topic":false,"confidence":"medium","status":"direct model label, unvalidated"}],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaresearch","metaepi_narrow","sts","research_integrity","insufficient_payload"],"consensus_categories":["research_integrity"],"category_scores_codex":[0.01895245,0.0005341971,0.001904095,0.0004022286,0.0001317672,0.00004099877,0.0009613207,0.002010192,0.001884864],"category_scores_gemma":[0.2121715,0.0003523615,0.00008339335,0.001132372,0.003533221,0.00003979265,0.0001464573,0.00693898,0.0001073361],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007692284,"about_ca_system_score_gemma":0.00002067393,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00003977977,"about_ca_topic_score_gemma":0.000004467031,"domain_scores_codex":[0.9904964,0.003491805,0.001776429,0.001620166,0.001448562,0.00116664],"domain_scores_gemma":[0.8490798,0.148744,0.0003678041,0.001164678,0.0004170343,0.0002266472],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0001774311,0.000194793,0.0005799933,0.0004344348,0.00002782756,0.0006531769,0.0001408196,4.846216e-9,0.0001970431,0.001458528,0.9490938,0.04704214],"study_design_scores_gemma":[0.001986343,0.002724087,0.005871047,0.001979573,0.00003322926,0.00004369817,0.00001602991,0.00002042193,0.000004771738,0.9573196,0.02963056,0.0003706352],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"commentary","genre_gemma":"methods","genre_scores_codex":[0.01175589,0.0004802854,0.01811397,0.9251283,0.003759101,0.002586379,0.00005392148,0.0002453805,0.03787681],"genre_scores_gemma":[0.002253613,0.0008899785,0.6147395,0.3580338,0.02291752,0.0002552691,0.000005176736,0.0001236638,0.0007815564],"genre_candidate":"commentary","genre_consensus":null,"teacher_disagreement_score":0.9558611,"threshold_uncertainty_score":0.9998928,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.9178104425000236,"score_gpt":0.7194051522256781,"score_spread":0.1984052902743455,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}