{"id":"W7117361401","doi":"10.51798/sijis.v6i4.1177","title":"Validation of tests using an argument-based approach: a review based on the PRISMA model","year":2025,"lang":"","type":"article","venue":"Sapienza International Journal of Interdisciplinary Studies","topic":"Student Assessment and Feedback","field":"Social Sciences","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"Barrie Urology Group","funders":"","keywords":"Relevance (law); Empirical research; Reliability (semiconductor); Quality (philosophy); Process (computing); Test (biology)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.4066257,0.0034289,0.01226347,0.03527448,0.004002382,0.007692531,0.01164792,0.005501953,0.004237104],"category_scores_gemma":[0.5448468,0.003119885,0.026921,0.02755648,0.007681658,0.01138665,0.009854433,0.0065086,0.001014973],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01385237,"about_ca_system_score_gemma":0.06857286,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005939658,"about_ca_topic_score_gemma":0.01192643,"domain_scores_codex":[0.5051938,0.2727278,0.1561977,0.01146135,0.05248361,0.001935702],"domain_scores_gemma":[0.3996008,0.454773,0.05327054,0.02398188,0.06646436,0.001909436],"domain_codex":null,"domain_gemma":"methods","domain_candidate":"methods","domain_consensus":null,"study_design_codex":"systematic_review","study_design_gemma":"systematic_review","study_design_scores_codex":[0.0003557462,0.00007279768,0.001831432,0.8265684,0.01066626,0.0002293902,0.002716337,0.001597733,0.0004654027,0.0125646,0.005493845,0.1374381],"study_design_scores_gemma":[0.001030532,0.0004157268,0.004285755,0.8623978,0.03198352,0.0005380941,0.00172538,0.002853299,0.001501934,0.02052324,0.07245172,0.0002930199],"study_design_candidate":"systematic_review","study_design_consensus":"systematic_review","genre_codex":"review","genre_gemma":"review","genre_scores_codex":[0.006493872,0.6862894,0.1844069,0.01999388,0.004875044,0.08336186,0.005761581,0.0008600698,0.007957399],"genre_scores_gemma":[0.08483338,0.2603994,0.4728304,0.005992074,0.000508732,0.1704865,0.004140469,0.0002198227,0.0005893845],"genre_candidate":"review","genre_consensus":"review","teacher_disagreement_score":0.5933743,"threshold_uncertainty_score":0.7317361,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1165376039611564,"score_gpt":0.4520148118121817,"score_spread":0.3354772078510253,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}