{"id":"W4408141487","doi":"10.1093/gpbjnl/qzaf017","title":"Evaluative Methodology for HRD Testing: Development of Standard Tools for Consistency Assessment","year":2025,"lang":"en","type":"article","venue":"Genomics Proteomics & Bioinformatics","topic":"Human Resource Development and Performance Evaluation","field":"Psychology","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"MD Precision (Canada)","funders":"","keywords":"Consistency (knowledge bases); Computer science; Reliability engineering; Artificial intelligence; Engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.1857675,0.002741154,0.002263663,0.01040755,0.001638512,0.00567235,0.004528649,0.00242298,0.002388934],"category_scores_gemma":[0.2199634,0.001283125,0.002279594,0.005609603,0.003578066,0.002961913,0.006752319,0.003521951,0.001884993],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00266503,"about_ca_system_score_gemma":0.004866678,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001841319,"about_ca_topic_score_gemma":0.002128579,"domain_scores_codex":[0.8538184,0.0626015,0.01877055,0.0135586,0.04982381,0.001427157],"domain_scores_gemma":[0.8158861,0.07686714,0.01896655,0.03350371,0.05330047,0.001476086],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0008925473,0.001063956,0.1069422,0.004855844,0.001902485,0.0005003143,0.004644541,0.03285921,0.08217607,0.03530372,0.01370585,0.7151533],"study_design_scores_gemma":[0.0003914859,0.003659684,0.1236742,0.004068877,0.001198444,0.002270816,0.00355414,0.334732,0.3396574,0.06783445,0.1176786,0.001279844],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01793728,0.001279156,0.9726421,0.0002676303,0.0001744099,0.001997797,0.000959835,0.002359836,0.002381944],"genre_scores_gemma":[0.1058278,0.0007628932,0.8845233,0.0002897226,0.00009627392,0.005095516,0.001907702,0.0006943165,0.0008025923],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.1857675,"threshold_uncertainty_score":0.9824443,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2234653144906807,"score_gpt":0.441635837035023,"score_spread":0.2181705225443422,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}