{"id":"W4409197441","doi":"10.1038/s41598-025-96369-w","title":"Developing a standardized framework for evaluating health apps using natural language processing","year":2025,"lang":"en","type":"article","venue":"Scientific Reports","topic":"Mobile Health and mHealth Applications","field":"Health Professions","cited_by":8,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"Argosy Foundation","keywords":"Computer science; Data science; MEDLINE; Natural language processing; World Wide Web; Biology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch"],"consensus_categories":["metaresearch"],"category_scores_codex":[0.3251365,0.005008366,0.005196209,0.04987268,0.004100683,0.01660437,0.006179484,0.004584686,0.004760759],"category_scores_gemma":[0.4051156,0.002416736,0.01300405,0.01954586,0.006711472,0.01410057,0.01192817,0.005434595,0.00198713],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.01920128,"about_ca_system_score_gemma":0.04098552,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009189009,"about_ca_topic_score_gemma":0.01464037,"domain_scores_codex":[0.5556632,0.2791379,0.1060384,0.01166405,0.04511828,0.002377996],"domain_scores_gemma":[0.5197245,0.3520379,0.03190224,0.01788153,0.07695467,0.001499196],"domain_codex":null,"domain_gemma":"evaluation","domain_candidate":"evaluation","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006730633,0.0006591063,0.01205007,0.1532777,0.003070729,0.0008091717,0.02284352,0.007991396,0.006367883,0.1304561,0.0223401,0.6394612],"study_design_scores_gemma":[0.002317904,0.00297296,0.0286788,0.2540138,0.01053353,0.001840816,0.02601996,0.04066428,0.02617192,0.2522946,0.3529394,0.001552074],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.01448208,0.03989107,0.7882614,0.008527235,0.0008812659,0.1113628,0.007216222,0.002016097,0.0273619],"genre_scores_gemma":[0.03574029,0.006482761,0.8607102,0.001223533,0.0001077891,0.09218913,0.002697852,0.0001592511,0.0006891462],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.6748635,"threshold_uncertainty_score":0.8322268,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1108274491389912,"score_gpt":0.5528486527740878,"score_spread":0.4420212036350966,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}