{"id":"W4280546471","doi":"10.1037/tms0000013","title":"An Evaluation of Math Applications in the App Store: Do they Contain Benchmarks of Educational Quality?","year":2022,"lang":"en","type":"article","venue":"TMS Proceedings 2021","topic":"Technology Adoption and User Behaviour","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Coding (social sciences); Context (archaeology); Computer science; Educational game; Benchmark (surveying); Quality (philosophy); Mathematics education; Multimedia; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.03553551,0.001093795,0.001190416,0.006915333,0.0009257354,0.005422045,0.001123219,0.0009348143,0.002029947],"category_scores_gemma":[0.1568921,0.0005039549,0.001231597,0.003378147,0.001219102,0.004821963,0.002135595,0.001210825,0.0009752138],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001562839,"about_ca_system_score_gemma":0.002098606,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00285937,"about_ca_topic_score_gemma":0.004724861,"domain_scores_codex":[0.9659846,0.008468782,0.004140834,0.001876121,0.01859761,0.0009320601],"domain_scores_gemma":[0.8162752,0.08896271,0.018513,0.007988499,0.06544477,0.002815812],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.00433736,0.002736571,0.4906259,0.00711308,0.0007737551,0.000630031,0.02889436,0.0009818593,0.01933251,0.002606291,0.00767644,0.4342919],"study_design_scores_gemma":[0.0002236726,0.006626925,0.9150985,0.00339677,0.000861707,0.0007018333,0.016116,0.007005003,0.02199277,0.001864415,0.02586377,0.0002485626],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9756999,0.001352738,0.008053009,0.0004002757,0.0001453582,0.001480282,0.001403605,0.0004787385,0.01098607],"genre_scores_gemma":[0.9657055,0.0008087772,0.02562366,0.000254977,0.00007197499,0.001825297,0.002461838,0.0002677939,0.002980282],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.03553551,"threshold_uncertainty_score":0.187932,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1116985507297271,"score_gpt":0.4422641609300444,"score_spread":0.3305656102003174,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}