{"id":"W4280546471","doi":"10.1037/tms0000013","title":"An Evaluation of Math Applications in the App Store: Do they Contain Benchmarks of Educational Quality?","year":2022,"lang":"en","type":"article","venue":"TMS Proceedings 2021","topic":"Technology Adoption and User Behaviour","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"McGill University","funders":"","keywords":"Coding (social sciences); Context (archaeology); Computer science; Educational game; Benchmark (surveying); Quality (philosophy); Mathematics education; Multimedia; Mathematics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.01000905,0.00008168169,0.0001815039,0.0003073489,0.0001673021,0.00004314968,0.0009960937,0.000059316,0.002401017],"category_scores_gemma":[0.0007758294,0.00005950188,0.00005905499,0.001035419,0.0001021053,0.0002032044,0.0001043521,0.0002277985,0.000007190055],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00007813642,"about_ca_system_score_gemma":0.0002271412,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00008289702,"about_ca_topic_score_gemma":0.00005507039,"domain_scores_codex":[0.9969906,0.0001811843,0.0006151504,0.0003136199,0.001783044,0.000116376],"domain_scores_gemma":[0.9980352,0.0003571941,0.000490544,0.0003364423,0.0007537988,0.00002683652],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"observational","study_design_scores_codex":[0.00002185239,0.0006617449,0.3357405,0.000005776532,0.00000644905,5.506572e-8,0.009363477,0.00007854518,0.004214105,0.6386407,0.001675514,0.009591201],"study_design_scores_gemma":[0.0004132896,0.00009565282,0.717789,0.000005561206,0.00002792414,0.00000524128,0.09213346,0.000754195,0.000208951,0.1862704,0.002194866,0.00010146],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9930614,0.0001076932,0.00002985654,0.002403861,0.00005699895,0.0005330289,0.00003938326,0.000007930188,0.003759824],"genre_scores_gemma":[0.998978,0.000002798774,0.0002811098,0.00006129344,0.00002477444,0.0005092816,0.0000382185,0.000004786271,0.00009974959],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.4523703,"threshold_uncertainty_score":0.9985109,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1116985507297271,"score_gpt":0.4422641609300444,"score_spread":0.3305656102003174,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}