{"id":"W4402028425","doi":"10.17605/osf.io/9qa2z","title":"Probing cross-linguistic differences in time representations using arithmetic word problems","year":2024,"lang":"en","type":"preprint","venue":"HAL (Le Centre pour la Communication Scientifique Directe)","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Arithmetic; Linguistics; Word (group theory); Computer science; Mathematics; Natural language processing; Philosophy","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001373562,0.0006063966,0.0004470245,0.001191739,0.0004614615,0.002774462,0.0006930455,0.0007318206,0.007726254],"category_scores_gemma":[0.01794794,0.0002911407,0.0004438112,0.0008500274,0.001061635,0.003200278,0.002068807,0.001228087,0.001174291],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0004294722,"about_ca_system_score_gemma":0.0002083312,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.001413638,"about_ca_topic_score_gemma":0.0009187568,"domain_scores_codex":[0.9986223,0.0003736437,0.0001496381,0.0004388928,0.0003241503,0.00009142179],"domain_scores_gemma":[0.9940336,0.003055909,0.001483122,0.0008177534,0.0003641949,0.0002454945],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"observational","study_design_scores_codex":[0.005020829,0.004734288,0.2513038,0.001805251,0.0006504925,0.001511033,0.1092228,0.007054984,0.2709663,0.05169215,0.004898582,0.2911395],"study_design_scores_gemma":[0.0004765743,0.004648743,0.7957991,0.0003095439,0.0003543248,0.002235844,0.03752878,0.01889281,0.03059274,0.08249043,0.02624642,0.0004246368],"study_design_candidate":"observational","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9867933,0.0001232765,0.005103914,0.00008361006,0.00002881212,0.00007000749,0.0002357218,0.00004141699,0.007519824],"genre_scores_gemma":[0.9905822,0.0001531963,0.005827437,0.0001041756,0.00002065884,0.0002502646,0.0005446217,0.00006452938,0.00245286],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.007726254,"threshold_uncertainty_score":0.0258469,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.02557441590622494,"score_gpt":0.2891422443992197,"score_spread":0.2635678284929947,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}