{"id":"W3046018217","doi":"10.48550/arxiv.2007.14461","title":"Modeling Behaviour to Predict User State: Self-Reports as Ground Truth","year":2020,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Emotion and Mood Recognition","field":"Psychology","cited_by":2,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Saskatchewan","funders":"","keywords":"Ground truth; State (computer science); Common ground; Computer science; Psychology; Artificial intelligence; Social psychology; Algorithm","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.0002130651,0.000394116,0.0003942651,0.0003132016,0.0001283718,0.00007173807,0.0003673593,0.000455351,0.00101802],"category_scores_gemma":[0.00003892832,0.0004894992,0.0002693388,0.0003495861,0.00003385499,0.0001425479,0.0005554614,0.0008173049,0.001129953],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0002209233,"about_ca_system_score_gemma":0.0001650526,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009906406,"about_ca_topic_score_gemma":0.00008822258,"domain_scores_codex":[0.9974306,0.0001960824,0.0003690061,0.001452291,0.0001291524,0.0004229228],"domain_scores_gemma":[0.9982901,0.00003883763,0.00022215,0.0007646117,0.0002082286,0.0004761032],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.002016436,0.003805975,0.06263868,0.0006513245,0.004010397,0.03351268,0.01988513,0.7894737,0.000159249,0.06505354,0.0161003,0.002692553],"study_design_scores_gemma":[0.01507898,0.003733577,0.08701815,0.001988864,0.009660451,0.001292706,0.01880172,0.445257,0.0005391173,0.3885368,0.01541927,0.01267342],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9161398,0.00002077229,0.0653136,0.0001277417,0.002724897,0.0006957805,0.00007120345,0.000598583,0.01430766],"genre_scores_gemma":[0.9928229,0.00005345167,0.0002372133,0.0004018793,0.0002112293,0.000005794224,0.0002125103,0.00006849589,0.005986499],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.3442167,"threshold_uncertainty_score":0.9998952,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.09784721173930638,"score_gpt":0.2384778792356532,"score_spread":0.1406306674963468,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}