{"id":"W4399648778","doi":"10.1007/978-981-97-3072-8_5","title":"Model-Based Assessment for Multi-subject and Multi-task Scenarios","year":2024,"lang":"en","type":"book-chapter","venue":"","topic":"Sleep and Work-Related Fatigue","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":false,"ca_institutions":"Artificial Intelligence in Medicine (Canada)","funders":"","keywords":"Subject (documents); Task (project management); Computer science; Systems engineering; Engineering; World Wide Web","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","insufficient_payload"],"consensus_categories":[],"category_scores_codex":[0.0002322399,0.0006228149,0.0005585203,0.0003128729,0.0001200175,0.00008176381,0.000220435,0.001195977,0.001134458],"category_scores_gemma":[0.000008075489,0.0005382712,0.0003368232,0.0000350244,0.0001233177,0.00003719802,0.00007311243,0.0008720892,0.0005465196],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001124272,"about_ca_system_score_gemma":0.0001271927,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00006126834,"about_ca_topic_score_gemma":0.0001891193,"domain_scores_codex":[0.9977967,0.00001907581,0.0005034759,0.00103916,0.0001913194,0.0004503402],"domain_scores_gemma":[0.9987632,0.0001944466,0.0001413875,0.0006197685,0.00009865175,0.0001825263],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003097774,0.0006191951,0.0002243611,0.0002513692,0.003508862,0.000211137,0.001070582,0.001723609,0.00006548878,0.8452216,0.04120616,0.1055878],"study_design_scores_gemma":[0.01405602,0.0007726982,0.0000834798,0.002161977,0.002494662,0.00001764534,0.0001637872,0.8809185,0.00003257039,0.005573548,0.09066415,0.00306095],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"other","genre_gemma":"other","genre_scores_codex":[0.00005014153,0.003569628,0.3369274,0.0005188761,0.002164059,0.002270681,0.0004546969,0.0004983799,0.6535461],"genre_scores_gemma":[0.02218552,0.0000686304,0.07886972,0.001024649,0.0002077444,0.0003640218,0.0003365907,0.0003099471,0.8966332],"genre_candidate":"other","genre_consensus":"other","teacher_disagreement_score":0.8791949,"threshold_uncertainty_score":0.9997786,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.08180886990868193,"score_gpt":0.3627268972816968,"score_spread":0.2809180273730149,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}