{"id":"W4416058056","doi":"10.48550/arxiv.2510.19788","title":"AutumnBench Public Benchmark","year":2025,"lang":"","type":"preprint","venue":"ArXiv.org","topic":"Social Robot Interaction and HRI","field":"Psychology","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Fonds de recherche du Québec – Nature et technologies","keywords":"Benchmarking; Suite; Maximization; Action (physics); Protocol (science); Reinforcement learning","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.003703982,0.005850812,0.002284614,0.003729996,0.001952313,0.004060342,0.008551961,0.004059858,0.08700875],"category_scores_gemma":[0.01365536,0.001436274,0.003426289,0.004856396,0.0008355742,0.004195573,0.004855046,0.00394408,0.10857],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003205269,"about_ca_system_score_gemma":0.003231636,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.03357302,"about_ca_topic_score_gemma":0.05417159,"domain_scores_codex":[0.996005,0.0009270041,0.0003883547,0.001023823,0.001201569,0.0004542466],"domain_scores_gemma":[0.9951644,0.001232668,0.0001890295,0.001648409,0.001349093,0.0004164857],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"not_applicable","study_design_gemma":"bench_or_experimental","study_design_scores_codex":[0.0002443078,0.000203499,0.0004632855,0.0005882052,0.00005374218,0.00004312976,0.00002775488,0.001683342,0.0003490495,0.000912322,0.9860101,0.00942134],"study_design_scores_gemma":[0.001173858,0.0002885362,0.004352281,0.0003805791,0.00007983293,0.0002263954,0.0002442699,0.01897761,0.003206006,0.007419228,0.9635414,0.0001099549],"study_design_candidate":"bench_or_experimental","study_design_consensus":null,"genre_codex":"dataset","genre_gemma":"dataset","genre_scores_codex":[0.004505419,0.001065674,0.005071643,0.0009435125,0.0007713293,0.0005289813,0.9251316,0.03694761,0.02503425],"genre_scores_gemma":[0.002445749,0.0001191942,0.004034424,0.0002321607,0.00003659969,0.000462838,0.9870276,0.001410284,0.004230988],"genre_candidate":"dataset","genre_consensus":"dataset","teacher_disagreement_score":0.08700875,"threshold_uncertainty_score":0.2910733,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.1379247616944918,"score_gpt":0.4053124748547807,"score_spread":0.2673877131602889,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}