{"id":"W7154608271","doi":"10.48448/tt6v-1975","title":"The emergence of flexible perspective reasoning in large language models","year":2025,"lang":"","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Interpretability; Perspective (graphical); Flexibility (engineering); Object (grammar); Antecedent (behavioral psychology); Subject (documents); Point (geometry); Relation (database); Character (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","insufficient_payload"],"consensus_categories":["sts","insufficient_payload"],"category_scores_codex":[0.008222444,0.0009261522,0.001046007,0.00289208,0.001309775,0.000317637,0.004916387,0.0004171813,0.004157946],"category_scores_gemma":[0.002747443,0.0007515021,0.0002516147,0.01252476,0.005667613,0.000994224,0.001800974,0.001427526,0.0009183781],"about_ca_system_candidate":true,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001496451,"about_ca_system_score_gemma":0.007992042,"about_ca_topic_candidate":true,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.007414664,"about_ca_topic_score_gemma":0.0100641,"domain_scores_codex":[0.9905421,0.0004506221,0.001421066,0.002379171,0.00272045,0.002486634],"domain_scores_gemma":[0.9937009,0.0004839328,0.001329096,0.00249929,0.001668004,0.0003188457],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0001150614,0.0006816463,0.0004178977,0.00007181536,0.0001199532,0.00003547134,0.01935516,0.01054004,0.003146138,0.9590428,0.003144762,0.003329238],"study_design_scores_gemma":[0.001717016,0.0002638407,0.0006481406,0.003002198,0.0001796439,0.00001245119,0.2618982,0.6961706,0.002995062,0.02855026,0.002969608,0.001592919],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.001813282,0.02381146,0.01712093,0.0004287294,0.001726571,0.001934368,0.0007440005,0.0001913643,0.9522293],"genre_scores_gemma":[0.7524133,0.001101361,0.005142492,0.00008507794,0.0002150647,0.00007553112,0.0000160761,0.0003530501,0.2405981],"genre_candidate":"other","genre_consensus":null,"teacher_disagreement_score":0.9304926,"threshold_uncertainty_score":0.9999904,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01785536929205611,"score_gpt":0.3355288894265334,"score_spread":0.3176735201344772,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}