{"id":"W7154608271","doi":"10.48448/tt6v-1975","title":"The emergence of flexible perspective reasoning in large language models","year":2025,"lang":"","type":"other","venue":"Underline Science Inc.","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Interpretability; Perspective (graphical); Flexibility (engineering); Object (grammar); Antecedent (behavioral psychology); Subject (documents); Point (geometry); Relation (database); Character (mathematics)","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004914009,0.0008109756,0.0006264849,0.0004199981,0.0004910436,0.0021165,0.001207772,0.0008062896,0.001913364],"category_scores_gemma":[0.02496434,0.0008138546,0.000741104,0.0003785572,0.00161984,0.003888436,0.002055883,0.003035055,0.000605541],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001286688,"about_ca_system_score_gemma":0.00089636,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004196977,"about_ca_topic_score_gemma":0.01030663,"domain_scores_codex":[0.9979488,0.001223107,0.00008199622,0.0004670358,0.000208009,0.00007110678],"domain_scores_gemma":[0.9834746,0.01208799,0.0006575184,0.003025127,0.0004788559,0.0002760896],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0005542829,0.0002171599,0.01194453,0.0004165909,0.0003088241,0.0005196286,0.003638956,0.6221203,0.04650481,0.05950867,0.003488487,0.2507778],"study_design_scores_gemma":[0.00001377527,0.00003998132,0.0007280136,0.00001499062,0.0000149056,0.00003359343,0.0000974795,0.9480152,0.003955798,0.04631257,0.0007586926,0.00001491883],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.3238003,0.0004187875,0.6642936,0.001500673,0.00006483855,0.0001031133,0.0003227115,0.003958979,0.005536996],"genre_scores_gemma":[0.872513,0.0001461426,0.1248132,0.000179823,0.00002248593,0.00009934757,0.0003531056,0.0003178899,0.00155512],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.004914009,"threshold_uncertainty_score":0.0259881,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.01785536929205611,"score_gpt":0.3355288894265334,"score_spread":0.3176735201344772,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}