{"id":"W4407309364","doi":"10.48550/arxiv.2502.04689","title":"Improving Language Models with Intentional Analysis","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada","keywords":"Question answering; Computer science; Natural language processing; Language model; Artificial intelligence; Information retrieval","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.004165682,0.001469617,0.0008070432,0.001635496,0.0006208805,0.003594198,0.001908333,0.001134449,0.004157028],"category_scores_gemma":[0.02251601,0.000633743,0.002124608,0.0008929831,0.0009241553,0.006914851,0.003851357,0.002761287,0.001940953],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001508653,"about_ca_system_score_gemma":0.002682886,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.005790345,"about_ca_topic_score_gemma":0.01032878,"domain_scores_codex":[0.996985,0.001614391,0.0002008801,0.0005409332,0.0005150044,0.0001438234],"domain_scores_gemma":[0.9884225,0.008312497,0.0004671575,0.001733143,0.0008343779,0.0002303584],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0004436462,0.0004354867,0.008441993,0.001046876,0.0003637944,0.0002315273,0.002828368,0.3361728,0.01585806,0.09032764,0.0185488,0.525301],"study_design_scores_gemma":[0.00003345243,0.00004350343,0.0001871542,0.00003741511,0.00004662073,0.00002597667,0.0001475262,0.9416749,0.002462797,0.05097254,0.004348261,0.00001980907],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02987524,0.0004251095,0.9561732,0.001009987,0.0000705728,0.0001423652,0.0005312703,0.008962345,0.002809889],"genre_scores_gemma":[0.378571,0.0004121049,0.6135015,0.0004199019,0.00008189162,0.0002900927,0.002760143,0.001232479,0.002730869],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.005790345,"threshold_uncertainty_score":0.02203047,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03361433793890273,"score_gpt":0.2642698443475976,"score_spread":0.2306555064086949,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}