{"id":"W4407309364","doi":"10.48550/arxiv.2502.04689","title":"Improving Language Models with Intentional Analysis","year":2025,"lang":"en","type":"preprint","venue":"ArXiv.org","topic":"Topic Modeling","field":"Computer Science","cited_by":1,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Alliance de recherche numérique du Canada","keywords":"Question answering; Computer science; Natural language processing; Language model; Artificial intelligence; Information retrieval","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0002528393,0.0002421512,0.0003624427,0.0004000101,0.00008304661,0.0001565941,0.001411469,0.0001561397,0.00002090629],"category_scores_gemma":[0.0000272153,0.0002132183,0.0002421132,0.0005602788,0.00002864342,0.0002667207,0.002123497,0.0005284252,0.00001416489],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001000953,"about_ca_system_score_gemma":0.0002594897,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009409417,"about_ca_topic_score_gemma":0.0001749699,"domain_scores_codex":[0.9981157,0.00005352997,0.0003099336,0.0009096176,0.0003337804,0.0002775074],"domain_scores_gemma":[0.9981181,0.00004749057,0.0001889191,0.001410761,0.0001599677,0.00007474514],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00001724781,0.0001357213,0.1776627,0.0003429859,0.002354517,0.0001247687,0.00366575,0.74519,0.0003318,0.04059429,0.00009098312,0.02948932],"study_design_scores_gemma":[0.0001303876,0.000009568063,0.007072831,0.00008399281,0.000236967,0.000001865802,0.00005885846,0.9903089,0.000160272,0.001656877,0.00001880174,0.0002606281],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.2727575,0.0001908109,0.72473,0.0002781911,0.0002360136,0.0001143504,0.000009445231,0.0001808784,0.001502724],"genre_scores_gemma":[0.8984644,0.000007300722,0.09881581,0.0002595132,0.00009334865,0.0000384173,0.00003096255,0.000008076008,0.002282098],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6259142,"threshold_uncertainty_score":0.869479,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.03361433793890273,"score_gpt":0.2642698443475976,"score_spread":0.2306555064086949,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}