{"id":"W6979372690","doi":"","title":"PERRY: Policy Evaluation with Confidence Intervals using Auxiliary Data","year":2025,"lang":"en","type":"article","venue":"ArXiv.org","topic":"Injection Molding Process and Properties","field":"Engineering","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"National Institutes of Health; National Human Genome Research Institute; Parker Institute for Cancer Immunotherapy; Canadian Institute for Advanced Research","keywords":"Task (project management); Confidence interval; Reinforcement learning; Value (mathematics); Ground truth; Construct (python library)","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01762736,0.002037719,0.002269482,0.002172277,0.0007284061,0.003456385,0.003407507,0.002821042,0.006242732],"category_scores_gemma":[0.1226242,0.001247997,0.001462919,0.001503629,0.002971928,0.005964743,0.004292029,0.00567508,0.000887906],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.002327056,"about_ca_system_score_gemma":0.002983603,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.004744139,"about_ca_topic_score_gemma":0.003288617,"domain_scores_codex":[0.9903616,0.005049709,0.0004863864,0.001735501,0.001975109,0.0003916167],"domain_scores_gemma":[0.8885505,0.09389629,0.004016079,0.007837908,0.004607133,0.001092062],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0003906311,0.0001463359,0.004510305,0.0002655351,0.0001586736,0.0001165461,0.0001867482,0.8252169,0.0007714761,0.07746467,0.003859828,0.08691231],"study_design_scores_gemma":[0.00003069534,0.00004859665,0.0002500097,0.00004817435,0.00001201689,0.00002079178,0.00001073449,0.9650924,0.0008846286,0.03266385,0.0009212387,0.00001691035],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.0111075,0.0004875491,0.9841644,0.0005590416,0.00009191423,0.00009905398,0.0003747589,0.001036307,0.002079556],"genre_scores_gemma":[0.6586113,0.000550706,0.3343517,0.0006531653,0.0002076744,0.0004721914,0.001925137,0.0007423331,0.002485764],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.01762736,"threshold_uncertainty_score":0.09322351,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.141702541937224,"score_gpt":0.3492679357619882,"score_spread":0.2075653938247643,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}