{"id":"W7106801466","doi":"10.48448/n0er-rq54","title":"A Monte-Carlo Sampling Framework For Reliable Evaluation of Large Language Models Using Behavioral Analysis","year":2025,"lang":"","type":"other","venue":"Open MIND","topic":"","field":"","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University Canada West; HEC Montréal","funders":"","keywords":"Cognition; Entropy (arrow of time); Behavioural sciences; Reliability (semiconductor); Sampling (signal processing); Benchmark (surveying); Behavioral economics","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.05495566,0.001510135,0.00180502,0.003440707,0.001128967,0.004362001,0.003726416,0.002824907,0.005978947],"category_scores_gemma":[0.2254767,0.0009908783,0.001678296,0.001498997,0.004212808,0.004712106,0.004096648,0.003771171,0.0009724643],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.003711784,"about_ca_system_score_gemma":0.003798223,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.008364365,"about_ca_topic_score_gemma":0.007951467,"domain_scores_codex":[0.9593529,0.03493423,0.0009309549,0.001693227,0.002541319,0.0005474619],"domain_scores_gemma":[0.7670203,0.2004092,0.008913658,0.01479917,0.006948429,0.001909308],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000319468,0.0002114946,0.00967669,0.0002910202,0.0002977438,0.0001951521,0.0008811566,0.3858481,0.001449865,0.5285955,0.00271939,0.06951451],"study_design_scores_gemma":[0.00002139122,0.00005313144,0.0006017789,0.00006033606,0.0000150053,0.00003241343,0.00004366998,0.832799,0.000410126,0.1650851,0.0008537221,0.00002437834],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.008647317,0.0001215512,0.9881633,0.0006342959,0.00002708318,0.0001562393,0.0001104904,0.0004105771,0.001729228],"genre_scores_gemma":[0.3612959,0.0001869076,0.634257,0.0004909388,0.0001429997,0.001281355,0.0004164422,0.0002940201,0.001634553],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.05495566,"threshold_uncertainty_score":0.2906368,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.239603421352328,"score_gpt":0.4892376824980191,"score_spread":0.249634261145691,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}