{"id":"W7083466133","doi":"10.5281/zenodo.17209747","title":"Zoran🦋 aSiM Benchmark: Empirical Validation of a Mimetic Intelligence Framework Across 2000+ Multidomain Questions","year":2025,"lang":"fr","type":"report","venue":"Zenodo (CERN European Organization for Nuclear Research)","topic":"Agriculture, Water, and Health","field":"Environmental Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"","keywords":"Baseline (sea); Benchmark (surveying); Qualitative analysis; Empirical research","routes":{"ca_aff":false,"ca_fund":false,"ca_venue":false,"about_ca":true,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["metaepi_narrow","sts","insufficient_payload"],"consensus_categories":["insufficient_payload"],"category_scores_codex":[0.00245449,0.0004768107,0.0006071987,0.000219924,0.003369821,0.0005699285,0.001776019,0.0006529029,0.0366484],"category_scores_gemma":[0.002532278,0.000457458,0.0002279016,0.001730978,0.001028631,0.000357618,0.002409897,0.001343809,0.01023133],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001762622,"about_ca_system_score_gemma":0.0000562088,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0009905749,"about_ca_topic_score_gemma":0.00001400891,"domain_scores_codex":[0.994145,0.001236954,0.001174348,0.001128537,0.001338258,0.0009769017],"domain_scores_gemma":[0.9970437,0.0002061185,0.0006271476,0.000924997,0.0007095435,0.0004885062],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"not_applicable","study_design_scores_codex":[0.0003164653,0.003213751,0.002899549,0.003213676,0.0004295597,0.0001166315,0.02987276,0.006498965,0.003234515,0.007438063,0.3210058,0.6217602],"study_design_scores_gemma":[0.0002299613,0.0004279266,0.01394292,0.000877356,0.000109517,0.0001550278,0.001122749,0.0005054698,0.001209896,0.001938739,0.9789596,0.0005208025],"study_design_candidate":"not_applicable","study_design_consensus":null,"genre_codex":"other","genre_gemma":"empirical","genre_scores_codex":[0.2171478,0.004291231,0.2432081,0.007737496,0.004084926,0.008436503,0.009523545,0.001998686,0.5035717],"genre_scores_gemma":[0.8911242,0.01699929,0.01297755,0.0005806086,0.001431944,0.000001364448,0.01730656,0.002937261,0.05664125],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.6739764,"threshold_uncertainty_score":0.9997877,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06040146508441012,"score_gpt":0.3355737193026329,"score_spread":0.2751722542182227,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}