{"id":"W4400376384","doi":"10.48550/arxiv.2407.02961","title":"Towards a Scalable Reference-Free Evaluation of Generative Models","year":2024,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Simulation Techniques and Applications","field":"Decision Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"","funders":"Natural Sciences and Engineering Research Council of Canada; Chinese University of Hong Kong","keywords":"Generative grammar; Computer science; Scalability; Generative model; Artificial intelligence; Database","routes":{"ca_aff":false,"ca_fund":true,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":true},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.01316004,0.001600372,0.001927238,0.003026718,0.0008081652,0.004457772,0.003029278,0.00224572,0.002859895],"category_scores_gemma":[0.05848764,0.001047394,0.001499931,0.002110966,0.001494484,0.004192233,0.004966902,0.003178912,0.001434513],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001975932,"about_ca_system_score_gemma":0.002233564,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0058467,"about_ca_topic_score_gemma":0.006654517,"domain_scores_codex":[0.9927216,0.003864743,0.0003558101,0.0009258093,0.001861591,0.0002702927],"domain_scores_gemma":[0.9772208,0.01456989,0.0009933866,0.0035279,0.003076886,0.0006112611],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"simulation_or_modeling","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0003626572,0.0001877412,0.007687345,0.0003232896,0.0002947845,0.0001779726,0.0002920381,0.7463672,0.007623213,0.04155557,0.006651402,0.1884768],"study_design_scores_gemma":[0.00001267862,0.00001853084,0.0002745104,0.00001647599,0.000007520303,0.00002538826,0.00001742929,0.9842026,0.001298347,0.01349506,0.0006186665,0.00001272741],"study_design_candidate":"simulation_or_modeling","study_design_consensus":"simulation_or_modeling","genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.01666866,0.0004323164,0.9785039,0.0002480609,0.00004030208,0.00007219562,0.0003679997,0.002781102,0.0008854602],"genre_scores_gemma":[0.3372466,0.0004137896,0.6558781,0.0002544682,0.00008617293,0.0002979015,0.00327365,0.001349787,0.001199461],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.01316004,"threshold_uncertainty_score":0.06959772,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.546135853014692,"score_gpt":0.3552762688511931,"score_spread":0.190859584163499,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}