{"id":"W4411120498","doi":"10.18653/v1/2025.in2writing-1.10","title":"An Analysis of Scoring Methods for Reranking in Large Language Model Story Generation","year":2025,"lang":"en","type":"article","venue":"","topic":"Computational and Text Analysis Methods","field":"Social Sciences","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Natural language processing; Language model; Artificial intelligence","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02143121,0.001493879,0.001076708,0.002797948,0.0008285658,0.001740565,0.002109219,0.001144157,0.003635895],"category_scores_gemma":[0.08147976,0.0005367371,0.0009244424,0.001909163,0.000944274,0.002620927,0.001575629,0.002064463,0.001295714],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001238056,"about_ca_system_score_gemma":0.001327079,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.00315834,"about_ca_topic_score_gemma":0.00652538,"domain_scores_codex":[0.9845368,0.01047542,0.0007381483,0.001040013,0.002918412,0.0002912891],"domain_scores_gemma":[0.9071429,0.07407127,0.002823842,0.006785812,0.008490926,0.0006852107],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.000391383,0.000294608,0.006949388,0.000484785,0.0002013516,0.00009794397,0.0008870917,0.1075447,0.008128662,0.02056427,0.00714142,0.8473145],"study_design_scores_gemma":[0.00008835406,0.0002057986,0.002862992,0.00006865303,0.00004603699,0.0001333338,0.0001435499,0.9688429,0.009500927,0.01388824,0.004165761,0.00005345673],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.02423437,0.0008397363,0.9703503,0.0002438855,0.00005700881,0.0002488869,0.0001351168,0.002022353,0.001868407],"genre_scores_gemma":[0.2737224,0.0004057307,0.7205899,0.0001236166,0.00009228797,0.0005368888,0.0007948377,0.0008231351,0.002911211],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.02143121,"threshold_uncertainty_score":0.1133404,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.07331420775418603,"score_gpt":0.5180781471380161,"score_spread":0.4447639393838301,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}