{"id":"W2997103344","doi":"10.48550/arxiv.1912.13082","title":"The Shmoop Corpus: A Dataset of Stories with Loosely Aligned Summaries","year":2019,"lang":"en","type":"preprint","venue":"arXiv (Cornell University)","topic":"Topic Modeling","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Toronto","funders":"","keywords":"Computer science; Automatic summarization; Paragraph; Natural language processing; Artificial intelligence; Set (abstract data type); Reading comprehension; Construct (python library); Exploit; Reading (process); Comprehension; Sentence; Exposition (narrative); Linguistics; World Wide Web; Literature; Art; Programming language","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.0003394419,0.0002661305,0.0003445296,0.0001049296,0.0001936097,0.0001685115,0.002964974,0.0001573815,0.000005382442],"category_scores_gemma":[0.0000377333,0.0002090912,0.00008549431,0.0003304089,0.00027233,0.0003762296,0.00254281,0.0002732214,0.00002157768],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00009602541,"about_ca_system_score_gemma":0.0004438164,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0005826275,"about_ca_topic_score_gemma":0.0004573668,"domain_scores_codex":[0.9983222,0.0001289096,0.0002367387,0.0008449008,0.0001605608,0.000306729],"domain_scores_gemma":[0.9965184,0.0002733819,0.0004178727,0.002504942,0.0002049847,0.00008037883],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"theoretical_or_conceptual","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0002143779,0.00008350606,0.01579057,0.0002343016,0.0003481032,0.0001562894,0.001061837,0.2960912,0.0000272416,0.6811615,0.003293574,0.001537489],"study_design_scores_gemma":[0.001559442,0.000293672,0.001393529,0.0003626188,0.0002182622,0.00001200185,0.0007083841,0.9034863,0.000548655,0.04648164,0.04366819,0.00126733],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.1348269,0.0002619862,0.862343,0.0001575914,0.0007282452,0.0004055102,0.0004194899,0.000124572,0.0007326728],"genre_scores_gemma":[0.9953628,0.0001872819,0.003068039,0.00005468436,0.00004021607,8.7636e-7,0.0001352214,0.00001391638,0.001136933],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.8605359,"threshold_uncertainty_score":0.8526492,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06948629363307242,"score_gpt":0.1862907546288985,"score_spread":0.1168044609958261,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}