{"id":"W4360799893","doi":"10.5281/zenodo.10804745","title":"Exploring Data Provenance in Handwritten Text Recognition Infrastructure: Sharing and Reusing Ground Truth Data, Referencing Models, and Acknowledging Contributions. Starting the Conversation on How We Could Get It Done","year":2023,"lang":"en","type":"article","venue":"Faculty Digital Archive (New York University Florence)","topic":"Research Data Management Practices","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"Nederlandse Organisatie voor Wetenschappelijk Onderzoek","keywords":"Conversation; Ground truth; Computer science; Reuse; Natural language processing; Provenance; Artificial intelligence; Speech recognition; Psychology; Engineering; Communication; Geology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":["scholarly_communication"],"consensus_categories":["scholarly_communication"],"category_scores_codex":[0.0008494723,0.0002001115,0.0001904895,0.0003495954,0.0005554298,0.003605445,0.002356466,0.00003778551,0.000001141371],"category_scores_gemma":[0.001093755,0.0001818788,0.0000169873,0.0008639348,0.0001603393,0.05263369,0.006506103,0.0004426285,0.000008822686],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.0001399121,"about_ca_system_score_gemma":0.0001318219,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0004631692,"about_ca_topic_score_gemma":0.0003985919,"domain_scores_codex":[0.9977254,0.0001147878,0.0001971246,0.001077394,0.0004386366,0.0004466414],"domain_scores_gemma":[0.9978544,0.0005873411,0.0002022996,0.001142023,0.00007509434,0.0001389016],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.0006523174,0.0001551298,0.008500098,0.0005205016,0.0003951413,0.0004269025,0.03587686,0.005035139,0.0005170593,0.0952634,0.02005792,0.8325995],"study_design_scores_gemma":[0.001611107,0.0001082129,0.01515091,0.0008823833,0.00003890965,0.00001021838,0.01719204,0.928378,0.00003689353,0.02271592,0.01336587,0.0005095704],"study_design_candidate":"simulation_or_modeling","study_design_consensus":null,"genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.5033103,0.0001023499,0.4636475,0.02374627,0.0003188543,0.001315591,0.005031565,0.0004024374,0.002125086],"genre_scores_gemma":[0.9887789,0.00138074,0.00455946,0.00005011843,0.00008228621,0.000002756482,0.004932024,0.00001204319,0.0002016287],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.9233428,"threshold_uncertainty_score":0.9974289,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.3832035130446835,"score_gpt":0.3216727170267064,"score_spread":0.06153079601797717,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}