{"id":"W4310505686","doi":"10.5281/zenodo.7267245","title":"Exploring Data Provenance in Handwritten Text Recognition Infrastructure: Sharing and Reusing Ground Truth Data, Referencing Models, and Acknowledging Contributions. Starting the Conversation on How We Could Get It Done","year":2022,"lang":"en","type":"article","venue":"Data Archiving and Networked Services (DANS)","topic":"Research Data Management Practices","field":"Computer Science","cited_by":6,"is_retracted":false,"has_abstract":true,"ca_institutions":"Université de Montréal","funders":"","keywords":"Conversation; Ground truth; Reuse; Computer science; Provenance; Natural language processing; Data science; Artificial intelligence; Linguistics; Engineering; Geology","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":["metaresearch","open_science"],"consensus_categories":[],"category_scores_codex":[0.1286815,0.001005491,0.0009548031,0.00909694,0.009159187,0.02210806,0.005399065,0.004507417,0.003890492],"category_scores_gemma":[0.3014784,0.001576994,0.001352,0.01023964,0.01659052,0.06010564,0.0244526,0.006143039,0.002243234],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.006029717,"about_ca_system_score_gemma":0.01717859,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.009355881,"about_ca_topic_score_gemma":0.01227891,"domain_scores_codex":[0.8989589,0.06327155,0.008839265,0.01038152,0.01630619,0.002242642],"domain_scores_gemma":[0.6152214,0.1398563,0.02245598,0.1731015,0.04468345,0.004681438],"domain_codex":null,"domain_gemma":"reproducibility","domain_candidate":"reproducibility","domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"theoretical_or_conceptual","study_design_scores_codex":[0.0004119583,0.0002342569,0.02291589,0.001561508,0.0001611354,0.001541892,0.2392265,0.0051132,0.008958131,0.2205365,0.01340142,0.4859375],"study_design_scores_gemma":[0.00005852752,0.0002466065,0.005935459,0.003748426,0.0001825864,0.001288056,0.08297715,0.0124401,0.02216258,0.4567014,0.413873,0.0003860281],"study_design_candidate":"theoretical_or_conceptual","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"empirical","genre_scores_codex":[0.07954518,0.00354128,0.8226824,0.05040426,0.001332853,0.001082213,0.001368561,0.003474368,0.03656885],"genre_scores_gemma":[0.3849654,0.001997627,0.5927703,0.00262368,0.0004551079,0.0007053994,0.001950179,0.002090805,0.01244144],"genre_candidate":"empirical","genre_consensus":null,"teacher_disagreement_score":0.994601,"threshold_uncertainty_score":0.6805411,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2746835107824994,"score_gpt":0.3274123300538482,"score_spread":0.05272881927134881,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}