{"id":"W2890631628","doi":"10.1016/j.dib.2018.08.099","title":"MiBio: A dataset for OCR post-processing evaluation","year":2018,"lang":"en","type":"article","venue":"Data in Brief","topic":"Natural Language Processing Techniques","field":"Computer Science","cited_by":4,"is_retracted":false,"has_abstract":true,"ca_institutions":"Dalhousie University","funders":"","keywords":"Computer science; Ground truth; Natural language processing; Preprocessor; Artificial intelligence; Benchmark (surveying); Sentence; Segmentation; Information retrieval; Optical character recognition; Word (group theory); Text segmentation; Linguistics; Image (mathematics); Cartography","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"codex-gemma-dda1882f352a","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.001475037,0.00009964844,0.00010372,0.0001104344,0.0001070792,0.0002885607,0.002493677,0.00005923976,0.00001491826],"category_scores_gemma":[0.0008713608,0.00009173377,0.000009487599,0.0003542878,0.0000653118,0.002027888,0.0009467617,0.00008959841,0.00001555102],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.00004901392,"about_ca_system_score_gemma":0.0001733553,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.0001866456,"about_ca_topic_score_gemma":0.0002204312,"domain_scores_codex":[0.9986487,0.0000458507,0.0001950279,0.0005676201,0.0002912473,0.0002515744],"domain_scores_gemma":[0.9980428,0.00006104478,0.00009559077,0.001505894,0.0002554531,0.0000392513],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"design_other","study_design_gemma":"simulation_or_modeling","study_design_scores_codex":[0.00002760316,0.00006500528,0.00008601671,0.00007249298,0.000004954078,0.000006311865,0.0003673685,3.382368e-7,0.004366961,0.003041904,0.1731883,0.8187727],"study_design_scores_gemma":[0.001293787,0.0002704967,0.0006896061,0.000289615,0.00003156291,0.00005334061,0.00003727019,0.6831161,0.02467163,0.04353651,0.2453678,0.0006422817],"study_design_candidate":"design_other","study_design_consensus":null,"genre_codex":"methods","genre_gemma":"methods","genre_scores_codex":[0.00154723,0.001731701,0.987125,0.002811476,0.0002664616,0.0006847011,0.005413932,0.0003100985,0.0001093906],"genre_scores_gemma":[0.1103724,0.000003065087,0.8681045,0.002457344,0.000191232,0.0000369901,0.01881162,0.00001140284,0.00001136993],"genre_candidate":"methods","genre_consensus":"methods","teacher_disagreement_score":0.8181305,"threshold_uncertainty_score":0.4633914,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.06855534351809031,"score_gpt":0.3859505036611572,"score_spread":0.3173951601430669,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}