{"id":"W4288406204","doi":"10.48550/arxiv.1903.11242","title":"An Empirical Study on Practicality of Specification Mining Algorithms on\\n a Real-world Application","year":2019,"lang":"","type":"preprint","venue":"arXiv (Cornell University)","topic":"Software Testing and Debugging Techniques","field":"Computer Science","cited_by":0,"is_retracted":false,"has_abstract":true,"ca_institutions":"University of Calgary","funders":"","keywords":"Debugging; Computer science; Program comprehension; Context (archaeology); Implementation; Inference; Programming language; Abstraction; Software engineering; Set (abstract data type); Root cause; Software; Algorithm; Artificial intelligence; Software system; Reliability engineering","routes":{"ca_aff":true,"ca_fund":false,"ca_venue":false,"about_ca":false,"invisible_to_affiliation_only":false},"retraction":null,"screen":null,"direct_labels":[],"prediction":{"model_version":"metacan-v3-hybrid-931329e0061c","candidate_categories":[],"consensus_categories":[],"category_scores_codex":[0.02834874,0.0008949707,0.0006463881,0.002225864,0.0008902073,0.002403131,0.002094378,0.001808004,0.002767748],"category_scores_gemma":[0.2885101,0.0004877446,0.0009411935,0.002984843,0.002097559,0.005335395,0.001674277,0.002745275,0.001030111],"about_ca_system_candidate":false,"about_ca_system_consensus":false,"about_ca_system_score_codex":0.001686851,"about_ca_system_score_gemma":0.001359975,"about_ca_topic_candidate":false,"about_ca_topic_consensus":false,"about_ca_topic_score_codex":0.002467357,"about_ca_topic_score_gemma":0.002893594,"domain_scores_codex":[0.9740233,0.01493227,0.002529435,0.003604739,0.004198044,0.0007122195],"domain_scores_gemma":[0.4254025,0.5207703,0.01007563,0.03052773,0.01128624,0.001937525],"domain_codex":null,"domain_gemma":null,"domain_candidate":null,"domain_consensus":null,"study_design_codex":"observational","study_design_gemma":"observational","study_design_scores_codex":[0.005682054,0.006078356,0.4264558,0.002445688,0.0009060954,0.0005360077,0.003547087,0.1135393,0.006807907,0.01115452,0.01783093,0.4050162],"study_design_scores_gemma":[0.0007815501,0.003840539,0.1537514,0.0004740728,0.0003759144,0.001437358,0.003862404,0.788905,0.01280497,0.01444282,0.01916987,0.0001540323],"study_design_candidate":"observational","study_design_consensus":"observational","genre_codex":"empirical","genre_gemma":"empirical","genre_scores_codex":[0.9712934,0.001579699,0.01780027,0.001457939,0.00007743448,0.0002571405,0.001556335,0.0008663319,0.005111307],"genre_scores_gemma":[0.977325,0.0003620272,0.01845197,0.0001252246,0.00003779634,0.0001319939,0.002750203,0.0001710919,0.0006447505],"genre_candidate":"empirical","genre_consensus":"empirical","teacher_disagreement_score":0.02834874,"threshold_uncertainty_score":0.1499243,"prediction_status":"machine_predicted_unvalidated"},"machine_scores":{"provisional":true,"baseline":true,"maturity_gate_passed":false,"score_opus":0.2026427942074081,"score_gpt":0.3137623839398878,"score_spread":0.1111195897324797,"validation_status":"score_only:v0-immature-baseline","note":"Baseline scores from an immature model (maturity gate not passed). Scores rank; they never assert a category."}}