@inproceedings{53d03f55dc18426fa8c7590bfe9b807a,
title = "Empirical evaluation of CRF-based bibliography extraction from reference strings",
abstract = "This paper reports an empirical evaluation of a CRF-based bibliography parser we have developed for reference strings of research papers. The parser uses a conditional random field (CRF) to estimate the correct bibliographic label such as an author's name and a title for each token in a reference string. We applied the parser specifically designed for reference strings to three academic journals, an English one and two Japanese ones, published in Japan. Experiments showed i) the parser correctly parsed from 90% to 94% of reference strings depending on the kinds of journals used and ii) segmentation errors induced by tokenization considerably degraded the final parsing accuracies. This paper also discusses some future directions of the bibliography extraction based on a detailed analysis of the experiments.",
keywords = "Bibliography extraction, Citation parsing, Conditional random field, Evaluation, Metadata",
author = "Manabu Ohta and Daiki Arauchi and Atsuhiro Takasu and Jun Adachi",
year = "2014",
month = jan,
day = "1",
doi = "10.1109/DAS.2014.64",
language = "English",
isbn = "9781479932436",
series = "Proceedings - 11th IAPR International Workshop on Document Analysis Systems, DAS 2014",
publisher = "IEEE Computer Society",
pages = "287--292",
booktitle = "Proceedings - 11th IAPR International Workshop on Document Analysis Systems, DAS 2014",
address = "United States",
note = "11th IAPR International Workshop on Document Analysis Systems, DAS 2014 ; Conference date: 07-04-2014 Through 10-04-2014",
}