Skip to content
Draft
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@ Note that this project **does not** adhere to [Semantic Versioning](https://semv
### Added

- We added `jabkit git merge-driver`, a Git merge driver that merges `.bib` files semantically. [#16838](https://github.com/JabRef/jabref/pull/16838)
- Extracting references from a PDF now recognizes bibliographies typeset with biblatex, including alphabetic labels such as `[AL26]`. [#775](https://github.com/JabRef/jabref-koppor/pull/775)

### Changed

Expand Down
13 changes: 13 additions & 0 deletions docs/requirements/import.md
Original file line number Diff line number Diff line change
Expand Up @@ -41,4 +41,17 @@ A file that parses with warnings is not affected: it still opens, and its warnin

Needs: impl, utest

## Rule-based reference extraction reads labelled BibTeX reference lists
`req~import.pdf.references.labelled~1`

The rule-based extraction splits a PDF's reference list into references at their labels, as typeset by BibTeX and biblatex styles: numeric (`[12]`) or alphabetic (`[AL26]`, `[Buc+23]`, `[BSG+ 23]`).
Each entry carries the printed reference in `comment`.

Fields are read for two layouts: IEEE (`J. Knaster et al., “Title”, Nucl. Fusion, vol. 57, …`) and the biblatex standard styles (`Authors. “Title”. In: …`).
There, field values such as the DOI are kept as printed, so faults of the typeset list stay visible.

Reference lists without labels, such as those of author-year styles, are not supported.

Needs: impl, utest

<!-- markdownlint-disable-file MD022 -->
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,9 @@
import java.nio.file.Path;
import java.util.ArrayList;
import java.util.List;
import java.util.Locale;
import java.util.Optional;
import java.util.regex.MatchResult;
import java.util.regex.Matcher;
import java.util.regex.Pattern;

Expand Down Expand Up @@ -35,7 +37,7 @@

/// Parses the references from the "References" section from a PDF.
///
/// Currently, IEEE two column format is supported.
/// Currently, IEEE two column format and the biblatex standard styles (e.g., `alphabetic`, `numeric`) are supported.
///
/// To extract a [BibEntry] matching the PDF, see [PdfContentImporter].
///
Expand All @@ -45,7 +47,12 @@ public class RuleBasedBibliographyPdfImporter extends BibliographyFromPdfImporte

private static final Logger LOGGER = LoggerFactory.getLogger(RuleBasedBibliographyPdfImporter.class);

private static final Pattern REFERENCE_PATTERN = Pattern.compile("\\[(\\d+)\\](.*?)(?=\\[|$)", Pattern.DOTALL);
/// Numeric labels (`[12]`) and alphabetic labels of BibTeX and biblatex styles (`[AL26]`, `[Buc+23]`, `[adr26]`, `[Kop+18a]`).
/// BibTeX's `alpha.bst` typesets "et al." as a raised "+", which the PDF text shows as `[BSG+ 23]`.
///
/// [impl->req~import.pdf.references.labelled~1]
private static final String LABEL = "\\[(\\d+|\\p{L}[\\p{L}'-]*(?:\\+ ?)?\\d{2}[a-z]?)\\]";
private static final Pattern REFERENCE_PATTERN = Pattern.compile(LABEL + "(.*?)(?=" + LABEL + "|$)", Pattern.DOTALL);
private static final Pattern YEAR_AT_END = Pattern.compile(", (\\d{4})\\.$");
private static final Pattern YEAR = Pattern.compile(", (\\d{4})(.*)");
private static final Pattern PAGES = Pattern.compile(", pp\\. (\\d+--?\\d+)\\.?(.*)");
Expand All @@ -60,6 +67,20 @@ public class RuleBasedBibliographyPdfImporter extends BibliographyFromPdfImporte
private static final Pattern AUTHORS_AND_TITLE_AT_BEGINNING = Pattern.compile("^([^“]+), “(.*?)(”,|,”) ");
private static final Pattern TITLE = Pattern.compile("“(.*?)”, (.*)");

// biblatex standard styles: `Authors. “Title”. In: Container. issn: …. doi: …. url: … (visited on …).`
private static final Pattern BIBLATEX_AUTHORS_AND_TITLE = Pattern.compile("^(.+?)\\. “(.+?)”\\.? ?(.*)$");
// Titles of books and online resources are printed in italics, not in quotes
private static final Pattern BIBLATEX_AUTHORS_AND_UNQUOTED_TITLE = Pattern.compile("^(.*?\\p{L}{2})\\. ?(.+?)\\. (.*)$");
private static final Pattern BIBLATEX_FIELD_LABEL = Pattern.compile("\\b(issn|isbn|doi|url): ", Pattern.CASE_INSENSITIVE);
private static final Pattern BIBLATEX_VISITED_ON = Pattern.compile("\\(visited on [^)]*\\)");
private static final Pattern BIBLATEX_ARTICLE = Pattern.compile("^In: (?:(.*?[^-\\s]) )?(?:(\\d+)(?:\\.(\\S+))? )?\\([^)]*?(\\d{4})\\)(?:, pp?\\. (\\S+?))?\\.?$");
private static final Pattern BIBLATEX_PAGES_AT_END = Pattern.compile(", pp?\\. (\\S+?)\\.?$");
private static final Pattern BIBLATEX_YEAR_AT_END = Pattern.compile("(?:^|[ ,])(\\d{4})\\.?$");
private static final Pattern BIBLATEX_VOLUME_AND_SERIES_AT_END = Pattern.compile("\\. Vol\\. (\\S+?)(?:\\. (.+))?$");
private static final Pattern BIBLATEX_EDITORS_AT_END = Pattern.compile("\\. Ed\\. by (.+)$");
// The page number printed below the last reference on a page ends up after the reference's final dot
private static final Pattern TRAILING_PAGE_NUMBER = Pattern.compile("\\.\\s+\\d{1,4}$");

@Nullable private final CitationKeyPatternPreferences citationKeyPatternPreferences;
private final NormalizeUnicodeFormatter normalizeUnicodeFormatter = new NormalizeUnicodeFormatter();

Expand Down Expand Up @@ -107,15 +128,15 @@ public ParserResult importDatabase(Path filePath, PDDocument document) throws IO
}

@VisibleForTesting
record IntermediateData(String number, String reference) {
record IntermediateData(String label, String reference) {
}

/// In: `[1] ...\n...\n...[2]...\n...\n...\n[3]...`
public List<BibEntry> getEntriesFromPDFContent(String contents) {
List<IntermediateData> referencesStrings = getIntermediateData(contents);

return referencesStrings.stream()
.map(data -> parsePlainCitation(data.number(), data.reference()))
.map(data -> parsePlainCitation(data.label(), data.reference()))
.toList();
}

Expand All @@ -125,7 +146,7 @@ static List<IntermediateData> getIntermediateData(String contents) {
Matcher matcher = REFERENCE_PATTERN.matcher(contents);
while (matcher.find()) {
String reference = matcher.group(2).replaceAll("\\r?\\n", " ").trim();
referencesStrings.add(new IntermediateData(matcher.group(1), reference));
referencesStrings.add(new IntermediateData(matcher.group(1).replace(" ", ""), reference));
}
return referencesStrings;
}
Expand All @@ -137,18 +158,19 @@ public Optional<BibEntry> parsePlainCitation(String reference) {

/// Example: `J. Knaster et al., “Overview of the IFMIF/EVEDA project”, Nucl. Fusion, vol. 57, p. 102016, 2017. doi:10.1088/ 1741-4326/aa6a6a`
@VisibleForTesting
BibEntry parsePlainCitation(String number, String reference) {
BibEntry parsePlainCitation(String label, String reference) {
reference = normalizeUnicodeFormatter.format(reference);
String originalReference = "[" + number + "] " + reference;
String originalReference = "[" + label + "] " + reference;

if (BIBLATEX_AUTHORS_AND_TITLE.matcher(reference).find()
|| (!reference.contains("“") && BIBLATEX_FIELD_LABEL.matcher(reference).find())) {
return parseBiblatexCitation(label, reference).withField(StandardField.COMMENT, originalReference);
}

BibEntry result = new BibEntry(StandardEntryType.Article)
.withCitationKey(number);
.withCitationKey(label);

reference = reference
.replace(".-", "-")
// Unicode en dash (used as page separator)
.replace("–", "-")
// Remove "- " introduced by linebreaks in the PDF
.replaceAll("([^ ])- ", "$1");
reference = removeLineBreakArtifacts(reference);

// Move URL to URL field
Matcher urlPatternMatcher = URLCleanup.URL_PATTERN.matcher(reference);
Expand Down Expand Up @@ -230,13 +252,7 @@ BibEntry parsePlainCitation(String number, String reference) {
// Y. Shimosaki et al., “Lattice design for 5 MeV – 125 mA CW RFQ operation in LIPAc”, in Proc. IPAC’19, Mel- bourne, Australia
matcher = AUTHORS_AND_TITLE_AT_BEGINNING.matcher(reference);
if (matcher.find()) {
String authors = matcher.group(1).replaceAll("et al\\.?", "and others");

// Alternative: AuthorList.fixAuthorFirstNameFirst(authors) only
// However, this does not work with special cases. Thus, we do a simple transformation only.
String fixedAuthors = AuthorListParser.normalizeSimply(authors).orElseGet(() -> AuthorList.fixAuthorFirstNameFirst(authors));

result.setField(StandardField.AUTHOR, fixedAuthors);
result.setField(StandardField.AUTHOR, normalizeAuthors(matcher.group(1)));
result.setField(StandardField.TITLE, matcher.group(2).replaceAll("et al\\.?", "and others"));
reference = reference.substring(matcher.end()).trim();
} else {
Expand Down Expand Up @@ -332,6 +348,138 @@ BibEntry parsePlainCitation(String number, String reference) {
return result;
}

private static String removeLineBreakArtifacts(String reference) {
return reference
.replace(".-", "-")
// Unicode en dash (used as page separator)
.replace("–", "-")
// Remove "- " introduced by linebreaks in the PDF
.replaceAll("([^ ])- ", "$1");
}

private static String normalizeAuthors(String authorsText) {
String authors = authorsText.replaceAll("et al\\.?", "and others");
// Alternative: AuthorList.fixAuthorFirstNameFirst(authors) only
// However, this does not work with special cases. Thus, we do a simple transformation only.
return AuthorListParser.normalizeSimply(authors).orElseGet(() -> AuthorList.fixAuthorFirstNameFirst(authors));
}

/// Parses a reference formatted by one of the biblatex standard styles.
///
/// [impl->req~import.pdf.references.labelled~1]
///
/// Example: `Aisha Alansari and Hamzah Luqman. “Large language models hallucination: A comprehensive survey”. In: Computer Science Review 61 (2026), p. 100970. issn: 1574-0137. doi: 10.1016/j.cosrev.2026.100970.`
private static BibEntry parseBiblatexCitation(String label, String reference) {
BibEntry result = new BibEntry(StandardEntryType.Misc).withCitationKey(label);
reference = TRAILING_PAGE_NUMBER.matcher(reference.trim()).replaceFirst(".");

// issn, isbn, doi, and url come last. Their values are taken as printed:
// there, a line break at a "-" does not hyphenate, the "-" is part of the value.
List<MatchResult> fieldLabels = BIBLATEX_FIELD_LABEL.matcher(reference).results().toList();
for (int i = 0; i < fieldLabels.size(); i++) {
MatchResult fieldLabel = fieldLabels.get(i);
Field field = switch (fieldLabel.group(1).toLowerCase(Locale.ROOT)) {
case "issn" ->
StandardField.ISSN;
case "isbn" ->
StandardField.ISBN;
case "doi" ->
StandardField.DOI;
default ->
StandardField.URL;
};
int valueEnd = i + 1 < fieldLabels.size() ? fieldLabels.get(i + 1).start() : reference.length();
String value = BIBLATEX_VISITED_ON.matcher(reference.substring(fieldLabel.end(), valueEnd)).replaceAll("")
.replaceAll("\\s", "")
.replaceAll("\\.$", "");
result.setField(field, value);
}

String head = fieldLabels.isEmpty() ? reference : reference.substring(0, fieldLabels.getFirst().start());
String rest = removeLineBreakArtifacts(head).trim();
Matcher authorsAndTitle = BIBLATEX_AUTHORS_AND_TITLE.matcher(rest);
if (!authorsAndTitle.matches()) {
authorsAndTitle = BIBLATEX_AUTHORS_AND_UNQUOTED_TITLE.matcher(rest);
}
if (authorsAndTitle.matches()) {
result.setField(StandardField.AUTHOR, normalizeBiblatexNames(authorsAndTitle.group(1)));
result.setField(StandardField.TITLE, authorsAndTitle.group(2));
rest = authorsAndTitle.group(3).trim();
}

Matcher article = BIBLATEX_ARTICLE.matcher(rest);
if (article.matches()) {
result.setType(StandardEntryType.Article);
setFieldIfPresent(result, StandardField.JOURNAL, article.group(1));
setFieldIfPresent(result, StandardField.VOLUME, article.group(2));
setFieldIfPresent(result, StandardField.NUMBER, article.group(3));
result.setField(StandardField.YEAR, article.group(4));
setFieldIfPresent(result, StandardField.PAGES, article.group(5));
return result;
}

boolean inContainer = rest.startsWith("In: ");
if (inContainer) {
result.setType(StandardEntryType.InProceedings);
rest = rest.substring("In: ".length());
}
rest = takeFromEnd(rest, BIBLATEX_PAGES_AT_END, result, StandardField.PAGES);
rest = takeFromEnd(rest, BIBLATEX_YEAR_AT_END, result, StandardField.YEAR);
if (rest.endsWith(",")) {
// "Publisher, 2019" or "Location: Publisher, 2019"
rest = rest.substring(0, rest.length() - 1);
int publisherStart = rest.lastIndexOf(". ") + 1;
String publisher = rest.substring(publisherStart).trim();
int locationEnd = publisher.indexOf(": ");
if (locationEnd >= 0) {
result.setField(StandardField.LOCATION, publisher.substring(0, locationEnd));
publisher = publisher.substring(locationEnd + 2);
}
result.setField(StandardField.PUBLISHER, publisher);
rest = rest.substring(0, publisherStart);
}
if (!inContainer) {
return result;
}

rest = rest.replaceAll("\\.$", "");
rest = takeFromEnd(rest, BIBLATEX_VOLUME_AND_SERIES_AT_END, result, StandardField.VOLUME, StandardField.SERIES);
Matcher editors = BIBLATEX_EDITORS_AT_END.matcher(rest);
if (editors.find()) {
result.setField(StandardField.EDITOR, normalizeBiblatexNames(editors.group(1)));
rest = rest.substring(0, editors.start());
}
setFieldIfPresent(result, StandardField.BOOKTITLE, rest);
return result;
}

/// biblatex prints names as "Given Family", thus a comma only separates names: "A, B, and C".
private static String normalizeBiblatexNames(String names) {
return names.replace(", and ", " and ")
.replace(", ", " and ")
.replaceAll(" et al\\.?$", " and others");
}

/// Moves the groups of `pattern`, which has to match at the end of `reference`, into `fields`.
///
/// @return `reference` without the matched part
private static String takeFromEnd(String reference, Pattern pattern, BibEntry entry, Field... fields) {
Matcher matcher = pattern.matcher(reference);
if (!matcher.find()) {
return reference;
}
for (int i = 0; i < fields.length; i++) {
setFieldIfPresent(entry, fields[i], matcher.group(i + 1));
}
return reference.substring(0, matcher.start()).trim();
}

private static void setFieldIfPresent(BibEntry entry, Field field, @Nullable String value) {
if (value != null && !value.isBlank()) {
entry.setField(field, value.trim());
}
}

/// @param pattern A pattern matching two groups: The first one to take, the second one to leave at the end of the string
private static EntryUpdateResult updateEntryAndReferenceIfMatches(String reference, Pattern pattern, BibEntry result, Field
field) {
Expand Down
Loading
Loading