- Bibliography heading now accepts bold markers (## **BIBLIOGRAPHY**), alternative names (Works Cited, References), and numbered sub-headings within the section. - Original heading text preserved in output instead of hardcoded "## Bibliography". - New find_blockquote_notes() parser for PDF-to-markdown footnote format (> N text), wired into CLI as third fallback after pandoc and numbered. - PDF junk filtered from bibliography entries: bare page numbers, blockquote footnotes, download banners, CC license URLs, and short running headers.
563 lines
18 KiB
Python
563 lines
18 KiB
Python
"""Tests for src/cmos/parser.py — extract the bibliography section from a draft.
|
|
|
|
v1 scope: locate a ``## Bibliography`` heading (case-insensitive) and return
|
|
the prose before it, the list of entries inside it, and the prose after it.
|
|
The section ends at the next ``##`` (level-2) heading or end of file.
|
|
Blank lines and bullet markers (``- `` or ``* ``) are stripped from entries.
|
|
|
|
v2 (notes) parser scope: two complementary functions for finding footnote
|
|
definitions in two distinct source formats:
|
|
|
|
- ``find_notes`` finds pandoc-style ``[^marker]: text`` definitions
|
|
(caret-prefixed marker, colon after the closing bracket).
|
|
- ``find_numbered_notes`` finds ``[N] text`` definitions (square bracket
|
|
numeric marker, no caret, no colon, just a space before the body) —
|
|
the format produced by docx-to-text conversion of footnoted Word docs.
|
|
|
|
Both functions return ``NotesParseResult`` with a list of ``NoteDefinition``
|
|
records carrying the marker, body text, line number, and the original
|
|
prefix string (so reassembly can round-trip the source's marker syntax).
|
|
|
|
Single-line definitions only in both formats; multi-line continuation is
|
|
deferred.
|
|
"""
|
|
|
|
import pytest
|
|
|
|
from cmos.parser import (
|
|
NoBibliographyError,
|
|
find_blockquote_notes,
|
|
find_notes,
|
|
find_numbered_notes,
|
|
split_bibliography,
|
|
)
|
|
|
|
|
|
def test_three_entries_under_bibliography_heading():
|
|
draft = """\
|
|
Some prose here.
|
|
|
|
More prose.
|
|
|
|
## Bibliography
|
|
|
|
yu, charles. interior chinatown. New York: Pantheon Books, 2020.
|
|
Kwon, Hyeyoung. "inclusion work." American Journal of Sociology 127, no. 6 (2022): 1818-1859.
|
|
google, "privacy policy," privacy & terms, nov 15 2023, policies.google.com/privacy
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 3
|
|
assert result.entries[0].startswith("yu, charles")
|
|
assert result.entries[1].startswith("Kwon, Hyeyoung")
|
|
assert result.entries[2].startswith("google")
|
|
assert "Some prose here." in result.before
|
|
assert result.after == ""
|
|
|
|
|
|
def test_section_ends_at_next_level_two_heading():
|
|
draft = """\
|
|
## Bibliography
|
|
|
|
Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
|
|
Kwon, Hyeyoung. "Inclusion Work," *American Journal of Sociology* 127 (2022).
|
|
|
|
## Appendix
|
|
|
|
Appendix content here.
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 2
|
|
assert result.entries[0].startswith("Yu, Charles")
|
|
assert "## Appendix" in result.after
|
|
assert "Appendix content here." in result.after
|
|
|
|
|
|
def test_bullet_prefixes_stripped():
|
|
draft = """\
|
|
## Bibliography
|
|
|
|
- Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
|
|
- Kwon, Hyeyoung. "Inclusion Work," *AJS* 127 (2022).
|
|
* Google. "Privacy Policy," Privacy & Terms, 2023.
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 3
|
|
assert result.entries[0].startswith("Yu, Charles")
|
|
assert result.entries[1].startswith("Kwon, Hyeyoung")
|
|
assert result.entries[2].startswith("Google")
|
|
|
|
|
|
def test_heading_match_is_case_insensitive():
|
|
draft = """\
|
|
## bibliography
|
|
|
|
Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 1
|
|
assert result.entries[0].startswith("Yu, Charles")
|
|
|
|
|
|
def test_heading_match_works_cited():
|
|
"""'Works Cited' is the standard CMOS heading alternative to
|
|
'Bibliography'. The parser should accept it."""
|
|
draft = """\
|
|
## Works Cited
|
|
|
|
Davidson, Donald. "On the Very Idea of a Conceptual Scheme."
|
|
Evans-Pritchard, E.E. *Witchcraft, oracles, and magic.*
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 2
|
|
assert result.entries[0].startswith("Davidson")
|
|
|
|
|
|
def test_heading_match_works_cited_bold():
|
|
"""Bold-wrapped variant: ``## **Works Cited**``."""
|
|
draft = """\
|
|
## **Works Cited**
|
|
|
|
- Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999).
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 1
|
|
assert result.entries[0].startswith("Frankenberry")
|
|
|
|
|
|
def test_heading_match_ignores_bold_markers():
|
|
"""Real-world drafts from PDF-to-markdown conversion often wrap the
|
|
heading text in bold: ``## **BIBLIOGRAPHY**``. The parser should
|
|
strip ``**`` before matching."""
|
|
draft = """\
|
|
## **BIBLIOGRAPHY**
|
|
|
|
- Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
|
|
- Kwon, Hyeyoung. "Inclusion Work," *AJS* 127 (2022).
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 2
|
|
assert result.entries[0].startswith("Yu, Charles")
|
|
|
|
|
|
def test_heading_match_ignores_numbered_sub_headings():
|
|
"""Subdivided bibliographies (common in dissertations) use numbered
|
|
``##`` sub-headings inside the bibliography section. These must NOT
|
|
terminate the section — the parser should skip them and keep
|
|
collecting entries until a non-numbered ``##`` heading or EOF."""
|
|
draft = """\
|
|
## **BIBLIOGRAPHY**
|
|
|
|
## **1. Primary Sources**
|
|
|
|
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
|
|
- Till, Walter C., ed., *Die koptischen Ostraka* (Vienna, 1960).
|
|
|
|
## **2. Secondary Literature**
|
|
|
|
- Marsham, Andrew, ed., *The Umayyad World* (London: Routledge, 2020).
|
|
- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 4
|
|
assert result.entries[0].startswith("Crum, Walter")
|
|
assert result.entries[3].startswith("Schulz, Fritz")
|
|
assert result.after == ""
|
|
|
|
|
|
def test_numbered_sub_headings_not_collected_as_entries():
|
|
"""The sub-heading lines themselves (e.g. ``## **1. Primary Sources**``)
|
|
should not appear in the entries list."""
|
|
draft = """\
|
|
## Bibliography
|
|
|
|
## 1. Primary Sources
|
|
|
|
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
|
|
|
|
## 2. Secondary Sources
|
|
|
|
- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 2
|
|
for entry in result.entries:
|
|
assert not entry.startswith("##")
|
|
assert "Primary Sources" not in entry
|
|
assert "Secondary Sources" not in entry
|
|
|
|
|
|
def test_non_numbered_heading_still_ends_section_after_sub_headings():
|
|
"""A non-numbered ``##`` heading after bibliography sub-sections
|
|
should still terminate the bibliography section."""
|
|
draft = """\
|
|
## Bibliography
|
|
|
|
## 1. Primary Sources
|
|
|
|
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
|
|
|
|
## Appendix
|
|
|
|
Appendix content.
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 1
|
|
assert result.entries[0].startswith("Crum, Walter")
|
|
assert "## Appendix" in result.after
|
|
assert "Appendix content." in result.after
|
|
|
|
|
|
def test_bare_page_numbers_filtered():
|
|
"""PDF-to-markdown conversion leaves bare page numbers (e.g. '370')
|
|
as standalone lines. These are not bibliography entries."""
|
|
draft = """\
|
|
## Bibliography
|
|
|
|
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
|
|
|
|
370
|
|
|
|
- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
|
|
|
|
371
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 2
|
|
assert result.entries[0].startswith("Crum")
|
|
assert result.entries[1].startswith("Schulz")
|
|
|
|
|
|
def test_blockquote_footnotes_filtered():
|
|
"""Blockquote footnotes (``> N text``) that appear inside a Works Cited
|
|
section are PDF artifacts, not bibliography entries."""
|
|
draft = """\
|
|
## Works Cited
|
|
|
|
Davidson, Donald. "On the Very Idea of a Conceptual Scheme."
|
|
|
|
> 87 Holbraad 2010.
|
|
> 88 Frankenberry 2018, 236.
|
|
|
|
Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999).
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 2
|
|
assert result.entries[0].startswith("Davidson")
|
|
assert result.entries[1].startswith("Frankenberry")
|
|
|
|
|
|
def test_download_banners_filtered():
|
|
"""Brill/publisher download banners are PDF junk, not entries."""
|
|
draft = """\
|
|
## Bibliography
|
|
|
|
Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
|
|
Downloaded from Brill.com 02/22/2024 05:33:29AM via Open Access. This is junk.
|
|
https://creativecommons.org/licenses/by/4.0/
|
|
Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 2
|
|
assert result.entries[0].startswith("Crum")
|
|
assert result.entries[1].startswith("Schulz")
|
|
|
|
|
|
def test_running_headers_filtered():
|
|
"""Short lines that are running headers or page footers (e.g. an
|
|
author surname or abbreviated title) are not entries."""
|
|
draft = """\
|
|
## Bibliography
|
|
|
|
Davidson, Donald. "On the Very Idea of a Conceptual Scheme."
|
|
hedrick
|
|
only words apart?
|
|
217
|
|
Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999).
|
|
"""
|
|
result = split_bibliography(draft)
|
|
assert len(result.entries) == 2
|
|
assert result.entries[0].startswith("Davidson")
|
|
assert result.entries[1].startswith("Frankenberry")
|
|
|
|
|
|
def test_no_bibliography_raises():
|
|
with pytest.raises(NoBibliographyError):
|
|
split_bibliography("## Introduction\n\nJust prose.\n")
|
|
|
|
|
|
# --- find_notes (v2 parser, pandoc-style markdown footnote definitions) ----
|
|
|
|
|
|
def test_find_notes_single_definition():
|
|
draft = """\
|
|
Some prose with a citation.[^1]
|
|
|
|
[^1]: Charles Yu, *Interior Chinatown* (Pantheon Books, 2020), 45.
|
|
"""
|
|
result = find_notes(draft)
|
|
assert len(result.definitions) == 1
|
|
assert result.definitions[0].marker == "1"
|
|
assert result.definitions[0].text == "Charles Yu, *Interior Chinatown* (Pantheon Books, 2020), 45."
|
|
|
|
|
|
def test_find_notes_multiple_definitions_in_order():
|
|
draft = """\
|
|
Prose.[^1] More prose.[^2] And again.[^3]
|
|
|
|
[^1]: First note.
|
|
[^2]: Second note.
|
|
[^3]: Third note.
|
|
"""
|
|
result = find_notes(draft)
|
|
assert len(result.definitions) == 3
|
|
assert [d.marker for d in result.definitions] == ["1", "2", "3"]
|
|
assert [d.text for d in result.definitions] == [
|
|
"First note.",
|
|
"Second note.",
|
|
"Third note.",
|
|
]
|
|
|
|
|
|
def test_find_notes_named_marker():
|
|
draft = """\
|
|
Citation.[^smith2020]
|
|
|
|
[^smith2020]: Smith, J. (2020). Some work.
|
|
"""
|
|
result = find_notes(draft)
|
|
assert len(result.definitions) == 1
|
|
assert result.definitions[0].marker == "smith2020"
|
|
assert result.definitions[0].text == "Smith, J. (2020). Some work."
|
|
|
|
|
|
def test_find_notes_returns_empty_when_no_definitions():
|
|
draft = "Just prose with no footnote definitions.\n"
|
|
result = find_notes(draft)
|
|
assert result.definitions == []
|
|
|
|
|
|
def test_find_notes_records_line_number():
|
|
draft = """\
|
|
Line zero.
|
|
Line one.
|
|
|
|
[^1]: Note on line three.
|
|
"""
|
|
result = find_notes(draft)
|
|
assert len(result.definitions) == 1
|
|
assert result.definitions[0].line_number == 3
|
|
|
|
|
|
def test_find_notes_strips_trailing_whitespace_from_text():
|
|
draft = "[^1]: Note text with trailing spaces. \n"
|
|
result = find_notes(draft)
|
|
assert result.definitions[0].text == "Note text with trailing spaces."
|
|
|
|
|
|
def test_find_notes_definition_can_appear_before_reference():
|
|
# Pandoc allows definitions anywhere in the doc, not just at the bottom.
|
|
draft = """\
|
|
[^1]: A definition before the reference.
|
|
|
|
Some prose that cites it.[^1]
|
|
"""
|
|
result = find_notes(draft)
|
|
assert len(result.definitions) == 1
|
|
assert result.definitions[0].text == "A definition before the reference."
|
|
|
|
|
|
def test_find_notes_records_original_prefix_for_pandoc_format():
|
|
"""Reassembly needs the literal prefix string so the round-trip
|
|
preserves the source's marker syntax. For pandoc this is `[^N]: `."""
|
|
draft = "[^1]: text.\n[^smith2020]: text.\n"
|
|
result = find_notes(draft)
|
|
assert result.definitions[0].original_prefix == "[^1]: "
|
|
assert result.definitions[1].original_prefix == "[^smith2020]: "
|
|
|
|
|
|
# --- find_numbered_notes ([N] text format, no caret, no colon) -------------
|
|
|
|
|
|
def test_find_numbered_notes_single_definition():
|
|
draft = """\
|
|
Some prose.
|
|
|
|
[1] Charles Yu, Interior Chinatown. Pantheon, 2020. p. 45.
|
|
"""
|
|
result = find_numbered_notes(draft)
|
|
assert len(result.definitions) == 1
|
|
assert result.definitions[0].marker == "1"
|
|
assert result.definitions[0].text == "Charles Yu, Interior Chinatown. Pantheon, 2020. p. 45."
|
|
|
|
|
|
def test_find_numbered_notes_multiple_sequential():
|
|
draft = """\
|
|
[1] First.
|
|
[2] Second.
|
|
[3] Third.
|
|
"""
|
|
result = find_numbered_notes(draft)
|
|
assert len(result.definitions) == 3
|
|
assert [d.marker for d in result.definitions] == ["1", "2", "3"]
|
|
assert [d.text for d in result.definitions] == ["First.", "Second.", "Third."]
|
|
|
|
|
|
def test_find_numbered_notes_multi_digit_markers():
|
|
"""The Anti-Communist draft has 107 notes, including [107]. Multi-digit
|
|
markers must work."""
|
|
draft = "[1] one.\n[42] forty-two.\n[107] one-oh-seven.\n"
|
|
result = find_numbered_notes(draft)
|
|
assert [d.marker for d in result.definitions] == ["1", "42", "107"]
|
|
|
|
|
|
def test_find_numbered_notes_returns_empty_when_no_definitions():
|
|
draft = "Just prose with no numbered notes.\n"
|
|
result = find_numbered_notes(draft)
|
|
assert result.definitions == []
|
|
|
|
|
|
def test_find_numbered_notes_records_line_number():
|
|
draft = """\
|
|
Line zero.
|
|
Line one.
|
|
|
|
[1] Note on line three.
|
|
"""
|
|
result = find_numbered_notes(draft)
|
|
assert result.definitions[0].line_number == 3
|
|
|
|
|
|
def test_find_numbered_notes_strips_trailing_whitespace_from_text():
|
|
draft = "[1] Note text with trailing spaces. \n"
|
|
result = find_numbered_notes(draft)
|
|
assert result.definitions[0].text == "Note text with trailing spaces."
|
|
|
|
|
|
def test_find_numbered_notes_ignores_pandoc_format():
|
|
"""`[^1]: text` is pandoc format, not numbered. find_numbered_notes
|
|
must NOT match it (or the two parsers would step on each other when
|
|
used in fallback fashion). The `^` after `[` disqualifies it."""
|
|
draft = "[^1]: pandoc-style note.\n"
|
|
result = find_numbered_notes(draft)
|
|
assert result.definitions == []
|
|
|
|
|
|
def test_find_numbered_notes_ignores_non_numeric_markers():
|
|
"""A marker like [Smith 2020] or [foo] is NOT a numbered footnote
|
|
definition — it could be many other things (citation reference, link
|
|
label, etc.). Only digit markers count."""
|
|
draft = """\
|
|
[Smith 2020] Some text — looks like an author-date reference.
|
|
[foo] some kind of label.
|
|
[1] Actual numbered note.
|
|
"""
|
|
result = find_numbered_notes(draft)
|
|
assert len(result.definitions) == 1
|
|
assert result.definitions[0].marker == "1"
|
|
|
|
|
|
def test_find_numbered_notes_records_original_prefix():
|
|
"""For numbered format the prefix is `[N] ` — square bracket, number,
|
|
closing bracket, single space. Reassembly will use this verbatim."""
|
|
draft = "[1] text.\n[107] text.\n"
|
|
result = find_numbered_notes(draft)
|
|
assert result.definitions[0].original_prefix == "[1] "
|
|
assert result.definitions[1].original_prefix == "[107] "
|
|
|
|
|
|
def test_find_numbered_notes_handles_anti_communist_draft_format():
|
|
"""Smoke test mirroring the actual Anti-Communist Formations draft:
|
|
notes follow a `## Notes` heading, separated by `________________` rule,
|
|
one definition per line, sequentially numbered."""
|
|
draft = """\
|
|
## Notes
|
|
________________
|
|
[1] First reference. https://example.org/path
|
|
[2] Second reference, abbreviated.
|
|
[3] Rosen, 7.
|
|
"""
|
|
result = find_numbered_notes(draft)
|
|
assert len(result.definitions) == 3
|
|
assert result.definitions[0].text == "First reference. https://example.org/path"
|
|
assert result.definitions[2].text == "Rosen, 7."
|
|
|
|
|
|
# --- find_blockquote_notes (> N text format, from PDF-to-markdown) ----------
|
|
|
|
|
|
def test_find_blockquote_notes_single_definition():
|
|
draft = """\
|
|
Some prose.
|
|
|
|
> 1 Donald Davidson, "On the Very Idea of a Conceptual Scheme."
|
|
"""
|
|
result = find_blockquote_notes(draft)
|
|
assert len(result.definitions) == 1
|
|
assert result.definitions[0].marker == "1"
|
|
assert result.definitions[0].text == 'Donald Davidson, "On the Very Idea of a Conceptual Scheme."'
|
|
|
|
|
|
def test_find_blockquote_notes_multiple_in_order():
|
|
draft = """\
|
|
> 1 First note.
|
|
> 2 Second note.
|
|
> 3 Third note.
|
|
"""
|
|
result = find_blockquote_notes(draft)
|
|
assert len(result.definitions) == 3
|
|
assert [d.marker for d in result.definitions] == ["1", "2", "3"]
|
|
assert [d.text for d in result.definitions] == [
|
|
"First note.",
|
|
"Second note.",
|
|
"Third note.",
|
|
]
|
|
|
|
|
|
def test_find_blockquote_notes_multi_digit_markers():
|
|
draft = "> 1 one.\n> 42 forty-two.\n> 88 eighty-eight.\n"
|
|
result = find_blockquote_notes(draft)
|
|
assert [d.marker for d in result.definitions] == ["1", "42", "88"]
|
|
|
|
|
|
def test_find_blockquote_notes_returns_empty_when_no_definitions():
|
|
draft = "Just prose.\n> A regular blockquote with no number.\n"
|
|
result = find_blockquote_notes(draft)
|
|
assert result.definitions == []
|
|
|
|
|
|
def test_find_blockquote_notes_records_line_number():
|
|
draft = """\
|
|
Line zero.
|
|
Line one.
|
|
|
|
> 1 Note on line three.
|
|
"""
|
|
result = find_blockquote_notes(draft)
|
|
assert result.definitions[0].line_number == 3
|
|
|
|
|
|
def test_find_blockquote_notes_strips_trailing_whitespace():
|
|
draft = "> 1 Note text with trailing spaces. \n"
|
|
result = find_blockquote_notes(draft)
|
|
assert result.definitions[0].text == "Note text with trailing spaces."
|
|
|
|
|
|
def test_find_blockquote_notes_records_original_prefix():
|
|
"""Reassembly prefix for blockquote format is ``'> N '``."""
|
|
draft = "> 1 text.\n> 88 text.\n"
|
|
result = find_blockquote_notes(draft)
|
|
assert result.definitions[0].original_prefix == "> 1 "
|
|
assert result.definitions[1].original_prefix == "> 88 "
|
|
|
|
|
|
def test_find_blockquote_notes_ignores_non_numeric_blockquotes():
|
|
"""A blockquote without a leading number is regular prose, not a note."""
|
|
draft = """\
|
|
> This is a regular blockquote.
|
|
> 1 This is a note.
|
|
> Another regular blockquote.
|
|
"""
|
|
result = find_blockquote_notes(draft)
|
|
assert len(result.definitions) == 1
|
|
assert result.definitions[0].marker == "1"
|