Files
cmos/tests/test_parser.py
T
Mark Eaton cad936f576 parser: support diverse md inputs (bold headings, Works Cited, blockquote notes, junk filtering)
- Bibliography heading now accepts bold markers (## **BIBLIOGRAPHY**),
  alternative names (Works Cited, References), and numbered sub-headings
  within the section.
- Original heading text preserved in output instead of hardcoded
  "## Bibliography".
- New find_blockquote_notes() parser for PDF-to-markdown footnote format
  (> N text), wired into CLI as third fallback after pandoc and numbered.
- PDF junk filtered from bibliography entries: bare page numbers,
  blockquote footnotes, download banners, CC license URLs, and short
  running headers.
2026-04-12 00:11:52 -04:00

563 lines
18 KiB
Python

"""Tests for src/cmos/parser.py — extract the bibliography section from a draft.
v1 scope: locate a ``## Bibliography`` heading (case-insensitive) and return
the prose before it, the list of entries inside it, and the prose after it.
The section ends at the next ``##`` (level-2) heading or end of file.
Blank lines and bullet markers (``- `` or ``* ``) are stripped from entries.
v2 (notes) parser scope: two complementary functions for finding footnote
definitions in two distinct source formats:
- ``find_notes`` finds pandoc-style ``[^marker]: text`` definitions
(caret-prefixed marker, colon after the closing bracket).
- ``find_numbered_notes`` finds ``[N] text`` definitions (square bracket
numeric marker, no caret, no colon, just a space before the body) —
the format produced by docx-to-text conversion of footnoted Word docs.
Both functions return ``NotesParseResult`` with a list of ``NoteDefinition``
records carrying the marker, body text, line number, and the original
prefix string (so reassembly can round-trip the source's marker syntax).
Single-line definitions only in both formats; multi-line continuation is
deferred.
"""
import pytest
from cmos.parser import (
NoBibliographyError,
find_blockquote_notes,
find_notes,
find_numbered_notes,
split_bibliography,
)
def test_three_entries_under_bibliography_heading():
draft = """\
Some prose here.
More prose.
## Bibliography
yu, charles. interior chinatown. New York: Pantheon Books, 2020.
Kwon, Hyeyoung. "inclusion work." American Journal of Sociology 127, no. 6 (2022): 1818-1859.
google, "privacy policy," privacy & terms, nov 15 2023, policies.google.com/privacy
"""
result = split_bibliography(draft)
assert len(result.entries) == 3
assert result.entries[0].startswith("yu, charles")
assert result.entries[1].startswith("Kwon, Hyeyoung")
assert result.entries[2].startswith("google")
assert "Some prose here." in result.before
assert result.after == ""
def test_section_ends_at_next_level_two_heading():
draft = """\
## Bibliography
Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
Kwon, Hyeyoung. "Inclusion Work," *American Journal of Sociology* 127 (2022).
## Appendix
Appendix content here.
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
assert result.entries[0].startswith("Yu, Charles")
assert "## Appendix" in result.after
assert "Appendix content here." in result.after
def test_bullet_prefixes_stripped():
draft = """\
## Bibliography
- Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
- Kwon, Hyeyoung. "Inclusion Work," *AJS* 127 (2022).
* Google. "Privacy Policy," Privacy & Terms, 2023.
"""
result = split_bibliography(draft)
assert len(result.entries) == 3
assert result.entries[0].startswith("Yu, Charles")
assert result.entries[1].startswith("Kwon, Hyeyoung")
assert result.entries[2].startswith("Google")
def test_heading_match_is_case_insensitive():
draft = """\
## bibliography
Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
"""
result = split_bibliography(draft)
assert len(result.entries) == 1
assert result.entries[0].startswith("Yu, Charles")
def test_heading_match_works_cited():
"""'Works Cited' is the standard CMOS heading alternative to
'Bibliography'. The parser should accept it."""
draft = """\
## Works Cited
Davidson, Donald. "On the Very Idea of a Conceptual Scheme."
Evans-Pritchard, E.E. *Witchcraft, oracles, and magic.*
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
assert result.entries[0].startswith("Davidson")
def test_heading_match_works_cited_bold():
"""Bold-wrapped variant: ``## **Works Cited**``."""
draft = """\
## **Works Cited**
- Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999).
"""
result = split_bibliography(draft)
assert len(result.entries) == 1
assert result.entries[0].startswith("Frankenberry")
def test_heading_match_ignores_bold_markers():
"""Real-world drafts from PDF-to-markdown conversion often wrap the
heading text in bold: ``## **BIBLIOGRAPHY**``. The parser should
strip ``**`` before matching."""
draft = """\
## **BIBLIOGRAPHY**
- Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
- Kwon, Hyeyoung. "Inclusion Work," *AJS* 127 (2022).
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
assert result.entries[0].startswith("Yu, Charles")
def test_heading_match_ignores_numbered_sub_headings():
"""Subdivided bibliographies (common in dissertations) use numbered
``##`` sub-headings inside the bibliography section. These must NOT
terminate the section — the parser should skip them and keep
collecting entries until a non-numbered ``##`` heading or EOF."""
draft = """\
## **BIBLIOGRAPHY**
## **1. Primary Sources**
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
- Till, Walter C., ed., *Die koptischen Ostraka* (Vienna, 1960).
## **2. Secondary Literature**
- Marsham, Andrew, ed., *The Umayyad World* (London: Routledge, 2020).
- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
"""
result = split_bibliography(draft)
assert len(result.entries) == 4
assert result.entries[0].startswith("Crum, Walter")
assert result.entries[3].startswith("Schulz, Fritz")
assert result.after == ""
def test_numbered_sub_headings_not_collected_as_entries():
"""The sub-heading lines themselves (e.g. ``## **1. Primary Sources**``)
should not appear in the entries list."""
draft = """\
## Bibliography
## 1. Primary Sources
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
## 2. Secondary Sources
- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
for entry in result.entries:
assert not entry.startswith("##")
assert "Primary Sources" not in entry
assert "Secondary Sources" not in entry
def test_non_numbered_heading_still_ends_section_after_sub_headings():
"""A non-numbered ``##`` heading after bibliography sub-sections
should still terminate the bibliography section."""
draft = """\
## Bibliography
## 1. Primary Sources
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
## Appendix
Appendix content.
"""
result = split_bibliography(draft)
assert len(result.entries) == 1
assert result.entries[0].startswith("Crum, Walter")
assert "## Appendix" in result.after
assert "Appendix content." in result.after
def test_bare_page_numbers_filtered():
"""PDF-to-markdown conversion leaves bare page numbers (e.g. '370')
as standalone lines. These are not bibliography entries."""
draft = """\
## Bibliography
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
370
- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
371
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
assert result.entries[0].startswith("Crum")
assert result.entries[1].startswith("Schulz")
def test_blockquote_footnotes_filtered():
"""Blockquote footnotes (``> N text``) that appear inside a Works Cited
section are PDF artifacts, not bibliography entries."""
draft = """\
## Works Cited
Davidson, Donald. "On the Very Idea of a Conceptual Scheme."
> 87 Holbraad 2010.
> 88 Frankenberry 2018, 236.
Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999).
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
assert result.entries[0].startswith("Davidson")
assert result.entries[1].startswith("Frankenberry")
def test_download_banners_filtered():
"""Brill/publisher download banners are PDF junk, not entries."""
draft = """\
## Bibliography
Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
Downloaded from Brill.com 02/22/2024 05:33:29AM via Open Access. This is junk.
https://creativecommons.org/licenses/by/4.0/
Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
assert result.entries[0].startswith("Crum")
assert result.entries[1].startswith("Schulz")
def test_running_headers_filtered():
"""Short lines that are running headers or page footers (e.g. an
author surname or abbreviated title) are not entries."""
draft = """\
## Bibliography
Davidson, Donald. "On the Very Idea of a Conceptual Scheme."
hedrick
only words apart?
217
Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999).
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
assert result.entries[0].startswith("Davidson")
assert result.entries[1].startswith("Frankenberry")
def test_no_bibliography_raises():
with pytest.raises(NoBibliographyError):
split_bibliography("## Introduction\n\nJust prose.\n")
# --- find_notes (v2 parser, pandoc-style markdown footnote definitions) ----
def test_find_notes_single_definition():
draft = """\
Some prose with a citation.[^1]
[^1]: Charles Yu, *Interior Chinatown* (Pantheon Books, 2020), 45.
"""
result = find_notes(draft)
assert len(result.definitions) == 1
assert result.definitions[0].marker == "1"
assert result.definitions[0].text == "Charles Yu, *Interior Chinatown* (Pantheon Books, 2020), 45."
def test_find_notes_multiple_definitions_in_order():
draft = """\
Prose.[^1] More prose.[^2] And again.[^3]
[^1]: First note.
[^2]: Second note.
[^3]: Third note.
"""
result = find_notes(draft)
assert len(result.definitions) == 3
assert [d.marker for d in result.definitions] == ["1", "2", "3"]
assert [d.text for d in result.definitions] == [
"First note.",
"Second note.",
"Third note.",
]
def test_find_notes_named_marker():
draft = """\
Citation.[^smith2020]
[^smith2020]: Smith, J. (2020). Some work.
"""
result = find_notes(draft)
assert len(result.definitions) == 1
assert result.definitions[0].marker == "smith2020"
assert result.definitions[0].text == "Smith, J. (2020). Some work."
def test_find_notes_returns_empty_when_no_definitions():
draft = "Just prose with no footnote definitions.\n"
result = find_notes(draft)
assert result.definitions == []
def test_find_notes_records_line_number():
draft = """\
Line zero.
Line one.
[^1]: Note on line three.
"""
result = find_notes(draft)
assert len(result.definitions) == 1
assert result.definitions[0].line_number == 3
def test_find_notes_strips_trailing_whitespace_from_text():
draft = "[^1]: Note text with trailing spaces. \n"
result = find_notes(draft)
assert result.definitions[0].text == "Note text with trailing spaces."
def test_find_notes_definition_can_appear_before_reference():
# Pandoc allows definitions anywhere in the doc, not just at the bottom.
draft = """\
[^1]: A definition before the reference.
Some prose that cites it.[^1]
"""
result = find_notes(draft)
assert len(result.definitions) == 1
assert result.definitions[0].text == "A definition before the reference."
def test_find_notes_records_original_prefix_for_pandoc_format():
"""Reassembly needs the literal prefix string so the round-trip
preserves the source's marker syntax. For pandoc this is `[^N]: `."""
draft = "[^1]: text.\n[^smith2020]: text.\n"
result = find_notes(draft)
assert result.definitions[0].original_prefix == "[^1]: "
assert result.definitions[1].original_prefix == "[^smith2020]: "
# --- find_numbered_notes ([N] text format, no caret, no colon) -------------
def test_find_numbered_notes_single_definition():
draft = """\
Some prose.
[1] Charles Yu, Interior Chinatown. Pantheon, 2020. p. 45.
"""
result = find_numbered_notes(draft)
assert len(result.definitions) == 1
assert result.definitions[0].marker == "1"
assert result.definitions[0].text == "Charles Yu, Interior Chinatown. Pantheon, 2020. p. 45."
def test_find_numbered_notes_multiple_sequential():
draft = """\
[1] First.
[2] Second.
[3] Third.
"""
result = find_numbered_notes(draft)
assert len(result.definitions) == 3
assert [d.marker for d in result.definitions] == ["1", "2", "3"]
assert [d.text for d in result.definitions] == ["First.", "Second.", "Third."]
def test_find_numbered_notes_multi_digit_markers():
"""The Anti-Communist draft has 107 notes, including [107]. Multi-digit
markers must work."""
draft = "[1] one.\n[42] forty-two.\n[107] one-oh-seven.\n"
result = find_numbered_notes(draft)
assert [d.marker for d in result.definitions] == ["1", "42", "107"]
def test_find_numbered_notes_returns_empty_when_no_definitions():
draft = "Just prose with no numbered notes.\n"
result = find_numbered_notes(draft)
assert result.definitions == []
def test_find_numbered_notes_records_line_number():
draft = """\
Line zero.
Line one.
[1] Note on line three.
"""
result = find_numbered_notes(draft)
assert result.definitions[0].line_number == 3
def test_find_numbered_notes_strips_trailing_whitespace_from_text():
draft = "[1] Note text with trailing spaces. \n"
result = find_numbered_notes(draft)
assert result.definitions[0].text == "Note text with trailing spaces."
def test_find_numbered_notes_ignores_pandoc_format():
"""`[^1]: text` is pandoc format, not numbered. find_numbered_notes
must NOT match it (or the two parsers would step on each other when
used in fallback fashion). The `^` after `[` disqualifies it."""
draft = "[^1]: pandoc-style note.\n"
result = find_numbered_notes(draft)
assert result.definitions == []
def test_find_numbered_notes_ignores_non_numeric_markers():
"""A marker like [Smith 2020] or [foo] is NOT a numbered footnote
definition — it could be many other things (citation reference, link
label, etc.). Only digit markers count."""
draft = """\
[Smith 2020] Some text — looks like an author-date reference.
[foo] some kind of label.
[1] Actual numbered note.
"""
result = find_numbered_notes(draft)
assert len(result.definitions) == 1
assert result.definitions[0].marker == "1"
def test_find_numbered_notes_records_original_prefix():
"""For numbered format the prefix is `[N] ` — square bracket, number,
closing bracket, single space. Reassembly will use this verbatim."""
draft = "[1] text.\n[107] text.\n"
result = find_numbered_notes(draft)
assert result.definitions[0].original_prefix == "[1] "
assert result.definitions[1].original_prefix == "[107] "
def test_find_numbered_notes_handles_anti_communist_draft_format():
"""Smoke test mirroring the actual Anti-Communist Formations draft:
notes follow a `## Notes` heading, separated by `________________` rule,
one definition per line, sequentially numbered."""
draft = """\
## Notes
________________
[1] First reference. https://example.org/path
[2] Second reference, abbreviated.
[3] Rosen, 7.
"""
result = find_numbered_notes(draft)
assert len(result.definitions) == 3
assert result.definitions[0].text == "First reference. https://example.org/path"
assert result.definitions[2].text == "Rosen, 7."
# --- find_blockquote_notes (> N text format, from PDF-to-markdown) ----------
def test_find_blockquote_notes_single_definition():
draft = """\
Some prose.
> 1 Donald Davidson, "On the Very Idea of a Conceptual Scheme."
"""
result = find_blockquote_notes(draft)
assert len(result.definitions) == 1
assert result.definitions[0].marker == "1"
assert result.definitions[0].text == 'Donald Davidson, "On the Very Idea of a Conceptual Scheme."'
def test_find_blockquote_notes_multiple_in_order():
draft = """\
> 1 First note.
> 2 Second note.
> 3 Third note.
"""
result = find_blockquote_notes(draft)
assert len(result.definitions) == 3
assert [d.marker for d in result.definitions] == ["1", "2", "3"]
assert [d.text for d in result.definitions] == [
"First note.",
"Second note.",
"Third note.",
]
def test_find_blockquote_notes_multi_digit_markers():
draft = "> 1 one.\n> 42 forty-two.\n> 88 eighty-eight.\n"
result = find_blockquote_notes(draft)
assert [d.marker for d in result.definitions] == ["1", "42", "88"]
def test_find_blockquote_notes_returns_empty_when_no_definitions():
draft = "Just prose.\n> A regular blockquote with no number.\n"
result = find_blockquote_notes(draft)
assert result.definitions == []
def test_find_blockquote_notes_records_line_number():
draft = """\
Line zero.
Line one.
> 1 Note on line three.
"""
result = find_blockquote_notes(draft)
assert result.definitions[0].line_number == 3
def test_find_blockquote_notes_strips_trailing_whitespace():
draft = "> 1 Note text with trailing spaces. \n"
result = find_blockquote_notes(draft)
assert result.definitions[0].text == "Note text with trailing spaces."
def test_find_blockquote_notes_records_original_prefix():
"""Reassembly prefix for blockquote format is ``'> N '``."""
draft = "> 1 text.\n> 88 text.\n"
result = find_blockquote_notes(draft)
assert result.definitions[0].original_prefix == "> 1 "
assert result.definitions[1].original_prefix == "> 88 "
def test_find_blockquote_notes_ignores_non_numeric_blockquotes():
"""A blockquote without a leading number is regular prose, not a note."""
draft = """\
> This is a regular blockquote.
> 1 This is a note.
> Another regular blockquote.
"""
result = find_blockquote_notes(draft)
assert len(result.definitions) == 1
assert result.definitions[0].marker == "1"