parser: support diverse md inputs (bold headings, Works Cited, blockquote notes, junk filtering)

- Bibliography heading now accepts bold markers (## **BIBLIOGRAPHY**),
  alternative names (Works Cited, References), and numbered sub-headings
  within the section.
- Original heading text preserved in output instead of hardcoded
  "## Bibliography".
- New find_blockquote_notes() parser for PDF-to-markdown footnote format
  (> N text), wired into CLI as third fallback after pandoc and numbered.
- PDF junk filtered from bibliography entries: bare page numbers,
  blockquote footnotes, download banners, CC license URLs, and short
  running headers.
This commit is contained in:
Mark Eaton
2026-04-12 00:11:52 -04:00
parent c21ca7b58e
commit cad936f576
4 changed files with 441 additions and 52 deletions
+51 -18
View File
@@ -51,14 +51,14 @@ def test_reformat_draft_preserves_sections_after_bibliography():
draft = """\
## Bibliography
one entry.
Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
## Appendix
appendix text.
"""
output = reformat_draft(draft, formatter=_fake_formatter)
assert "FORMATTED(one entry.)" in output
assert "FORMATTED(Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).)" in output
assert "## Appendix" in output
assert "appendix text." in output
@@ -69,11 +69,27 @@ def test_reformat_draft_raises_when_no_bibliography():
def test_reformat_draft_bibliography_heading_preserved():
draft = "## Bibliography\n\nentry.\n"
draft = '## Bibliography\n\nYu, Charles. *Interior Chinatown* (Pantheon, 2020).\n'
output = reformat_draft(draft, formatter=_fake_formatter)
assert "## Bibliography" in output
def test_reformat_draft_preserves_works_cited_heading():
"""When the input heading is ``## Works Cited``, the output must
keep that heading — not replace it with ``## Bibliography``."""
draft = '## Works Cited\n\nDavidson, Donald. "On the Very Idea." (1974).\n'
output = reformat_draft(draft, formatter=_fake_formatter)
assert "## Works Cited" in output
assert "## Bibliography" not in output
def test_reformat_draft_preserves_bold_wrapped_heading():
"""Bold markers in the heading should be preserved in the output."""
draft = '## **Works Cited**\n\nDavidson, Donald. "On the Very Idea." (1974).\n'
output = reformat_draft(draft, formatter=_fake_formatter)
assert "## **Works Cited**" in output
def test_python_dash_m_invocation_actually_runs_main():
"""Regression test: `python -m cmos.cli` must actually invoke main().
@@ -216,6 +232,23 @@ def test_reformat_notes_auto_detects_numbered_format():
assert "[^2]" not in output
def test_reformat_notes_auto_detects_blockquote_format():
"""PDF-to-markdown drafts use ``> N text`` blockquote footnotes.
reformat_notes must auto-detect after pandoc and numbered fail."""
draft = """\
Some prose.
> 1 Davidson, "On the Very Idea of a Conceptual Scheme."
> 2 Frankenberry 1999, 526.
"""
output = reformat_notes(draft, formatter=_fake_note_formatter)
assert '> 1 NOTE_FORMATTED(Davidson, "On the Very Idea of a Conceptual Scheme.")' in output
assert "> 2 NOTE_FORMATTED(Frankenberry 1999, 526.)" in output
# No pandoc or numbered markers should appear.
assert "[^1]" not in output
assert "[1] " not in output
def test_reformat_notes_pandoc_input_still_round_trips_as_pandoc():
"""Regression: after the auto-detect change, pandoc-format input
must still produce pandoc-format output. The original_prefix path
@@ -265,28 +298,28 @@ def test_reformat_draft_preserves_order_under_concurrency():
# later ones if order were naively tied to completion.
import time
# Map each entry to a sleep duration so earlier entries finish last.
order = {"2020": 4, "2021": 3, "2022": 2, "2023": 1, "2024": 0}
def slow_fake(messy: str) -> str:
# Entries with lower index sleep longer so they finish last.
n = int(messy.split()[-1])
time.sleep(0.05 * (5 - n))
year = messy.strip()[-5:-1] # extract "2020" etc.
time.sleep(0.05 * order.get(year, 0))
return f"FORMATTED({messy})"
draft = """\
## Bibliography
entry 0
entry 1
entry 2
entry 3
entry 4
Author, A. *Title Zero* (Publisher, 2020).
Author, B. *Title One* (Publisher, 2021).
Author, C. *Title Two* (Publisher, 2022).
Author, D. *Title Three* (Publisher, 2023).
Author, E. *Title Four* (Publisher, 2024).
"""
output = reformat_draft(draft, formatter=slow_fake, concurrency=4)
# Entries must appear in input order.
lines = [l for l in output.splitlines() if l.startswith("FORMATTED(")]
assert lines == [
"FORMATTED(entry 0)",
"FORMATTED(entry 1)",
"FORMATTED(entry 2)",
"FORMATTED(entry 3)",
"FORMATTED(entry 4)",
]
assert lines[0].startswith("FORMATTED(Author, A.")
assert lines[1].startswith("FORMATTED(Author, B.")
assert lines[2].startswith("FORMATTED(Author, C.")
assert lines[3].startswith("FORMATTED(Author, D.")
assert lines[4].startswith("FORMATTED(Author, E.")
+277 -9
View File
@@ -26,6 +26,7 @@ import pytest
from cmos.parser import (
NoBibliographyError,
find_blockquote_notes,
find_notes,
find_numbered_notes,
split_bibliography,
@@ -57,15 +58,16 @@ def test_section_ends_at_next_level_two_heading():
draft = """\
## Bibliography
First entry.
Second entry.
Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
Kwon, Hyeyoung. "Inclusion Work," *American Journal of Sociology* 127 (2022).
## Appendix
Appendix content here.
"""
result = split_bibliography(draft)
assert result.entries == ["First entry.", "Second entry."]
assert len(result.entries) == 2
assert result.entries[0].startswith("Yu, Charles")
assert "## Appendix" in result.after
assert "Appendix content here." in result.after
@@ -74,22 +76,208 @@ def test_bullet_prefixes_stripped():
draft = """\
## Bibliography
- First entry.
- Second entry.
* Third entry.
- Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
- Kwon, Hyeyoung. "Inclusion Work," *AJS* 127 (2022).
* Google. "Privacy Policy," Privacy & Terms, 2023.
"""
result = split_bibliography(draft)
assert result.entries == ["First entry.", "Second entry.", "Third entry."]
assert len(result.entries) == 3
assert result.entries[0].startswith("Yu, Charles")
assert result.entries[1].startswith("Kwon, Hyeyoung")
assert result.entries[2].startswith("Google")
def test_heading_match_is_case_insensitive():
draft = """\
## bibliography
Only entry.
Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
"""
result = split_bibliography(draft)
assert result.entries == ["Only entry."]
assert len(result.entries) == 1
assert result.entries[0].startswith("Yu, Charles")
def test_heading_match_works_cited():
"""'Works Cited' is the standard CMOS heading alternative to
'Bibliography'. The parser should accept it."""
draft = """\
## Works Cited
Davidson, Donald. "On the Very Idea of a Conceptual Scheme."
Evans-Pritchard, E.E. *Witchcraft, oracles, and magic.*
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
assert result.entries[0].startswith("Davidson")
def test_heading_match_works_cited_bold():
"""Bold-wrapped variant: ``## **Works Cited**``."""
draft = """\
## **Works Cited**
- Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999).
"""
result = split_bibliography(draft)
assert len(result.entries) == 1
assert result.entries[0].startswith("Frankenberry")
def test_heading_match_ignores_bold_markers():
"""Real-world drafts from PDF-to-markdown conversion often wrap the
heading text in bold: ``## **BIBLIOGRAPHY**``. The parser should
strip ``**`` before matching."""
draft = """\
## **BIBLIOGRAPHY**
- Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
- Kwon, Hyeyoung. "Inclusion Work," *AJS* 127 (2022).
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
assert result.entries[0].startswith("Yu, Charles")
def test_heading_match_ignores_numbered_sub_headings():
"""Subdivided bibliographies (common in dissertations) use numbered
``##`` sub-headings inside the bibliography section. These must NOT
terminate the section — the parser should skip them and keep
collecting entries until a non-numbered ``##`` heading or EOF."""
draft = """\
## **BIBLIOGRAPHY**
## **1. Primary Sources**
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
- Till, Walter C., ed., *Die koptischen Ostraka* (Vienna, 1960).
## **2. Secondary Literature**
- Marsham, Andrew, ed., *The Umayyad World* (London: Routledge, 2020).
- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
"""
result = split_bibliography(draft)
assert len(result.entries) == 4
assert result.entries[0].startswith("Crum, Walter")
assert result.entries[3].startswith("Schulz, Fritz")
assert result.after == ""
def test_numbered_sub_headings_not_collected_as_entries():
"""The sub-heading lines themselves (e.g. ``## **1. Primary Sources**``)
should not appear in the entries list."""
draft = """\
## Bibliography
## 1. Primary Sources
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
## 2. Secondary Sources
- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
for entry in result.entries:
assert not entry.startswith("##")
assert "Primary Sources" not in entry
assert "Secondary Sources" not in entry
def test_non_numbered_heading_still_ends_section_after_sub_headings():
"""A non-numbered ``##`` heading after bibliography sub-sections
should still terminate the bibliography section."""
draft = """\
## Bibliography
## 1. Primary Sources
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
## Appendix
Appendix content.
"""
result = split_bibliography(draft)
assert len(result.entries) == 1
assert result.entries[0].startswith("Crum, Walter")
assert "## Appendix" in result.after
assert "Appendix content." in result.after
def test_bare_page_numbers_filtered():
"""PDF-to-markdown conversion leaves bare page numbers (e.g. '370')
as standalone lines. These are not bibliography entries."""
draft = """\
## Bibliography
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
370
- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
371
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
assert result.entries[0].startswith("Crum")
assert result.entries[1].startswith("Schulz")
def test_blockquote_footnotes_filtered():
"""Blockquote footnotes (``> N text``) that appear inside a Works Cited
section are PDF artifacts, not bibliography entries."""
draft = """\
## Works Cited
Davidson, Donald. "On the Very Idea of a Conceptual Scheme."
> 87 Holbraad 2010.
> 88 Frankenberry 2018, 236.
Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999).
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
assert result.entries[0].startswith("Davidson")
assert result.entries[1].startswith("Frankenberry")
def test_download_banners_filtered():
"""Brill/publisher download banners are PDF junk, not entries."""
draft = """\
## Bibliography
Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
Downloaded from Brill.com 02/22/2024 05:33:29AM via Open Access. This is junk.
https://creativecommons.org/licenses/by/4.0/
Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
assert result.entries[0].startswith("Crum")
assert result.entries[1].startswith("Schulz")
def test_running_headers_filtered():
"""Short lines that are running headers or page footers (e.g. an
author surname or abbreviated title) are not entries."""
draft = """\
## Bibliography
Davidson, Donald. "On the Very Idea of a Conceptual Scheme."
hedrick
only words apart?
217
Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999).
"""
result = split_bibliography(draft)
assert len(result.entries) == 2
assert result.entries[0].startswith("Davidson")
assert result.entries[1].startswith("Frankenberry")
def test_no_bibliography_raises():
@@ -292,3 +480,83 @@ ________________
assert len(result.definitions) == 3
assert result.definitions[0].text == "First reference. https://example.org/path"
assert result.definitions[2].text == "Rosen, 7."
# --- find_blockquote_notes (> N text format, from PDF-to-markdown) ----------
def test_find_blockquote_notes_single_definition():
draft = """\
Some prose.
> 1 Donald Davidson, "On the Very Idea of a Conceptual Scheme."
"""
result = find_blockquote_notes(draft)
assert len(result.definitions) == 1
assert result.definitions[0].marker == "1"
assert result.definitions[0].text == 'Donald Davidson, "On the Very Idea of a Conceptual Scheme."'
def test_find_blockquote_notes_multiple_in_order():
draft = """\
> 1 First note.
> 2 Second note.
> 3 Third note.
"""
result = find_blockquote_notes(draft)
assert len(result.definitions) == 3
assert [d.marker for d in result.definitions] == ["1", "2", "3"]
assert [d.text for d in result.definitions] == [
"First note.",
"Second note.",
"Third note.",
]
def test_find_blockquote_notes_multi_digit_markers():
draft = "> 1 one.\n> 42 forty-two.\n> 88 eighty-eight.\n"
result = find_blockquote_notes(draft)
assert [d.marker for d in result.definitions] == ["1", "42", "88"]
def test_find_blockquote_notes_returns_empty_when_no_definitions():
draft = "Just prose.\n> A regular blockquote with no number.\n"
result = find_blockquote_notes(draft)
assert result.definitions == []
def test_find_blockquote_notes_records_line_number():
draft = """\
Line zero.
Line one.
> 1 Note on line three.
"""
result = find_blockquote_notes(draft)
assert result.definitions[0].line_number == 3
def test_find_blockquote_notes_strips_trailing_whitespace():
draft = "> 1 Note text with trailing spaces. \n"
result = find_blockquote_notes(draft)
assert result.definitions[0].text == "Note text with trailing spaces."
def test_find_blockquote_notes_records_original_prefix():
"""Reassembly prefix for blockquote format is ``'> N '``."""
draft = "> 1 text.\n> 88 text.\n"
result = find_blockquote_notes(draft)
assert result.definitions[0].original_prefix == "> 1 "
assert result.definitions[1].original_prefix == "> 88 "
def test_find_blockquote_notes_ignores_non_numeric_blockquotes():
"""A blockquote without a leading number is regular prose, not a note."""
draft = """\
> This is a regular blockquote.
> 1 This is a note.
> Another regular blockquote.
"""
result = find_blockquote_notes(draft)
assert len(result.definitions) == 1
assert result.definitions[0].marker == "1"