diff --git a/src/cmos/cli.py b/src/cmos/cli.py index e03dfcb..99880c2 100644 --- a/src/cmos/cli.py +++ b/src/cmos/cli.py @@ -23,7 +23,12 @@ from typing import Callable from cmos.formatter import format_bibliography_entry from cmos.note_formatter import format_note_entry -from cmos.parser import find_notes, find_numbered_notes, split_bibliography +from cmos.parser import ( + find_blockquote_notes, + find_notes, + find_numbered_notes, + split_bibliography, +) FormatterFn = Callable[[str], str] @@ -61,7 +66,7 @@ def reformat_draft( pieces: list[str] = [] if parsed.before: pieces.append(parsed.before) - pieces.append("## Bibliography") + pieces.append(parsed.heading) pieces.append("") pieces.extend(rewritten) if parsed.after: @@ -77,13 +82,14 @@ def reformat_notes( ) -> str: """Rewrite footnote definitions in ``text`` to CMOS 18 note form. - Auto-detects the source format by trying ``cmos.parser.find_notes`` - (pandoc-style ``[^marker]: text``) first and falling back to - ``cmos.parser.find_numbered_notes`` (``[N] text``) if no pandoc - definitions were found. Both formats are common in real drafts: - pandoc is used by writers who author directly in markdown, while - the numbered form is what docx-to-text conversion typically - produces from footnoted Word documents. + Auto-detects the source format by trying three parsers in order: + ``find_notes`` (pandoc-style ``[^marker]: text``), then + ``find_numbered_notes`` (``[N] text``), then + ``find_blockquote_notes`` (``> N text``). All three formats are + common in real drafts: pandoc is used by writers who author + directly in markdown, the numbered form comes from docx-to-text + conversion, and the blockquote form comes from PDF-to-markdown + conversion. Each definition's body is formatted via ``formatter`` (the v2 note formatter by default) and substituted back into the same @@ -115,6 +121,8 @@ def reformat_notes( parsed = find_notes(text) if not parsed.definitions: parsed = find_numbered_notes(text) + if not parsed.definitions: + parsed = find_blockquote_notes(text) if not parsed.definitions: return text diff --git a/src/cmos/parser.py b/src/cmos/parser.py index 45d45bc..aaf1f4b 100644 --- a/src/cmos/parser.py +++ b/src/cmos/parser.py @@ -3,33 +3,38 @@ v1 scope: find a ``## Bibliography`` heading (case-insensitive), treat the section as running until the next level-2 (``##``) heading or end of file, and return each non-blank line as one entry. Bullet markers (``- ``, ``* ``) -at the start of a line are stripped. +at the start of a line are stripped. The heading may also be +``## Works Cited`` or ``## References`` (case-insensitive), and may be +wrapped in bold markers (``## **Bibliography**``). Numbered sub-headings +within the bibliography (e.g., ``## **1. Primary Sources**``) are skipped +rather than treated as section terminators. -v2 scope (notes): two complementary functions for finding footnote -definitions in two distinct source formats. Both return the same +v2 scope (notes): three complementary functions for finding footnote +definitions in distinct source formats. All return the same ``NotesParseResult`` shape; the consumer (typically the CLI) can call -one and fall back to the other to auto-detect format. +them in fallback order to auto-detect format. - ``find_notes`` finds pandoc-style markdown footnote definitions ``[^marker]: text``. The marker can be numeric or named. - ``find_numbered_notes`` finds plain ``[N] text`` definitions where the marker is digit-only and there is no caret or colon — the format produced by docx-to-text conversion of footnoted Word docs. +- ``find_blockquote_notes`` finds blockquote-style ``> N text`` + definitions — the format produced by PDF-to-markdown conversion. Each ``NoteDefinition`` carries the marker, the body text (with surrounding whitespace stripped), the line number where it appeared (0-indexed, for future reassembly), and the literal ``original_prefix`` -string (e.g., ``"[^1]: "`` or ``"[1] "``) so that reassembly can -round-trip the source's marker syntax without the consumer needing to -know which format was matched. +string (e.g., ``"[^1]: "``, ``"[1] "``, or ``"> 1 "``) so that +reassembly can round-trip the source's marker syntax without the +consumer needing to know which format was matched. -Single-line definitions only in both formats — multi-line continuation +Single-line definitions only in all formats — multi-line continuation (indented continuation lines) is deferred. The reference markers in prose are NOT collected; only the definitions need reformatting. -Out of scope in v1: in-prose citation rewriting, multi-line entries, -nested sections, alternative heading names (e.g., "Works Cited", -"References"). Out of scope in v2 (so far): multi-line note definitions, +Out of scope in v1: in-prose citation rewriting, multi-line entries. +Out of scope in v2 (so far): multi-line note definitions, shortened-form generation. """ @@ -48,6 +53,7 @@ class ParseResult: before: str entries: list[str] after: str + heading: str = "## Bibliography" @dataclass @@ -77,11 +83,43 @@ class NotesParseResult: definitions: list[NoteDefinition] -_HEADING_RE = re.compile(r"^##\s+bibliography\s*$", re.IGNORECASE) +_HEADING_RE = re.compile( + r"^##\s+\*{0,2}(?:bibliography|works\s+cited|references)\*{0,2}\s*$", + re.IGNORECASE, +) _LEVEL_TWO_RE = re.compile(r"^##\s+\S") +_NUMBERED_SUB_HEADING_RE = re.compile( + r"^##\s+[_*]*\d", re.IGNORECASE +) _BULLET_RE = re.compile(r"^[-*]\s+") +_BARE_PAGE_RE = re.compile(r"^\d+$") +_DOWNLOAD_BANNER_RE = re.compile(r"^Downloaded from ", re.IGNORECASE) +_CC_LICENSE_RE = re.compile(r"creativecommons\.org") +_CITATION_MARKERS_RE = re.compile(r"""[,*_"(]|://""") _NOTE_DEF_RE = re.compile(r"^\[\^([^\]]+)\]:\s*(.*?)\s*$") _NUMBERED_NOTE_RE = re.compile(r"^\[(\d+)\]\s+(.*?)\s*$") +_BLOCKQUOTE_NOTE_RE = re.compile(r"^>\s*(\d+)\s+(.*?)\s*$") + + +def _is_pdf_junk(line: str) -> bool: + """Return True if ``line`` looks like PDF-to-markdown noise rather + than a bibliography entry. + + Catches: bare page numbers, blockquote footnotes, publisher download + banners, Creative Commons license URLs, and short running headers + (under 30 chars with no citation-like punctuation). + """ + if _BARE_PAGE_RE.match(line): + return True + if _BLOCKQUOTE_NOTE_RE.match(line): + return True + if _DOWNLOAD_BANNER_RE.match(line): + return True + if _CC_LICENSE_RE.search(line): + return True + if len(line) < 30 and not _CITATION_MARKERS_RE.search(line): + return True + return False def split_bibliography(text: str) -> ParseResult: @@ -96,14 +134,18 @@ def split_bibliography(text: str) -> ParseResult: if heading_idx is None: raise NoBibliographyError("no '## Bibliography' heading found") - # Find the end of the section: next level-two heading after the bibliography, - # or end of file. + # Find the end of the section: next level-two heading that is NOT a + # numbered sub-heading (e.g., "## 1. Primary Sources"). Numbered + # sub-headings are part of the bibliography and should be skipped. section_end = len(lines) for j in range(heading_idx + 1, len(lines)): - if _LEVEL_TWO_RE.match(lines[j]): + if _LEVEL_TWO_RE.match(lines[j]) and not _NUMBERED_SUB_HEADING_RE.match( + lines[j] + ): section_end = j break + heading = lines[heading_idx].rstrip() before = "\n".join(lines[:heading_idx]) after_lines = lines[section_end:] after = "\n".join(after_lines) @@ -113,10 +155,17 @@ def split_bibliography(text: str) -> ParseResult: stripped = raw.strip() if not stripped: continue + # Skip numbered sub-headings — they subdivide the bibliography + # but are not entries themselves. + if _NUMBERED_SUB_HEADING_RE.match(raw): + continue stripped = _BULLET_RE.sub("", stripped) + # Filter PDF-to-markdown junk. + if _is_pdf_junk(stripped): + continue entries.append(stripped) - return ParseResult(before=before, entries=entries, after=after) + return ParseResult(before=before, entries=entries, after=after, heading=heading) def find_notes(text: str) -> NotesParseResult: @@ -194,3 +243,34 @@ def find_numbered_notes(text: str) -> NotesParseResult: ) ) return NotesParseResult(definitions=definitions) + + +def find_blockquote_notes(text: str) -> NotesParseResult: + """Find blockquote-style footnote definitions: ``> N text``. + + Targets the format produced by PDF-to-markdown conversion where + footnotes appear as blockquoted lines beginning with a bare number. + The marker must be all digits — a blockquote line without a leading + number is regular prose and is not matched. + + Returns a ``NotesParseResult`` with definitions in source order. + Each definition's ``original_prefix`` is set to ``"> N "`` so that + downstream reassembly can write back the formatted body using the + same marker syntax. + """ + definitions: list[NoteDefinition] = [] + for line_number, line in enumerate(text.splitlines()): + match = _BLOCKQUOTE_NOTE_RE.match(line) + if match is None: + continue + marker = match.group(1) + body = match.group(2) + definitions.append( + NoteDefinition( + marker=marker, + text=body, + line_number=line_number, + original_prefix=f"> {marker} ", + ) + ) + return NotesParseResult(definitions=definitions) diff --git a/tests/test_cli.py b/tests/test_cli.py index 09712d9..5dda935 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -51,14 +51,14 @@ def test_reformat_draft_preserves_sections_after_bibliography(): draft = """\ ## Bibliography -one entry. +Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020). ## Appendix appendix text. """ output = reformat_draft(draft, formatter=_fake_formatter) - assert "FORMATTED(one entry.)" in output + assert "FORMATTED(Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).)" in output assert "## Appendix" in output assert "appendix text." in output @@ -69,11 +69,27 @@ def test_reformat_draft_raises_when_no_bibliography(): def test_reformat_draft_bibliography_heading_preserved(): - draft = "## Bibliography\n\nentry.\n" + draft = '## Bibliography\n\nYu, Charles. *Interior Chinatown* (Pantheon, 2020).\n' output = reformat_draft(draft, formatter=_fake_formatter) assert "## Bibliography" in output +def test_reformat_draft_preserves_works_cited_heading(): + """When the input heading is ``## Works Cited``, the output must + keep that heading — not replace it with ``## Bibliography``.""" + draft = '## Works Cited\n\nDavidson, Donald. "On the Very Idea." (1974).\n' + output = reformat_draft(draft, formatter=_fake_formatter) + assert "## Works Cited" in output + assert "## Bibliography" not in output + + +def test_reformat_draft_preserves_bold_wrapped_heading(): + """Bold markers in the heading should be preserved in the output.""" + draft = '## **Works Cited**\n\nDavidson, Donald. "On the Very Idea." (1974).\n' + output = reformat_draft(draft, formatter=_fake_formatter) + assert "## **Works Cited**" in output + + def test_python_dash_m_invocation_actually_runs_main(): """Regression test: `python -m cmos.cli` must actually invoke main(). @@ -216,6 +232,23 @@ def test_reformat_notes_auto_detects_numbered_format(): assert "[^2]" not in output +def test_reformat_notes_auto_detects_blockquote_format(): + """PDF-to-markdown drafts use ``> N text`` blockquote footnotes. + reformat_notes must auto-detect after pandoc and numbered fail.""" + draft = """\ +Some prose. + +> 1 Davidson, "On the Very Idea of a Conceptual Scheme." +> 2 Frankenberry 1999, 526. +""" + output = reformat_notes(draft, formatter=_fake_note_formatter) + assert '> 1 NOTE_FORMATTED(Davidson, "On the Very Idea of a Conceptual Scheme.")' in output + assert "> 2 NOTE_FORMATTED(Frankenberry 1999, 526.)" in output + # No pandoc or numbered markers should appear. + assert "[^1]" not in output + assert "[1] " not in output + + def test_reformat_notes_pandoc_input_still_round_trips_as_pandoc(): """Regression: after the auto-detect change, pandoc-format input must still produce pandoc-format output. The original_prefix path @@ -265,28 +298,28 @@ def test_reformat_draft_preserves_order_under_concurrency(): # later ones if order were naively tied to completion. import time + # Map each entry to a sleep duration so earlier entries finish last. + order = {"2020": 4, "2021": 3, "2022": 2, "2023": 1, "2024": 0} + def slow_fake(messy: str) -> str: - # Entries with lower index sleep longer so they finish last. - n = int(messy.split()[-1]) - time.sleep(0.05 * (5 - n)) + year = messy.strip()[-5:-1] # extract "2020" etc. + time.sleep(0.05 * order.get(year, 0)) return f"FORMATTED({messy})" draft = """\ ## Bibliography -entry 0 -entry 1 -entry 2 -entry 3 -entry 4 +Author, A. *Title Zero* (Publisher, 2020). +Author, B. *Title One* (Publisher, 2021). +Author, C. *Title Two* (Publisher, 2022). +Author, D. *Title Three* (Publisher, 2023). +Author, E. *Title Four* (Publisher, 2024). """ output = reformat_draft(draft, formatter=slow_fake, concurrency=4) # Entries must appear in input order. lines = [l for l in output.splitlines() if l.startswith("FORMATTED(")] - assert lines == [ - "FORMATTED(entry 0)", - "FORMATTED(entry 1)", - "FORMATTED(entry 2)", - "FORMATTED(entry 3)", - "FORMATTED(entry 4)", - ] + assert lines[0].startswith("FORMATTED(Author, A.") + assert lines[1].startswith("FORMATTED(Author, B.") + assert lines[2].startswith("FORMATTED(Author, C.") + assert lines[3].startswith("FORMATTED(Author, D.") + assert lines[4].startswith("FORMATTED(Author, E.") diff --git a/tests/test_parser.py b/tests/test_parser.py index c00df2d..d450004 100644 --- a/tests/test_parser.py +++ b/tests/test_parser.py @@ -26,6 +26,7 @@ import pytest from cmos.parser import ( NoBibliographyError, + find_blockquote_notes, find_notes, find_numbered_notes, split_bibliography, @@ -57,15 +58,16 @@ def test_section_ends_at_next_level_two_heading(): draft = """\ ## Bibliography -First entry. -Second entry. +Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020). +Kwon, Hyeyoung. "Inclusion Work," *American Journal of Sociology* 127 (2022). ## Appendix Appendix content here. """ result = split_bibliography(draft) - assert result.entries == ["First entry.", "Second entry."] + assert len(result.entries) == 2 + assert result.entries[0].startswith("Yu, Charles") assert "## Appendix" in result.after assert "Appendix content here." in result.after @@ -74,22 +76,208 @@ def test_bullet_prefixes_stripped(): draft = """\ ## Bibliography -- First entry. -- Second entry. -* Third entry. +- Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020). +- Kwon, Hyeyoung. "Inclusion Work," *AJS* 127 (2022). +* Google. "Privacy Policy," Privacy & Terms, 2023. """ result = split_bibliography(draft) - assert result.entries == ["First entry.", "Second entry.", "Third entry."] + assert len(result.entries) == 3 + assert result.entries[0].startswith("Yu, Charles") + assert result.entries[1].startswith("Kwon, Hyeyoung") + assert result.entries[2].startswith("Google") def test_heading_match_is_case_insensitive(): draft = """\ ## bibliography -Only entry. +Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020). """ result = split_bibliography(draft) - assert result.entries == ["Only entry."] + assert len(result.entries) == 1 + assert result.entries[0].startswith("Yu, Charles") + + +def test_heading_match_works_cited(): + """'Works Cited' is the standard CMOS heading alternative to + 'Bibliography'. The parser should accept it.""" + draft = """\ +## Works Cited + +Davidson, Donald. "On the Very Idea of a Conceptual Scheme." +Evans-Pritchard, E.E. *Witchcraft, oracles, and magic.* +""" + result = split_bibliography(draft) + assert len(result.entries) == 2 + assert result.entries[0].startswith("Davidson") + + +def test_heading_match_works_cited_bold(): + """Bold-wrapped variant: ``## **Works Cited**``.""" + draft = """\ +## **Works Cited** + +- Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999). +""" + result = split_bibliography(draft) + assert len(result.entries) == 1 + assert result.entries[0].startswith("Frankenberry") + + +def test_heading_match_ignores_bold_markers(): + """Real-world drafts from PDF-to-markdown conversion often wrap the + heading text in bold: ``## **BIBLIOGRAPHY**``. The parser should + strip ``**`` before matching.""" + draft = """\ +## **BIBLIOGRAPHY** + +- Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020). +- Kwon, Hyeyoung. "Inclusion Work," *AJS* 127 (2022). +""" + result = split_bibliography(draft) + assert len(result.entries) == 2 + assert result.entries[0].startswith("Yu, Charles") + + +def test_heading_match_ignores_numbered_sub_headings(): + """Subdivided bibliographies (common in dissertations) use numbered + ``##`` sub-headings inside the bibliography section. These must NOT + terminate the section — the parser should skip them and keep + collecting entries until a non-numbered ``##`` heading or EOF.""" + draft = """\ +## **BIBLIOGRAPHY** + +## **1. Primary Sources** + +- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939). +- Till, Walter C., ed., *Die koptischen Ostraka* (Vienna, 1960). + +## **2. Secondary Literature** + +- Marsham, Andrew, ed., *The Umayyad World* (London: Routledge, 2020). +- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951). +""" + result = split_bibliography(draft) + assert len(result.entries) == 4 + assert result.entries[0].startswith("Crum, Walter") + assert result.entries[3].startswith("Schulz, Fritz") + assert result.after == "" + + +def test_numbered_sub_headings_not_collected_as_entries(): + """The sub-heading lines themselves (e.g. ``## **1. Primary Sources**``) + should not appear in the entries list.""" + draft = """\ +## Bibliography + +## 1. Primary Sources + +- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939). + +## 2. Secondary Sources + +- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951). +""" + result = split_bibliography(draft) + assert len(result.entries) == 2 + for entry in result.entries: + assert not entry.startswith("##") + assert "Primary Sources" not in entry + assert "Secondary Sources" not in entry + + +def test_non_numbered_heading_still_ends_section_after_sub_headings(): + """A non-numbered ``##`` heading after bibliography sub-sections + should still terminate the bibliography section.""" + draft = """\ +## Bibliography + +## 1. Primary Sources + +- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939). + +## Appendix + +Appendix content. +""" + result = split_bibliography(draft) + assert len(result.entries) == 1 + assert result.entries[0].startswith("Crum, Walter") + assert "## Appendix" in result.after + assert "Appendix content." in result.after + + +def test_bare_page_numbers_filtered(): + """PDF-to-markdown conversion leaves bare page numbers (e.g. '370') + as standalone lines. These are not bibliography entries.""" + draft = """\ +## Bibliography + +- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939). + +370 + +- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951). + +371 +""" + result = split_bibliography(draft) + assert len(result.entries) == 2 + assert result.entries[0].startswith("Crum") + assert result.entries[1].startswith("Schulz") + + +def test_blockquote_footnotes_filtered(): + """Blockquote footnotes (``> N text``) that appear inside a Works Cited + section are PDF artifacts, not bibliography entries.""" + draft = """\ +## Works Cited + +Davidson, Donald. "On the Very Idea of a Conceptual Scheme." + +> 87 Holbraad 2010. +> 88 Frankenberry 2018, 236. + +Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999). +""" + result = split_bibliography(draft) + assert len(result.entries) == 2 + assert result.entries[0].startswith("Davidson") + assert result.entries[1].startswith("Frankenberry") + + +def test_download_banners_filtered(): + """Brill/publisher download banners are PDF junk, not entries.""" + draft = """\ +## Bibliography + +Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939). +Downloaded from Brill.com 02/22/2024 05:33:29AM via Open Access. This is junk. +https://creativecommons.org/licenses/by/4.0/ +Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951). +""" + result = split_bibliography(draft) + assert len(result.entries) == 2 + assert result.entries[0].startswith("Crum") + assert result.entries[1].startswith("Schulz") + + +def test_running_headers_filtered(): + """Short lines that are running headers or page footers (e.g. an + author surname or abbreviated title) are not entries.""" + draft = """\ +## Bibliography + +Davidson, Donald. "On the Very Idea of a Conceptual Scheme." +hedrick +only words apart? +217 +Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999). +""" + result = split_bibliography(draft) + assert len(result.entries) == 2 + assert result.entries[0].startswith("Davidson") + assert result.entries[1].startswith("Frankenberry") def test_no_bibliography_raises(): @@ -292,3 +480,83 @@ ________________ assert len(result.definitions) == 3 assert result.definitions[0].text == "First reference. https://example.org/path" assert result.definitions[2].text == "Rosen, 7." + + +# --- find_blockquote_notes (> N text format, from PDF-to-markdown) ---------- + + +def test_find_blockquote_notes_single_definition(): + draft = """\ +Some prose. + +> 1 Donald Davidson, "On the Very Idea of a Conceptual Scheme." +""" + result = find_blockquote_notes(draft) + assert len(result.definitions) == 1 + assert result.definitions[0].marker == "1" + assert result.definitions[0].text == 'Donald Davidson, "On the Very Idea of a Conceptual Scheme."' + + +def test_find_blockquote_notes_multiple_in_order(): + draft = """\ +> 1 First note. +> 2 Second note. +> 3 Third note. +""" + result = find_blockquote_notes(draft) + assert len(result.definitions) == 3 + assert [d.marker for d in result.definitions] == ["1", "2", "3"] + assert [d.text for d in result.definitions] == [ + "First note.", + "Second note.", + "Third note.", + ] + + +def test_find_blockquote_notes_multi_digit_markers(): + draft = "> 1 one.\n> 42 forty-two.\n> 88 eighty-eight.\n" + result = find_blockquote_notes(draft) + assert [d.marker for d in result.definitions] == ["1", "42", "88"] + + +def test_find_blockquote_notes_returns_empty_when_no_definitions(): + draft = "Just prose.\n> A regular blockquote with no number.\n" + result = find_blockquote_notes(draft) + assert result.definitions == [] + + +def test_find_blockquote_notes_records_line_number(): + draft = """\ +Line zero. +Line one. + +> 1 Note on line three. +""" + result = find_blockquote_notes(draft) + assert result.definitions[0].line_number == 3 + + +def test_find_blockquote_notes_strips_trailing_whitespace(): + draft = "> 1 Note text with trailing spaces. \n" + result = find_blockquote_notes(draft) + assert result.definitions[0].text == "Note text with trailing spaces." + + +def test_find_blockquote_notes_records_original_prefix(): + """Reassembly prefix for blockquote format is ``'> N '``.""" + draft = "> 1 text.\n> 88 text.\n" + result = find_blockquote_notes(draft) + assert result.definitions[0].original_prefix == "> 1 " + assert result.definitions[1].original_prefix == "> 88 " + + +def test_find_blockquote_notes_ignores_non_numeric_blockquotes(): + """A blockquote without a leading number is regular prose, not a note.""" + draft = """\ +> This is a regular blockquote. +> 1 This is a note. +> Another regular blockquote. +""" + result = find_blockquote_notes(draft) + assert len(result.definitions) == 1 + assert result.definitions[0].marker == "1"