"""Tests for src/cmos/parser.py — extract the bibliography section from a draft. v1 scope: locate a ``## Bibliography`` heading (case-insensitive) and return the prose before it, the list of entries inside it, and the prose after it. The section ends at the next ``##`` (level-2) heading or end of file. Blank lines and bullet markers (``- `` or ``* ``) are stripped from entries. v2 (notes) parser scope: two complementary functions for finding footnote definitions in two distinct source formats: - ``find_notes`` finds pandoc-style ``[^marker]: text`` definitions (caret-prefixed marker, colon after the closing bracket). - ``find_numbered_notes`` finds ``[N] text`` definitions (square bracket numeric marker, no caret, no colon, just a space before the body) — the format produced by docx-to-text conversion of footnoted Word docs. Both functions return ``NotesParseResult`` with a list of ``NoteDefinition`` records carrying the marker, body text, line number, and the original prefix string (so reassembly can round-trip the source's marker syntax). Single-line definitions only in both formats; multi-line continuation is deferred. """ import pytest from cmos.parser import ( NoBibliographyError, find_blockquote_notes, find_notes, find_numbered_notes, split_bibliography, ) def test_three_entries_under_bibliography_heading(): draft = """\ Some prose here. More prose. ## Bibliography yu, charles. interior chinatown. New York: Pantheon Books, 2020. Kwon, Hyeyoung. "inclusion work." American Journal of Sociology 127, no. 6 (2022): 1818-1859. google, "privacy policy," privacy & terms, nov 15 2023, policies.google.com/privacy """ result = split_bibliography(draft) assert len(result.entries) == 3 assert result.entries[0].startswith("yu, charles") assert result.entries[1].startswith("Kwon, Hyeyoung") assert result.entries[2].startswith("google") assert "Some prose here." in result.before assert result.after == "" def test_section_ends_at_next_level_two_heading(): draft = """\ ## Bibliography Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020). Kwon, Hyeyoung. "Inclusion Work," *American Journal of Sociology* 127 (2022). ## Appendix Appendix content here. """ result = split_bibliography(draft) assert len(result.entries) == 2 assert result.entries[0].startswith("Yu, Charles") assert "## Appendix" in result.after assert "Appendix content here." in result.after def test_bullet_prefixes_stripped(): draft = """\ ## Bibliography - Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020). - Kwon, Hyeyoung. "Inclusion Work," *AJS* 127 (2022). * Google. "Privacy Policy," Privacy & Terms, 2023. """ result = split_bibliography(draft) assert len(result.entries) == 3 assert result.entries[0].startswith("Yu, Charles") assert result.entries[1].startswith("Kwon, Hyeyoung") assert result.entries[2].startswith("Google") def test_heading_match_is_case_insensitive(): draft = """\ ## bibliography Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020). """ result = split_bibliography(draft) assert len(result.entries) == 1 assert result.entries[0].startswith("Yu, Charles") def test_heading_match_works_cited(): """'Works Cited' is the standard CMOS heading alternative to 'Bibliography'. The parser should accept it.""" draft = """\ ## Works Cited Davidson, Donald. "On the Very Idea of a Conceptual Scheme." Evans-Pritchard, E.E. *Witchcraft, oracles, and magic.* """ result = split_bibliography(draft) assert len(result.entries) == 2 assert result.entries[0].startswith("Davidson") def test_heading_match_works_cited_bold(): """Bold-wrapped variant: ``## **Works Cited**``.""" draft = """\ ## **Works Cited** - Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999). """ result = split_bibliography(draft) assert len(result.entries) == 1 assert result.entries[0].startswith("Frankenberry") def test_heading_match_ignores_bold_markers(): """Real-world drafts from PDF-to-markdown conversion often wrap the heading text in bold: ``## **BIBLIOGRAPHY**``. The parser should strip ``**`` before matching.""" draft = """\ ## **BIBLIOGRAPHY** - Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020). - Kwon, Hyeyoung. "Inclusion Work," *AJS* 127 (2022). """ result = split_bibliography(draft) assert len(result.entries) == 2 assert result.entries[0].startswith("Yu, Charles") def test_heading_match_ignores_numbered_sub_headings(): """Subdivided bibliographies (common in dissertations) use numbered ``##`` sub-headings inside the bibliography section. These must NOT terminate the section — the parser should skip them and keep collecting entries until a non-numbered ``##`` heading or EOF.""" draft = """\ ## **BIBLIOGRAPHY** ## **1. Primary Sources** - Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939). - Till, Walter C., ed., *Die koptischen Ostraka* (Vienna, 1960). ## **2. Secondary Literature** - Marsham, Andrew, ed., *The Umayyad World* (London: Routledge, 2020). - Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951). """ result = split_bibliography(draft) assert len(result.entries) == 4 assert result.entries[0].startswith("Crum, Walter") assert result.entries[3].startswith("Schulz, Fritz") assert result.after == "" def test_numbered_sub_headings_not_collected_as_entries(): """The sub-heading lines themselves (e.g. ``## **1. Primary Sources**``) should not appear in the entries list.""" draft = """\ ## Bibliography ## 1. Primary Sources - Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939). ## 2. Secondary Sources - Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951). """ result = split_bibliography(draft) assert len(result.entries) == 2 for entry in result.entries: assert not entry.startswith("##") assert "Primary Sources" not in entry assert "Secondary Sources" not in entry def test_non_numbered_heading_still_ends_section_after_sub_headings(): """A non-numbered ``##`` heading after bibliography sub-sections should still terminate the bibliography section.""" draft = """\ ## Bibliography ## 1. Primary Sources - Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939). ## Appendix Appendix content. """ result = split_bibliography(draft) assert len(result.entries) == 1 assert result.entries[0].startswith("Crum, Walter") assert "## Appendix" in result.after assert "Appendix content." in result.after def test_bare_page_numbers_filtered(): """PDF-to-markdown conversion leaves bare page numbers (e.g. '370') as standalone lines. These are not bibliography entries.""" draft = """\ ## Bibliography - Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939). 370 - Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951). 371 """ result = split_bibliography(draft) assert len(result.entries) == 2 assert result.entries[0].startswith("Crum") assert result.entries[1].startswith("Schulz") def test_blockquote_footnotes_filtered(): """Blockquote footnotes (``> N text``) that appear inside a Works Cited section are PDF artifacts, not bibliography entries.""" draft = """\ ## Works Cited Davidson, Donald. "On the Very Idea of a Conceptual Scheme." > 87 Holbraad 2010. > 88 Frankenberry 2018, 236. Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999). """ result = split_bibliography(draft) assert len(result.entries) == 2 assert result.entries[0].startswith("Davidson") assert result.entries[1].startswith("Frankenberry") def test_download_banners_filtered(): """Brill/publisher download banners are PDF junk, not entries.""" draft = """\ ## Bibliography Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939). Downloaded from Brill.com 02/22/2024 05:33:29AM via Open Access. This is junk. https://creativecommons.org/licenses/by/4.0/ Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951). """ result = split_bibliography(draft) assert len(result.entries) == 2 assert result.entries[0].startswith("Crum") assert result.entries[1].startswith("Schulz") def test_running_headers_filtered(): """Short lines that are running headers or page footers (e.g. an author surname or abbreviated title) are not entries.""" draft = """\ ## Bibliography Davidson, Donald. "On the Very Idea of a Conceptual Scheme." hedrick only words apart? 217 Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999). """ result = split_bibliography(draft) assert len(result.entries) == 2 assert result.entries[0].startswith("Davidson") assert result.entries[1].startswith("Frankenberry") def test_no_bibliography_raises(): with pytest.raises(NoBibliographyError): split_bibliography("## Introduction\n\nJust prose.\n") # --- find_notes (v2 parser, pandoc-style markdown footnote definitions) ---- def test_find_notes_single_definition(): draft = """\ Some prose with a citation.[^1] [^1]: Charles Yu, *Interior Chinatown* (Pantheon Books, 2020), 45. """ result = find_notes(draft) assert len(result.definitions) == 1 assert result.definitions[0].marker == "1" assert result.definitions[0].text == "Charles Yu, *Interior Chinatown* (Pantheon Books, 2020), 45." def test_find_notes_multiple_definitions_in_order(): draft = """\ Prose.[^1] More prose.[^2] And again.[^3] [^1]: First note. [^2]: Second note. [^3]: Third note. """ result = find_notes(draft) assert len(result.definitions) == 3 assert [d.marker for d in result.definitions] == ["1", "2", "3"] assert [d.text for d in result.definitions] == [ "First note.", "Second note.", "Third note.", ] def test_find_notes_named_marker(): draft = """\ Citation.[^smith2020] [^smith2020]: Smith, J. (2020). Some work. """ result = find_notes(draft) assert len(result.definitions) == 1 assert result.definitions[0].marker == "smith2020" assert result.definitions[0].text == "Smith, J. (2020). Some work." def test_find_notes_returns_empty_when_no_definitions(): draft = "Just prose with no footnote definitions.\n" result = find_notes(draft) assert result.definitions == [] def test_find_notes_records_line_number(): draft = """\ Line zero. Line one. [^1]: Note on line three. """ result = find_notes(draft) assert len(result.definitions) == 1 assert result.definitions[0].line_number == 3 def test_find_notes_strips_trailing_whitespace_from_text(): draft = "[^1]: Note text with trailing spaces. \n" result = find_notes(draft) assert result.definitions[0].text == "Note text with trailing spaces." def test_find_notes_definition_can_appear_before_reference(): # Pandoc allows definitions anywhere in the doc, not just at the bottom. draft = """\ [^1]: A definition before the reference. Some prose that cites it.[^1] """ result = find_notes(draft) assert len(result.definitions) == 1 assert result.definitions[0].text == "A definition before the reference." def test_find_notes_records_original_prefix_for_pandoc_format(): """Reassembly needs the literal prefix string so the round-trip preserves the source's marker syntax. For pandoc this is `[^N]: `.""" draft = "[^1]: text.\n[^smith2020]: text.\n" result = find_notes(draft) assert result.definitions[0].original_prefix == "[^1]: " assert result.definitions[1].original_prefix == "[^smith2020]: " # --- find_numbered_notes ([N] text format, no caret, no colon) ------------- def test_find_numbered_notes_single_definition(): draft = """\ Some prose. [1] Charles Yu, Interior Chinatown. Pantheon, 2020. p. 45. """ result = find_numbered_notes(draft) assert len(result.definitions) == 1 assert result.definitions[0].marker == "1" assert result.definitions[0].text == "Charles Yu, Interior Chinatown. Pantheon, 2020. p. 45." def test_find_numbered_notes_multiple_sequential(): draft = """\ [1] First. [2] Second. [3] Third. """ result = find_numbered_notes(draft) assert len(result.definitions) == 3 assert [d.marker for d in result.definitions] == ["1", "2", "3"] assert [d.text for d in result.definitions] == ["First.", "Second.", "Third."] def test_find_numbered_notes_multi_digit_markers(): """The Anti-Communist draft has 107 notes, including [107]. Multi-digit markers must work.""" draft = "[1] one.\n[42] forty-two.\n[107] one-oh-seven.\n" result = find_numbered_notes(draft) assert [d.marker for d in result.definitions] == ["1", "42", "107"] def test_find_numbered_notes_returns_empty_when_no_definitions(): draft = "Just prose with no numbered notes.\n" result = find_numbered_notes(draft) assert result.definitions == [] def test_find_numbered_notes_records_line_number(): draft = """\ Line zero. Line one. [1] Note on line three. """ result = find_numbered_notes(draft) assert result.definitions[0].line_number == 3 def test_find_numbered_notes_strips_trailing_whitespace_from_text(): draft = "[1] Note text with trailing spaces. \n" result = find_numbered_notes(draft) assert result.definitions[0].text == "Note text with trailing spaces." def test_find_numbered_notes_ignores_pandoc_format(): """`[^1]: text` is pandoc format, not numbered. find_numbered_notes must NOT match it (or the two parsers would step on each other when used in fallback fashion). The `^` after `[` disqualifies it.""" draft = "[^1]: pandoc-style note.\n" result = find_numbered_notes(draft) assert result.definitions == [] def test_find_numbered_notes_ignores_non_numeric_markers(): """A marker like [Smith 2020] or [foo] is NOT a numbered footnote definition — it could be many other things (citation reference, link label, etc.). Only digit markers count.""" draft = """\ [Smith 2020] Some text — looks like an author-date reference. [foo] some kind of label. [1] Actual numbered note. """ result = find_numbered_notes(draft) assert len(result.definitions) == 1 assert result.definitions[0].marker == "1" def test_find_numbered_notes_records_original_prefix(): """For numbered format the prefix is `[N] ` — square bracket, number, closing bracket, single space. Reassembly will use this verbatim.""" draft = "[1] text.\n[107] text.\n" result = find_numbered_notes(draft) assert result.definitions[0].original_prefix == "[1] " assert result.definitions[1].original_prefix == "[107] " def test_find_numbered_notes_handles_anti_communist_draft_format(): """Smoke test mirroring the actual Anti-Communist Formations draft: notes follow a `## Notes` heading, separated by `________________` rule, one definition per line, sequentially numbered.""" draft = """\ ## Notes ________________ [1] First reference. https://example.org/path [2] Second reference, abbreviated. [3] Rosen, 7. """ result = find_numbered_notes(draft) assert len(result.definitions) == 3 assert result.definitions[0].text == "First reference. https://example.org/path" assert result.definitions[2].text == "Rosen, 7." # --- find_blockquote_notes (> N text format, from PDF-to-markdown) ---------- def test_find_blockquote_notes_single_definition(): draft = """\ Some prose. > 1 Donald Davidson, "On the Very Idea of a Conceptual Scheme." """ result = find_blockquote_notes(draft) assert len(result.definitions) == 1 assert result.definitions[0].marker == "1" assert result.definitions[0].text == 'Donald Davidson, "On the Very Idea of a Conceptual Scheme."' def test_find_blockquote_notes_multiple_in_order(): draft = """\ > 1 First note. > 2 Second note. > 3 Third note. """ result = find_blockquote_notes(draft) assert len(result.definitions) == 3 assert [d.marker for d in result.definitions] == ["1", "2", "3"] assert [d.text for d in result.definitions] == [ "First note.", "Second note.", "Third note.", ] def test_find_blockquote_notes_multi_digit_markers(): draft = "> 1 one.\n> 42 forty-two.\n> 88 eighty-eight.\n" result = find_blockquote_notes(draft) assert [d.marker for d in result.definitions] == ["1", "42", "88"] def test_find_blockquote_notes_returns_empty_when_no_definitions(): draft = "Just prose.\n> A regular blockquote with no number.\n" result = find_blockquote_notes(draft) assert result.definitions == [] def test_find_blockquote_notes_records_line_number(): draft = """\ Line zero. Line one. > 1 Note on line three. """ result = find_blockquote_notes(draft) assert result.definitions[0].line_number == 3 def test_find_blockquote_notes_strips_trailing_whitespace(): draft = "> 1 Note text with trailing spaces. \n" result = find_blockquote_notes(draft) assert result.definitions[0].text == "Note text with trailing spaces." def test_find_blockquote_notes_records_original_prefix(): """Reassembly prefix for blockquote format is ``'> N '``.""" draft = "> 1 text.\n> 88 text.\n" result = find_blockquote_notes(draft) assert result.definitions[0].original_prefix == "> 1 " assert result.definitions[1].original_prefix == "> 88 " def test_find_blockquote_notes_ignores_non_numeric_blockquotes(): """A blockquote without a leading number is regular prose, not a note.""" draft = """\ > This is a regular blockquote. > 1 This is a note. > Another regular blockquote. """ result = find_blockquote_notes(draft) assert len(result.definitions) == 1 assert result.definitions[0].marker == "1"