parser: support diverse md inputs (bold headings, Works Cited, blockquote notes, junk filtering)
- Bibliography heading now accepts bold markers (## **BIBLIOGRAPHY**), alternative names (Works Cited, References), and numbered sub-headings within the section. - Original heading text preserved in output instead of hardcoded "## Bibliography". - New find_blockquote_notes() parser for PDF-to-markdown footnote format (> N text), wired into CLI as third fallback after pandoc and numbered. - PDF junk filtered from bibliography entries: bare page numbers, blockquote footnotes, download banners, CC license URLs, and short running headers.
This commit is contained in:
+17
-9
@@ -23,7 +23,12 @@ from typing import Callable
|
|||||||
|
|
||||||
from cmos.formatter import format_bibliography_entry
|
from cmos.formatter import format_bibliography_entry
|
||||||
from cmos.note_formatter import format_note_entry
|
from cmos.note_formatter import format_note_entry
|
||||||
from cmos.parser import find_notes, find_numbered_notes, split_bibliography
|
from cmos.parser import (
|
||||||
|
find_blockquote_notes,
|
||||||
|
find_notes,
|
||||||
|
find_numbered_notes,
|
||||||
|
split_bibliography,
|
||||||
|
)
|
||||||
|
|
||||||
FormatterFn = Callable[[str], str]
|
FormatterFn = Callable[[str], str]
|
||||||
|
|
||||||
@@ -61,7 +66,7 @@ def reformat_draft(
|
|||||||
pieces: list[str] = []
|
pieces: list[str] = []
|
||||||
if parsed.before:
|
if parsed.before:
|
||||||
pieces.append(parsed.before)
|
pieces.append(parsed.before)
|
||||||
pieces.append("## Bibliography")
|
pieces.append(parsed.heading)
|
||||||
pieces.append("")
|
pieces.append("")
|
||||||
pieces.extend(rewritten)
|
pieces.extend(rewritten)
|
||||||
if parsed.after:
|
if parsed.after:
|
||||||
@@ -77,13 +82,14 @@ def reformat_notes(
|
|||||||
) -> str:
|
) -> str:
|
||||||
"""Rewrite footnote definitions in ``text`` to CMOS 18 note form.
|
"""Rewrite footnote definitions in ``text`` to CMOS 18 note form.
|
||||||
|
|
||||||
Auto-detects the source format by trying ``cmos.parser.find_notes``
|
Auto-detects the source format by trying three parsers in order:
|
||||||
(pandoc-style ``[^marker]: text``) first and falling back to
|
``find_notes`` (pandoc-style ``[^marker]: text``), then
|
||||||
``cmos.parser.find_numbered_notes`` (``[N] text``) if no pandoc
|
``find_numbered_notes`` (``[N] text``), then
|
||||||
definitions were found. Both formats are common in real drafts:
|
``find_blockquote_notes`` (``> N text``). All three formats are
|
||||||
pandoc is used by writers who author directly in markdown, while
|
common in real drafts: pandoc is used by writers who author
|
||||||
the numbered form is what docx-to-text conversion typically
|
directly in markdown, the numbered form comes from docx-to-text
|
||||||
produces from footnoted Word documents.
|
conversion, and the blockquote form comes from PDF-to-markdown
|
||||||
|
conversion.
|
||||||
|
|
||||||
Each definition's body is formatted via ``formatter`` (the v2
|
Each definition's body is formatted via ``formatter`` (the v2
|
||||||
note formatter by default) and substituted back into the same
|
note formatter by default) and substituted back into the same
|
||||||
@@ -115,6 +121,8 @@ def reformat_notes(
|
|||||||
parsed = find_notes(text)
|
parsed = find_notes(text)
|
||||||
if not parsed.definitions:
|
if not parsed.definitions:
|
||||||
parsed = find_numbered_notes(text)
|
parsed = find_numbered_notes(text)
|
||||||
|
if not parsed.definitions:
|
||||||
|
parsed = find_blockquote_notes(text)
|
||||||
|
|
||||||
if not parsed.definitions:
|
if not parsed.definitions:
|
||||||
return text
|
return text
|
||||||
|
|||||||
+96
-16
@@ -3,33 +3,38 @@
|
|||||||
v1 scope: find a ``## Bibliography`` heading (case-insensitive), treat the
|
v1 scope: find a ``## Bibliography`` heading (case-insensitive), treat the
|
||||||
section as running until the next level-2 (``##``) heading or end of file,
|
section as running until the next level-2 (``##``) heading or end of file,
|
||||||
and return each non-blank line as one entry. Bullet markers (``- ``, ``* ``)
|
and return each non-blank line as one entry. Bullet markers (``- ``, ``* ``)
|
||||||
at the start of a line are stripped.
|
at the start of a line are stripped. The heading may also be
|
||||||
|
``## Works Cited`` or ``## References`` (case-insensitive), and may be
|
||||||
|
wrapped in bold markers (``## **Bibliography**``). Numbered sub-headings
|
||||||
|
within the bibliography (e.g., ``## **1. Primary Sources**``) are skipped
|
||||||
|
rather than treated as section terminators.
|
||||||
|
|
||||||
v2 scope (notes): two complementary functions for finding footnote
|
v2 scope (notes): three complementary functions for finding footnote
|
||||||
definitions in two distinct source formats. Both return the same
|
definitions in distinct source formats. All return the same
|
||||||
``NotesParseResult`` shape; the consumer (typically the CLI) can call
|
``NotesParseResult`` shape; the consumer (typically the CLI) can call
|
||||||
one and fall back to the other to auto-detect format.
|
them in fallback order to auto-detect format.
|
||||||
|
|
||||||
- ``find_notes`` finds pandoc-style markdown footnote definitions
|
- ``find_notes`` finds pandoc-style markdown footnote definitions
|
||||||
``[^marker]: text``. The marker can be numeric or named.
|
``[^marker]: text``. The marker can be numeric or named.
|
||||||
- ``find_numbered_notes`` finds plain ``[N] text`` definitions where
|
- ``find_numbered_notes`` finds plain ``[N] text`` definitions where
|
||||||
the marker is digit-only and there is no caret or colon — the format
|
the marker is digit-only and there is no caret or colon — the format
|
||||||
produced by docx-to-text conversion of footnoted Word docs.
|
produced by docx-to-text conversion of footnoted Word docs.
|
||||||
|
- ``find_blockquote_notes`` finds blockquote-style ``> N text``
|
||||||
|
definitions — the format produced by PDF-to-markdown conversion.
|
||||||
|
|
||||||
Each ``NoteDefinition`` carries the marker, the body text (with
|
Each ``NoteDefinition`` carries the marker, the body text (with
|
||||||
surrounding whitespace stripped), the line number where it appeared
|
surrounding whitespace stripped), the line number where it appeared
|
||||||
(0-indexed, for future reassembly), and the literal ``original_prefix``
|
(0-indexed, for future reassembly), and the literal ``original_prefix``
|
||||||
string (e.g., ``"[^1]: "`` or ``"[1] "``) so that reassembly can
|
string (e.g., ``"[^1]: "``, ``"[1] "``, or ``"> 1 "``) so that
|
||||||
round-trip the source's marker syntax without the consumer needing to
|
reassembly can round-trip the source's marker syntax without the
|
||||||
know which format was matched.
|
consumer needing to know which format was matched.
|
||||||
|
|
||||||
Single-line definitions only in both formats — multi-line continuation
|
Single-line definitions only in all formats — multi-line continuation
|
||||||
(indented continuation lines) is deferred. The reference markers in
|
(indented continuation lines) is deferred. The reference markers in
|
||||||
prose are NOT collected; only the definitions need reformatting.
|
prose are NOT collected; only the definitions need reformatting.
|
||||||
|
|
||||||
Out of scope in v1: in-prose citation rewriting, multi-line entries,
|
Out of scope in v1: in-prose citation rewriting, multi-line entries.
|
||||||
nested sections, alternative heading names (e.g., "Works Cited",
|
Out of scope in v2 (so far): multi-line note definitions,
|
||||||
"References"). Out of scope in v2 (so far): multi-line note definitions,
|
|
||||||
shortened-form generation.
|
shortened-form generation.
|
||||||
"""
|
"""
|
||||||
|
|
||||||
@@ -48,6 +53,7 @@ class ParseResult:
|
|||||||
before: str
|
before: str
|
||||||
entries: list[str]
|
entries: list[str]
|
||||||
after: str
|
after: str
|
||||||
|
heading: str = "## Bibliography"
|
||||||
|
|
||||||
|
|
||||||
@dataclass
|
@dataclass
|
||||||
@@ -77,11 +83,43 @@ class NotesParseResult:
|
|||||||
definitions: list[NoteDefinition]
|
definitions: list[NoteDefinition]
|
||||||
|
|
||||||
|
|
||||||
_HEADING_RE = re.compile(r"^##\s+bibliography\s*$", re.IGNORECASE)
|
_HEADING_RE = re.compile(
|
||||||
|
r"^##\s+\*{0,2}(?:bibliography|works\s+cited|references)\*{0,2}\s*$",
|
||||||
|
re.IGNORECASE,
|
||||||
|
)
|
||||||
_LEVEL_TWO_RE = re.compile(r"^##\s+\S")
|
_LEVEL_TWO_RE = re.compile(r"^##\s+\S")
|
||||||
|
_NUMBERED_SUB_HEADING_RE = re.compile(
|
||||||
|
r"^##\s+[_*]*\d", re.IGNORECASE
|
||||||
|
)
|
||||||
_BULLET_RE = re.compile(r"^[-*]\s+")
|
_BULLET_RE = re.compile(r"^[-*]\s+")
|
||||||
|
_BARE_PAGE_RE = re.compile(r"^\d+$")
|
||||||
|
_DOWNLOAD_BANNER_RE = re.compile(r"^Downloaded from ", re.IGNORECASE)
|
||||||
|
_CC_LICENSE_RE = re.compile(r"creativecommons\.org")
|
||||||
|
_CITATION_MARKERS_RE = re.compile(r"""[,*_"(]|://""")
|
||||||
_NOTE_DEF_RE = re.compile(r"^\[\^([^\]]+)\]:\s*(.*?)\s*$")
|
_NOTE_DEF_RE = re.compile(r"^\[\^([^\]]+)\]:\s*(.*?)\s*$")
|
||||||
_NUMBERED_NOTE_RE = re.compile(r"^\[(\d+)\]\s+(.*?)\s*$")
|
_NUMBERED_NOTE_RE = re.compile(r"^\[(\d+)\]\s+(.*?)\s*$")
|
||||||
|
_BLOCKQUOTE_NOTE_RE = re.compile(r"^>\s*(\d+)\s+(.*?)\s*$")
|
||||||
|
|
||||||
|
|
||||||
|
def _is_pdf_junk(line: str) -> bool:
|
||||||
|
"""Return True if ``line`` looks like PDF-to-markdown noise rather
|
||||||
|
than a bibliography entry.
|
||||||
|
|
||||||
|
Catches: bare page numbers, blockquote footnotes, publisher download
|
||||||
|
banners, Creative Commons license URLs, and short running headers
|
||||||
|
(under 30 chars with no citation-like punctuation).
|
||||||
|
"""
|
||||||
|
if _BARE_PAGE_RE.match(line):
|
||||||
|
return True
|
||||||
|
if _BLOCKQUOTE_NOTE_RE.match(line):
|
||||||
|
return True
|
||||||
|
if _DOWNLOAD_BANNER_RE.match(line):
|
||||||
|
return True
|
||||||
|
if _CC_LICENSE_RE.search(line):
|
||||||
|
return True
|
||||||
|
if len(line) < 30 and not _CITATION_MARKERS_RE.search(line):
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
def split_bibliography(text: str) -> ParseResult:
|
def split_bibliography(text: str) -> ParseResult:
|
||||||
@@ -96,14 +134,18 @@ def split_bibliography(text: str) -> ParseResult:
|
|||||||
if heading_idx is None:
|
if heading_idx is None:
|
||||||
raise NoBibliographyError("no '## Bibliography' heading found")
|
raise NoBibliographyError("no '## Bibliography' heading found")
|
||||||
|
|
||||||
# Find the end of the section: next level-two heading after the bibliography,
|
# Find the end of the section: next level-two heading that is NOT a
|
||||||
# or end of file.
|
# numbered sub-heading (e.g., "## 1. Primary Sources"). Numbered
|
||||||
|
# sub-headings are part of the bibliography and should be skipped.
|
||||||
section_end = len(lines)
|
section_end = len(lines)
|
||||||
for j in range(heading_idx + 1, len(lines)):
|
for j in range(heading_idx + 1, len(lines)):
|
||||||
if _LEVEL_TWO_RE.match(lines[j]):
|
if _LEVEL_TWO_RE.match(lines[j]) and not _NUMBERED_SUB_HEADING_RE.match(
|
||||||
|
lines[j]
|
||||||
|
):
|
||||||
section_end = j
|
section_end = j
|
||||||
break
|
break
|
||||||
|
|
||||||
|
heading = lines[heading_idx].rstrip()
|
||||||
before = "\n".join(lines[:heading_idx])
|
before = "\n".join(lines[:heading_idx])
|
||||||
after_lines = lines[section_end:]
|
after_lines = lines[section_end:]
|
||||||
after = "\n".join(after_lines)
|
after = "\n".join(after_lines)
|
||||||
@@ -113,10 +155,17 @@ def split_bibliography(text: str) -> ParseResult:
|
|||||||
stripped = raw.strip()
|
stripped = raw.strip()
|
||||||
if not stripped:
|
if not stripped:
|
||||||
continue
|
continue
|
||||||
|
# Skip numbered sub-headings — they subdivide the bibliography
|
||||||
|
# but are not entries themselves.
|
||||||
|
if _NUMBERED_SUB_HEADING_RE.match(raw):
|
||||||
|
continue
|
||||||
stripped = _BULLET_RE.sub("", stripped)
|
stripped = _BULLET_RE.sub("", stripped)
|
||||||
|
# Filter PDF-to-markdown junk.
|
||||||
|
if _is_pdf_junk(stripped):
|
||||||
|
continue
|
||||||
entries.append(stripped)
|
entries.append(stripped)
|
||||||
|
|
||||||
return ParseResult(before=before, entries=entries, after=after)
|
return ParseResult(before=before, entries=entries, after=after, heading=heading)
|
||||||
|
|
||||||
|
|
||||||
def find_notes(text: str) -> NotesParseResult:
|
def find_notes(text: str) -> NotesParseResult:
|
||||||
@@ -194,3 +243,34 @@ def find_numbered_notes(text: str) -> NotesParseResult:
|
|||||||
)
|
)
|
||||||
)
|
)
|
||||||
return NotesParseResult(definitions=definitions)
|
return NotesParseResult(definitions=definitions)
|
||||||
|
|
||||||
|
|
||||||
|
def find_blockquote_notes(text: str) -> NotesParseResult:
|
||||||
|
"""Find blockquote-style footnote definitions: ``> N text``.
|
||||||
|
|
||||||
|
Targets the format produced by PDF-to-markdown conversion where
|
||||||
|
footnotes appear as blockquoted lines beginning with a bare number.
|
||||||
|
The marker must be all digits — a blockquote line without a leading
|
||||||
|
number is regular prose and is not matched.
|
||||||
|
|
||||||
|
Returns a ``NotesParseResult`` with definitions in source order.
|
||||||
|
Each definition's ``original_prefix`` is set to ``"> N "`` so that
|
||||||
|
downstream reassembly can write back the formatted body using the
|
||||||
|
same marker syntax.
|
||||||
|
"""
|
||||||
|
definitions: list[NoteDefinition] = []
|
||||||
|
for line_number, line in enumerate(text.splitlines()):
|
||||||
|
match = _BLOCKQUOTE_NOTE_RE.match(line)
|
||||||
|
if match is None:
|
||||||
|
continue
|
||||||
|
marker = match.group(1)
|
||||||
|
body = match.group(2)
|
||||||
|
definitions.append(
|
||||||
|
NoteDefinition(
|
||||||
|
marker=marker,
|
||||||
|
text=body,
|
||||||
|
line_number=line_number,
|
||||||
|
original_prefix=f"> {marker} ",
|
||||||
|
)
|
||||||
|
)
|
||||||
|
return NotesParseResult(definitions=definitions)
|
||||||
|
|||||||
+51
-18
@@ -51,14 +51,14 @@ def test_reformat_draft_preserves_sections_after_bibliography():
|
|||||||
draft = """\
|
draft = """\
|
||||||
## Bibliography
|
## Bibliography
|
||||||
|
|
||||||
one entry.
|
Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
|
||||||
|
|
||||||
## Appendix
|
## Appendix
|
||||||
|
|
||||||
appendix text.
|
appendix text.
|
||||||
"""
|
"""
|
||||||
output = reformat_draft(draft, formatter=_fake_formatter)
|
output = reformat_draft(draft, formatter=_fake_formatter)
|
||||||
assert "FORMATTED(one entry.)" in output
|
assert "FORMATTED(Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).)" in output
|
||||||
assert "## Appendix" in output
|
assert "## Appendix" in output
|
||||||
assert "appendix text." in output
|
assert "appendix text." in output
|
||||||
|
|
||||||
@@ -69,11 +69,27 @@ def test_reformat_draft_raises_when_no_bibliography():
|
|||||||
|
|
||||||
|
|
||||||
def test_reformat_draft_bibliography_heading_preserved():
|
def test_reformat_draft_bibliography_heading_preserved():
|
||||||
draft = "## Bibliography\n\nentry.\n"
|
draft = '## Bibliography\n\nYu, Charles. *Interior Chinatown* (Pantheon, 2020).\n'
|
||||||
output = reformat_draft(draft, formatter=_fake_formatter)
|
output = reformat_draft(draft, formatter=_fake_formatter)
|
||||||
assert "## Bibliography" in output
|
assert "## Bibliography" in output
|
||||||
|
|
||||||
|
|
||||||
|
def test_reformat_draft_preserves_works_cited_heading():
|
||||||
|
"""When the input heading is ``## Works Cited``, the output must
|
||||||
|
keep that heading — not replace it with ``## Bibliography``."""
|
||||||
|
draft = '## Works Cited\n\nDavidson, Donald. "On the Very Idea." (1974).\n'
|
||||||
|
output = reformat_draft(draft, formatter=_fake_formatter)
|
||||||
|
assert "## Works Cited" in output
|
||||||
|
assert "## Bibliography" not in output
|
||||||
|
|
||||||
|
|
||||||
|
def test_reformat_draft_preserves_bold_wrapped_heading():
|
||||||
|
"""Bold markers in the heading should be preserved in the output."""
|
||||||
|
draft = '## **Works Cited**\n\nDavidson, Donald. "On the Very Idea." (1974).\n'
|
||||||
|
output = reformat_draft(draft, formatter=_fake_formatter)
|
||||||
|
assert "## **Works Cited**" in output
|
||||||
|
|
||||||
|
|
||||||
def test_python_dash_m_invocation_actually_runs_main():
|
def test_python_dash_m_invocation_actually_runs_main():
|
||||||
"""Regression test: `python -m cmos.cli` must actually invoke main().
|
"""Regression test: `python -m cmos.cli` must actually invoke main().
|
||||||
|
|
||||||
@@ -216,6 +232,23 @@ def test_reformat_notes_auto_detects_numbered_format():
|
|||||||
assert "[^2]" not in output
|
assert "[^2]" not in output
|
||||||
|
|
||||||
|
|
||||||
|
def test_reformat_notes_auto_detects_blockquote_format():
|
||||||
|
"""PDF-to-markdown drafts use ``> N text`` blockquote footnotes.
|
||||||
|
reformat_notes must auto-detect after pandoc and numbered fail."""
|
||||||
|
draft = """\
|
||||||
|
Some prose.
|
||||||
|
|
||||||
|
> 1 Davidson, "On the Very Idea of a Conceptual Scheme."
|
||||||
|
> 2 Frankenberry 1999, 526.
|
||||||
|
"""
|
||||||
|
output = reformat_notes(draft, formatter=_fake_note_formatter)
|
||||||
|
assert '> 1 NOTE_FORMATTED(Davidson, "On the Very Idea of a Conceptual Scheme.")' in output
|
||||||
|
assert "> 2 NOTE_FORMATTED(Frankenberry 1999, 526.)" in output
|
||||||
|
# No pandoc or numbered markers should appear.
|
||||||
|
assert "[^1]" not in output
|
||||||
|
assert "[1] " not in output
|
||||||
|
|
||||||
|
|
||||||
def test_reformat_notes_pandoc_input_still_round_trips_as_pandoc():
|
def test_reformat_notes_pandoc_input_still_round_trips_as_pandoc():
|
||||||
"""Regression: after the auto-detect change, pandoc-format input
|
"""Regression: after the auto-detect change, pandoc-format input
|
||||||
must still produce pandoc-format output. The original_prefix path
|
must still produce pandoc-format output. The original_prefix path
|
||||||
@@ -265,28 +298,28 @@ def test_reformat_draft_preserves_order_under_concurrency():
|
|||||||
# later ones if order were naively tied to completion.
|
# later ones if order were naively tied to completion.
|
||||||
import time
|
import time
|
||||||
|
|
||||||
|
# Map each entry to a sleep duration so earlier entries finish last.
|
||||||
|
order = {"2020": 4, "2021": 3, "2022": 2, "2023": 1, "2024": 0}
|
||||||
|
|
||||||
def slow_fake(messy: str) -> str:
|
def slow_fake(messy: str) -> str:
|
||||||
# Entries with lower index sleep longer so they finish last.
|
year = messy.strip()[-5:-1] # extract "2020" etc.
|
||||||
n = int(messy.split()[-1])
|
time.sleep(0.05 * order.get(year, 0))
|
||||||
time.sleep(0.05 * (5 - n))
|
|
||||||
return f"FORMATTED({messy})"
|
return f"FORMATTED({messy})"
|
||||||
|
|
||||||
draft = """\
|
draft = """\
|
||||||
## Bibliography
|
## Bibliography
|
||||||
|
|
||||||
entry 0
|
Author, A. *Title Zero* (Publisher, 2020).
|
||||||
entry 1
|
Author, B. *Title One* (Publisher, 2021).
|
||||||
entry 2
|
Author, C. *Title Two* (Publisher, 2022).
|
||||||
entry 3
|
Author, D. *Title Three* (Publisher, 2023).
|
||||||
entry 4
|
Author, E. *Title Four* (Publisher, 2024).
|
||||||
"""
|
"""
|
||||||
output = reformat_draft(draft, formatter=slow_fake, concurrency=4)
|
output = reformat_draft(draft, formatter=slow_fake, concurrency=4)
|
||||||
# Entries must appear in input order.
|
# Entries must appear in input order.
|
||||||
lines = [l for l in output.splitlines() if l.startswith("FORMATTED(")]
|
lines = [l for l in output.splitlines() if l.startswith("FORMATTED(")]
|
||||||
assert lines == [
|
assert lines[0].startswith("FORMATTED(Author, A.")
|
||||||
"FORMATTED(entry 0)",
|
assert lines[1].startswith("FORMATTED(Author, B.")
|
||||||
"FORMATTED(entry 1)",
|
assert lines[2].startswith("FORMATTED(Author, C.")
|
||||||
"FORMATTED(entry 2)",
|
assert lines[3].startswith("FORMATTED(Author, D.")
|
||||||
"FORMATTED(entry 3)",
|
assert lines[4].startswith("FORMATTED(Author, E.")
|
||||||
"FORMATTED(entry 4)",
|
|
||||||
]
|
|
||||||
|
|||||||
+277
-9
@@ -26,6 +26,7 @@ import pytest
|
|||||||
|
|
||||||
from cmos.parser import (
|
from cmos.parser import (
|
||||||
NoBibliographyError,
|
NoBibliographyError,
|
||||||
|
find_blockquote_notes,
|
||||||
find_notes,
|
find_notes,
|
||||||
find_numbered_notes,
|
find_numbered_notes,
|
||||||
split_bibliography,
|
split_bibliography,
|
||||||
@@ -57,15 +58,16 @@ def test_section_ends_at_next_level_two_heading():
|
|||||||
draft = """\
|
draft = """\
|
||||||
## Bibliography
|
## Bibliography
|
||||||
|
|
||||||
First entry.
|
Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
|
||||||
Second entry.
|
Kwon, Hyeyoung. "Inclusion Work," *American Journal of Sociology* 127 (2022).
|
||||||
|
|
||||||
## Appendix
|
## Appendix
|
||||||
|
|
||||||
Appendix content here.
|
Appendix content here.
|
||||||
"""
|
"""
|
||||||
result = split_bibliography(draft)
|
result = split_bibliography(draft)
|
||||||
assert result.entries == ["First entry.", "Second entry."]
|
assert len(result.entries) == 2
|
||||||
|
assert result.entries[0].startswith("Yu, Charles")
|
||||||
assert "## Appendix" in result.after
|
assert "## Appendix" in result.after
|
||||||
assert "Appendix content here." in result.after
|
assert "Appendix content here." in result.after
|
||||||
|
|
||||||
@@ -74,22 +76,208 @@ def test_bullet_prefixes_stripped():
|
|||||||
draft = """\
|
draft = """\
|
||||||
## Bibliography
|
## Bibliography
|
||||||
|
|
||||||
- First entry.
|
- Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
|
||||||
- Second entry.
|
- Kwon, Hyeyoung. "Inclusion Work," *AJS* 127 (2022).
|
||||||
* Third entry.
|
* Google. "Privacy Policy," Privacy & Terms, 2023.
|
||||||
"""
|
"""
|
||||||
result = split_bibliography(draft)
|
result = split_bibliography(draft)
|
||||||
assert result.entries == ["First entry.", "Second entry.", "Third entry."]
|
assert len(result.entries) == 3
|
||||||
|
assert result.entries[0].startswith("Yu, Charles")
|
||||||
|
assert result.entries[1].startswith("Kwon, Hyeyoung")
|
||||||
|
assert result.entries[2].startswith("Google")
|
||||||
|
|
||||||
|
|
||||||
def test_heading_match_is_case_insensitive():
|
def test_heading_match_is_case_insensitive():
|
||||||
draft = """\
|
draft = """\
|
||||||
## bibliography
|
## bibliography
|
||||||
|
|
||||||
Only entry.
|
Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
|
||||||
"""
|
"""
|
||||||
result = split_bibliography(draft)
|
result = split_bibliography(draft)
|
||||||
assert result.entries == ["Only entry."]
|
assert len(result.entries) == 1
|
||||||
|
assert result.entries[0].startswith("Yu, Charles")
|
||||||
|
|
||||||
|
|
||||||
|
def test_heading_match_works_cited():
|
||||||
|
"""'Works Cited' is the standard CMOS heading alternative to
|
||||||
|
'Bibliography'. The parser should accept it."""
|
||||||
|
draft = """\
|
||||||
|
## Works Cited
|
||||||
|
|
||||||
|
Davidson, Donald. "On the Very Idea of a Conceptual Scheme."
|
||||||
|
Evans-Pritchard, E.E. *Witchcraft, oracles, and magic.*
|
||||||
|
"""
|
||||||
|
result = split_bibliography(draft)
|
||||||
|
assert len(result.entries) == 2
|
||||||
|
assert result.entries[0].startswith("Davidson")
|
||||||
|
|
||||||
|
|
||||||
|
def test_heading_match_works_cited_bold():
|
||||||
|
"""Bold-wrapped variant: ``## **Works Cited**``."""
|
||||||
|
draft = """\
|
||||||
|
## **Works Cited**
|
||||||
|
|
||||||
|
- Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999).
|
||||||
|
"""
|
||||||
|
result = split_bibliography(draft)
|
||||||
|
assert len(result.entries) == 1
|
||||||
|
assert result.entries[0].startswith("Frankenberry")
|
||||||
|
|
||||||
|
|
||||||
|
def test_heading_match_ignores_bold_markers():
|
||||||
|
"""Real-world drafts from PDF-to-markdown conversion often wrap the
|
||||||
|
heading text in bold: ``## **BIBLIOGRAPHY**``. The parser should
|
||||||
|
strip ``**`` before matching."""
|
||||||
|
draft = """\
|
||||||
|
## **BIBLIOGRAPHY**
|
||||||
|
|
||||||
|
- Yu, Charles. *Interior Chinatown* (New York: Pantheon, 2020).
|
||||||
|
- Kwon, Hyeyoung. "Inclusion Work," *AJS* 127 (2022).
|
||||||
|
"""
|
||||||
|
result = split_bibliography(draft)
|
||||||
|
assert len(result.entries) == 2
|
||||||
|
assert result.entries[0].startswith("Yu, Charles")
|
||||||
|
|
||||||
|
|
||||||
|
def test_heading_match_ignores_numbered_sub_headings():
|
||||||
|
"""Subdivided bibliographies (common in dissertations) use numbered
|
||||||
|
``##`` sub-headings inside the bibliography section. These must NOT
|
||||||
|
terminate the section — the parser should skip them and keep
|
||||||
|
collecting entries until a non-numbered ``##`` heading or EOF."""
|
||||||
|
draft = """\
|
||||||
|
## **BIBLIOGRAPHY**
|
||||||
|
|
||||||
|
## **1. Primary Sources**
|
||||||
|
|
||||||
|
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
|
||||||
|
- Till, Walter C., ed., *Die koptischen Ostraka* (Vienna, 1960).
|
||||||
|
|
||||||
|
## **2. Secondary Literature**
|
||||||
|
|
||||||
|
- Marsham, Andrew, ed., *The Umayyad World* (London: Routledge, 2020).
|
||||||
|
- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
|
||||||
|
"""
|
||||||
|
result = split_bibliography(draft)
|
||||||
|
assert len(result.entries) == 4
|
||||||
|
assert result.entries[0].startswith("Crum, Walter")
|
||||||
|
assert result.entries[3].startswith("Schulz, Fritz")
|
||||||
|
assert result.after == ""
|
||||||
|
|
||||||
|
|
||||||
|
def test_numbered_sub_headings_not_collected_as_entries():
|
||||||
|
"""The sub-heading lines themselves (e.g. ``## **1. Primary Sources**``)
|
||||||
|
should not appear in the entries list."""
|
||||||
|
draft = """\
|
||||||
|
## Bibliography
|
||||||
|
|
||||||
|
## 1. Primary Sources
|
||||||
|
|
||||||
|
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
|
||||||
|
|
||||||
|
## 2. Secondary Sources
|
||||||
|
|
||||||
|
- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
|
||||||
|
"""
|
||||||
|
result = split_bibliography(draft)
|
||||||
|
assert len(result.entries) == 2
|
||||||
|
for entry in result.entries:
|
||||||
|
assert not entry.startswith("##")
|
||||||
|
assert "Primary Sources" not in entry
|
||||||
|
assert "Secondary Sources" not in entry
|
||||||
|
|
||||||
|
|
||||||
|
def test_non_numbered_heading_still_ends_section_after_sub_headings():
|
||||||
|
"""A non-numbered ``##`` heading after bibliography sub-sections
|
||||||
|
should still terminate the bibliography section."""
|
||||||
|
draft = """\
|
||||||
|
## Bibliography
|
||||||
|
|
||||||
|
## 1. Primary Sources
|
||||||
|
|
||||||
|
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
|
||||||
|
|
||||||
|
## Appendix
|
||||||
|
|
||||||
|
Appendix content.
|
||||||
|
"""
|
||||||
|
result = split_bibliography(draft)
|
||||||
|
assert len(result.entries) == 1
|
||||||
|
assert result.entries[0].startswith("Crum, Walter")
|
||||||
|
assert "## Appendix" in result.after
|
||||||
|
assert "Appendix content." in result.after
|
||||||
|
|
||||||
|
|
||||||
|
def test_bare_page_numbers_filtered():
|
||||||
|
"""PDF-to-markdown conversion leaves bare page numbers (e.g. '370')
|
||||||
|
as standalone lines. These are not bibliography entries."""
|
||||||
|
draft = """\
|
||||||
|
## Bibliography
|
||||||
|
|
||||||
|
- Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
|
||||||
|
|
||||||
|
370
|
||||||
|
|
||||||
|
- Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
|
||||||
|
|
||||||
|
371
|
||||||
|
"""
|
||||||
|
result = split_bibliography(draft)
|
||||||
|
assert len(result.entries) == 2
|
||||||
|
assert result.entries[0].startswith("Crum")
|
||||||
|
assert result.entries[1].startswith("Schulz")
|
||||||
|
|
||||||
|
|
||||||
|
def test_blockquote_footnotes_filtered():
|
||||||
|
"""Blockquote footnotes (``> N text``) that appear inside a Works Cited
|
||||||
|
section are PDF artifacts, not bibliography entries."""
|
||||||
|
draft = """\
|
||||||
|
## Works Cited
|
||||||
|
|
||||||
|
Davidson, Donald. "On the Very Idea of a Conceptual Scheme."
|
||||||
|
|
||||||
|
> 87 Holbraad 2010.
|
||||||
|
> 88 Frankenberry 2018, 236.
|
||||||
|
|
||||||
|
Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999).
|
||||||
|
"""
|
||||||
|
result = split_bibliography(draft)
|
||||||
|
assert len(result.entries) == 2
|
||||||
|
assert result.entries[0].startswith("Davidson")
|
||||||
|
assert result.entries[1].startswith("Frankenberry")
|
||||||
|
|
||||||
|
|
||||||
|
def test_download_banners_filtered():
|
||||||
|
"""Brill/publisher download banners are PDF junk, not entries."""
|
||||||
|
draft = """\
|
||||||
|
## Bibliography
|
||||||
|
|
||||||
|
Crum, Walter E., *A Coptic Dictionary* (Oxford: Clarendon Press, 1939).
|
||||||
|
Downloaded from Brill.com 02/22/2024 05:33:29AM via Open Access. This is junk.
|
||||||
|
https://creativecommons.org/licenses/by/4.0/
|
||||||
|
Schulz, Fritz, *Classical Roman Law* (Oxford: Clarendon Press, 1951).
|
||||||
|
"""
|
||||||
|
result = split_bibliography(draft)
|
||||||
|
assert len(result.entries) == 2
|
||||||
|
assert result.entries[0].startswith("Crum")
|
||||||
|
assert result.entries[1].startswith("Schulz")
|
||||||
|
|
||||||
|
|
||||||
|
def test_running_headers_filtered():
|
||||||
|
"""Short lines that are running headers or page footers (e.g. an
|
||||||
|
author surname or abbreviated title) are not entries."""
|
||||||
|
draft = """\
|
||||||
|
## Bibliography
|
||||||
|
|
||||||
|
Davidson, Donald. "On the Very Idea of a Conceptual Scheme."
|
||||||
|
hedrick
|
||||||
|
only words apart?
|
||||||
|
217
|
||||||
|
Frankenberry, Nancy. "Pragmatism, Truth, and Subjectivity" (1999).
|
||||||
|
"""
|
||||||
|
result = split_bibliography(draft)
|
||||||
|
assert len(result.entries) == 2
|
||||||
|
assert result.entries[0].startswith("Davidson")
|
||||||
|
assert result.entries[1].startswith("Frankenberry")
|
||||||
|
|
||||||
|
|
||||||
def test_no_bibliography_raises():
|
def test_no_bibliography_raises():
|
||||||
@@ -292,3 +480,83 @@ ________________
|
|||||||
assert len(result.definitions) == 3
|
assert len(result.definitions) == 3
|
||||||
assert result.definitions[0].text == "First reference. https://example.org/path"
|
assert result.definitions[0].text == "First reference. https://example.org/path"
|
||||||
assert result.definitions[2].text == "Rosen, 7."
|
assert result.definitions[2].text == "Rosen, 7."
|
||||||
|
|
||||||
|
|
||||||
|
# --- find_blockquote_notes (> N text format, from PDF-to-markdown) ----------
|
||||||
|
|
||||||
|
|
||||||
|
def test_find_blockquote_notes_single_definition():
|
||||||
|
draft = """\
|
||||||
|
Some prose.
|
||||||
|
|
||||||
|
> 1 Donald Davidson, "On the Very Idea of a Conceptual Scheme."
|
||||||
|
"""
|
||||||
|
result = find_blockquote_notes(draft)
|
||||||
|
assert len(result.definitions) == 1
|
||||||
|
assert result.definitions[0].marker == "1"
|
||||||
|
assert result.definitions[0].text == 'Donald Davidson, "On the Very Idea of a Conceptual Scheme."'
|
||||||
|
|
||||||
|
|
||||||
|
def test_find_blockquote_notes_multiple_in_order():
|
||||||
|
draft = """\
|
||||||
|
> 1 First note.
|
||||||
|
> 2 Second note.
|
||||||
|
> 3 Third note.
|
||||||
|
"""
|
||||||
|
result = find_blockquote_notes(draft)
|
||||||
|
assert len(result.definitions) == 3
|
||||||
|
assert [d.marker for d in result.definitions] == ["1", "2", "3"]
|
||||||
|
assert [d.text for d in result.definitions] == [
|
||||||
|
"First note.",
|
||||||
|
"Second note.",
|
||||||
|
"Third note.",
|
||||||
|
]
|
||||||
|
|
||||||
|
|
||||||
|
def test_find_blockquote_notes_multi_digit_markers():
|
||||||
|
draft = "> 1 one.\n> 42 forty-two.\n> 88 eighty-eight.\n"
|
||||||
|
result = find_blockquote_notes(draft)
|
||||||
|
assert [d.marker for d in result.definitions] == ["1", "42", "88"]
|
||||||
|
|
||||||
|
|
||||||
|
def test_find_blockquote_notes_returns_empty_when_no_definitions():
|
||||||
|
draft = "Just prose.\n> A regular blockquote with no number.\n"
|
||||||
|
result = find_blockquote_notes(draft)
|
||||||
|
assert result.definitions == []
|
||||||
|
|
||||||
|
|
||||||
|
def test_find_blockquote_notes_records_line_number():
|
||||||
|
draft = """\
|
||||||
|
Line zero.
|
||||||
|
Line one.
|
||||||
|
|
||||||
|
> 1 Note on line three.
|
||||||
|
"""
|
||||||
|
result = find_blockquote_notes(draft)
|
||||||
|
assert result.definitions[0].line_number == 3
|
||||||
|
|
||||||
|
|
||||||
|
def test_find_blockquote_notes_strips_trailing_whitespace():
|
||||||
|
draft = "> 1 Note text with trailing spaces. \n"
|
||||||
|
result = find_blockquote_notes(draft)
|
||||||
|
assert result.definitions[0].text == "Note text with trailing spaces."
|
||||||
|
|
||||||
|
|
||||||
|
def test_find_blockquote_notes_records_original_prefix():
|
||||||
|
"""Reassembly prefix for blockquote format is ``'> N '``."""
|
||||||
|
draft = "> 1 text.\n> 88 text.\n"
|
||||||
|
result = find_blockquote_notes(draft)
|
||||||
|
assert result.definitions[0].original_prefix == "> 1 "
|
||||||
|
assert result.definitions[1].original_prefix == "> 88 "
|
||||||
|
|
||||||
|
|
||||||
|
def test_find_blockquote_notes_ignores_non_numeric_blockquotes():
|
||||||
|
"""A blockquote without a leading number is regular prose, not a note."""
|
||||||
|
draft = """\
|
||||||
|
> This is a regular blockquote.
|
||||||
|
> 1 This is a note.
|
||||||
|
> Another regular blockquote.
|
||||||
|
"""
|
||||||
|
result = find_blockquote_notes(draft)
|
||||||
|
assert len(result.definitions) == 1
|
||||||
|
assert result.definitions[0].marker == "1"
|
||||||
|
|||||||
Reference in New Issue
Block a user