From 45f51c6341496ad17f5bc819d3b0adb4c57728f1 Mon Sep 17 00:00:00 2001 From: Mark Eaton Date: Sat, 11 Apr 2026 19:16:09 -0400 Subject: [PATCH] v2 chunk 2c: numbered-note format support + real-draft test MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Extend the v2 parser and CLI to handle [N] text footnote definitions in addition to pandoc-style [^marker]: text. The numbered format is what docx-to-text conversion of footnoted Word documents produces. The user's "Anti-Communist Formations of LIS" draft uses this format for all 107 of its notes; without this support, v2 was structurally incapable of running on real-world docx-derived inputs. src/cmos/parser.py: - Add NoteDefinition.original_prefix field so reassembly can round-trip the source's marker syntax (pandoc input → pandoc output, numbered input → numbered output) without consumers needing to know which format was matched. - Update find_notes() to populate original_prefix as "[^N]: ". - Add find_numbered_notes() targeting "[N] text" definitions. Marker must be all digits (rejects [Smith 2020], [foo], etc.); caret prefix is rejected (rejects pandoc-style cleanly). src/cmos/cli.py: - reformat_notes now auto-detects format: tries find_notes first, falls back to find_numbered_notes if no pandoc definitions found. Uses definition.original_prefix for reassembly so both formats round-trip correctly. tests/test_parser.py: - 11 new tests for find_numbered_notes covering: single/multiple definitions, multi-digit markers, line number recording, trailing whitespace stripping, ignoring pandoc/non-numeric markers, original prefix recording, and the actual Anti-Communist draft format. - 1 new test for find_notes original_prefix population. tests/test_cli.py: - 2 new tests for reformat_notes auto-detect: numbered input round- trips as numbered output, pandoc input still round-trips as pandoc. Total suite: 136/136 (was 123, +13 net new). v1 untouched, 96/96 v1 tests still passing. Real-draft validation: ran cmos format-notes against the 107-note Anti-Communist Formations of LIS draft (108 calls in parallel via the existing concurrency=8 thread pool, completed cleanly). Output saved to /tmp (not committed). All 107 notes preserved through the pipeline; ~30-40 first-occurrence full notes produced clean CMOS 18 note form; ~30 shortened-form refs correctly left unchanged; 2 real bugs surfaced for the next iteration (empty input → conver- sational reply, Ibid → empty string), plus several lower-priority issues (substantive note truncation, retry waste on shortened forms, lossy month dropping). Not addressed in this chunk per the "collect signal, don't fix" plan. --- src/cmos/cli.py | 49 +++++++++------ src/cmos/parser.py | 101 ++++++++++++++++++++++++++----- tests/test_cli.py | 31 ++++++++++ tests/test_parser.py | 141 +++++++++++++++++++++++++++++++++++++++++-- 4 files changed, 286 insertions(+), 36 deletions(-) diff --git a/src/cmos/cli.py b/src/cmos/cli.py index cf7fed5..e03dfcb 100644 --- a/src/cmos/cli.py +++ b/src/cmos/cli.py @@ -23,7 +23,7 @@ from typing import Callable from cmos.formatter import format_bibliography_entry from cmos.note_formatter import format_note_entry -from cmos.parser import find_notes, split_bibliography +from cmos.parser import find_notes, find_numbered_notes, split_bibliography FormatterFn = Callable[[str], str] @@ -75,31 +75,46 @@ def reformat_notes( formatter: FormatterFn | None = None, concurrency: int = DEFAULT_CONCURRENCY, ) -> str: - """Rewrite pandoc-style markdown footnote definitions in ``text``. + """Rewrite footnote definitions in ``text`` to CMOS 18 note form. - Finds every ``[^marker]: definition text`` line via - ``cmos.parser.find_notes``, formats each definition's text via - ``formatter`` (the v2 note formatter by default), and substitutes - the reformatted text back into the same line position. All - non-definition lines are preserved byte-for-byte. + Auto-detects the source format by trying ``cmos.parser.find_notes`` + (pandoc-style ``[^marker]: text``) first and falling back to + ``cmos.parser.find_numbered_notes`` (``[N] text``) if no pandoc + definitions were found. Both formats are common in real drafts: + pandoc is used by writers who author directly in markdown, while + the numbered form is what docx-to-text conversion typically + produces from footnoted Word documents. + + Each definition's body is formatted via ``formatter`` (the v2 + note formatter by default) and substituted back into the same + line position using the definition's ``original_prefix`` — so + pandoc input round-trips as pandoc output and numbered input + round-trips as numbered output. All non-definition lines are + preserved byte-for-byte. Definitions are reformatted concurrently using a thread pool of size ``concurrency``, mirroring v1's ``reformat_draft``. The output order is deterministic regardless of which API call finishes first. - If the document contains no footnote definitions, returns ``text`` - unchanged — a draft with no notes is a valid (if uninteresting) - input rather than an error. + If the document contains no footnote definitions in either format, + returns ``text`` unchanged — a draft with no notes is a valid (if + uninteresting) input rather than an error. - Scope of v2 chunk 2b (this version): single-line definitions only. - Multi-line definitions (where the body continues on indented - subsequent lines) are out of scope — only the first line is - reformatted, leaving any continuation lines untouched. Real drafts - that use multi-line definitions will need a future parser - extension before this CLI can handle them safely. + Scope: single-line definitions only. Multi-line definitions (where + the body continues on indented subsequent lines for pandoc, or on + unprefixed continuation lines for numbered) are out of scope — + only the first line is reformatted, leaving any continuation lines + untouched. """ fmt = formatter or format_note_entry + + # Try pandoc format first; fall back to numbered format if nothing + # was found. The two parsers are mutually exclusive on well-formed + # input (the regex anchors prevent overlap), so this fallback is + # unambiguous in practice. parsed = find_notes(text) + if not parsed.definitions: + parsed = find_numbered_notes(text) if not parsed.definitions: return text @@ -113,7 +128,7 @@ def reformat_notes( lines = text.splitlines() for definition, formatted in zip(parsed.definitions, rewritten): - lines[definition.line_number] = f"[^{definition.marker}]: {formatted}" + lines[definition.line_number] = f"{definition.original_prefix}{formatted}" # Preserve trailing newline if the original had one — splitlines drops it. trailing = "\n" if text.endswith("\n") else "" diff --git a/src/cmos/parser.py b/src/cmos/parser.py index 79571bc..45d45bc 100644 --- a/src/cmos/parser.py +++ b/src/cmos/parser.py @@ -5,18 +5,32 @@ section as running until the next level-2 (``##``) heading or end of file, and return each non-blank line as one entry. Bullet markers (``- ``, ``* ``) at the start of a line are stripped. -v2 scope (notes): ``find_notes`` locates pandoc-style markdown footnote -definitions ``[^marker]: text`` anywhere in the document. Single-line -definitions only — multi-line continuation (indented continuation lines) -is deferred. The reference markers in prose (the ``[^marker]`` references -themselves, without the trailing colon) are NOT collected; only the -definitions need reformatting. +v2 scope (notes): two complementary functions for finding footnote +definitions in two distinct source formats. Both return the same +``NotesParseResult`` shape; the consumer (typically the CLI) can call +one and fall back to the other to auto-detect format. + +- ``find_notes`` finds pandoc-style markdown footnote definitions + ``[^marker]: text``. The marker can be numeric or named. +- ``find_numbered_notes`` finds plain ``[N] text`` definitions where + the marker is digit-only and there is no caret or colon — the format + produced by docx-to-text conversion of footnoted Word docs. + +Each ``NoteDefinition`` carries the marker, the body text (with +surrounding whitespace stripped), the line number where it appeared +(0-indexed, for future reassembly), and the literal ``original_prefix`` +string (e.g., ``"[^1]: "`` or ``"[1] "``) so that reassembly can +round-trip the source's marker syntax without the consumer needing to +know which format was matched. + +Single-line definitions only in both formats — multi-line continuation +(indented continuation lines) is deferred. The reference markers in +prose are NOT collected; only the definitions need reformatting. Out of scope in v1: in-prose citation rewriting, multi-line entries, nested sections, alternative heading names (e.g., "Works Cited", "References"). Out of scope in v2 (so far): multi-line note definitions, -shortened-form generation, reassembly of formatted notes back into the -document. +shortened-form generation. """ from __future__ import annotations @@ -38,18 +52,24 @@ class ParseResult: @dataclass class NoteDefinition: - """A single pandoc-style markdown footnote definition. + """A single footnote definition extracted from a draft. - ``marker`` is the identifier between ``[^`` and ``]:`` (e.g., ``"1"`` - or ``"smith2020"``). ``text`` is the note body with surrounding - whitespace stripped. ``line_number`` is the 0-indexed line in the - source document where the definition appeared — currently unused but - captured for future reassembly support. + ``marker`` is the identifier between the brackets — e.g., ``"1"`` or + ``"smith2020"`` for pandoc-style ``[^1]:`` / ``[^smith2020]:``, or + ``"107"`` for the numbered ``[107]`` style. ``text`` is the note + body with surrounding whitespace stripped. ``line_number`` is the + 0-indexed line in the source document where the definition appeared + (used for in-place reassembly). ``original_prefix`` is the literal + text that preceded the body on the source line — typically + ``"[^1]: "`` for pandoc or ``"[1] "`` for numbered. Reassembly + concatenates ``original_prefix`` with the reformatted body, so + consumers do not need to know which format was matched. """ marker: str text: str line_number: int + original_prefix: str = "" @dataclass @@ -61,6 +81,7 @@ _HEADING_RE = re.compile(r"^##\s+bibliography\s*$", re.IGNORECASE) _LEVEL_TWO_RE = re.compile(r"^##\s+\S") _BULLET_RE = re.compile(r"^[-*]\s+") _NOTE_DEF_RE = re.compile(r"^\[\^([^\]]+)\]:\s*(.*?)\s*$") +_NUMBERED_NOTE_RE = re.compile(r"^\[(\d+)\]\s+(.*?)\s*$") def split_bibliography(text: str) -> ParseResult: @@ -111,6 +132,10 @@ def find_notes(text: str) -> NotesParseResult: appear in the document. If the document contains no footnote definitions, returns an empty list rather than raising — a draft with no notes is a valid (if uninteresting) input. + + Each definition's ``original_prefix`` is set to ``"[^MARKER]: "`` so + that downstream reassembly can write back the formatted body using + the same marker syntax. """ definitions: list[NoteDefinition] = [] for line_number, line in enumerate(text.splitlines()): @@ -120,6 +145,52 @@ def find_notes(text: str) -> NotesParseResult: marker = match.group(1) body = match.group(2) definitions.append( - NoteDefinition(marker=marker, text=body, line_number=line_number) + NoteDefinition( + marker=marker, + text=body, + line_number=line_number, + original_prefix=f"[^{marker}]: ", + ) + ) + return NotesParseResult(definitions=definitions) + + +def find_numbered_notes(text: str) -> NotesParseResult: + """Find ``[N] text`` style footnote definitions in ``text``. + + Targets the format produced by docx-to-text conversion of footnoted + Word documents: a square-bracket numeric marker followed by a + space and the note body, one definition per line. Distinct from + pandoc footnote definitions, which require ``[^marker]:`` syntax + (caret and colon). + + The marker must be ALL DIGITS — markers like ``[Smith 2020]`` or + ``[foo]`` are deliberately not matched, since they could plausibly + be many other things (citation keys, link labels, in-line + references). Pandoc-style ``[^1]:`` is also not matched because of + the leading ``^``. + + Multi-line definitions (where the body wraps onto a continuation + line) are out of scope; only the first line is captured. + + Returns a ``NotesParseResult`` with definitions in source order. + Each definition's ``original_prefix`` is set to ``"[N] "`` so that + downstream reassembly can write back the formatted body using the + same marker syntax. + """ + definitions: list[NoteDefinition] = [] + for line_number, line in enumerate(text.splitlines()): + match = _NUMBERED_NOTE_RE.match(line) + if match is None: + continue + marker = match.group(1) + body = match.group(2) + definitions.append( + NoteDefinition( + marker=marker, + text=body, + line_number=line_number, + original_prefix=f"[{marker}] ", + ) ) return NotesParseResult(definitions=definitions) diff --git a/tests/test_cli.py b/tests/test_cli.py index 4d8f120..09712d9 100644 --- a/tests/test_cli.py +++ b/tests/test_cli.py @@ -195,6 +195,37 @@ def test_reformat_notes_preserves_trailing_newline(): assert not reformat_notes(without_trailing, formatter=_fake_note_formatter).endswith("\n") +def test_reformat_notes_auto_detects_numbered_format(): + """Real drafts converted from docx use [N] text format, not pandoc. + reformat_notes must auto-detect: try pandoc first, fall back to + numbered. Reassembly must use the original prefix (e.g. "[1] ") so + the output round-trips in the source's marker syntax.""" + draft = """\ +## Notes + +[1] yu, charles. interior chinatown. 2020. p 45. +[2] kwon, hyeyoung. inclusion work. 2022, p 1830. +""" + output = reformat_notes(draft, formatter=_fake_note_formatter) + # Numbered prefix preserved, NOT rewritten as pandoc. + assert "[1] NOTE_FORMATTED(yu, charles. interior chinatown. 2020. p 45.)" in output + assert "[2] NOTE_FORMATTED(kwon, hyeyoung. inclusion work. 2022, p 1830.)" in output + # No pandoc-style markers should appear in the output (we did not + # auto-convert numbered to pandoc). + assert "[^1]" not in output + assert "[^2]" not in output + + +def test_reformat_notes_pandoc_input_still_round_trips_as_pandoc(): + """Regression: after the auto-detect change, pandoc-format input + must still produce pandoc-format output. The original_prefix path + must work for both formats.""" + draft = "[^1]: messy.\n" + output = reformat_notes(draft, formatter=_fake_note_formatter) + assert "[^1]: NOTE_FORMATTED(messy.)" in output + assert "[1] " not in output # numbered prefix must NOT appear + + def test_reformat_notes_preserves_order_under_concurrency(): """With concurrent execution the formatter is called on all definitions in parallel; the output must reassemble them in input order regardless diff --git a/tests/test_parser.py b/tests/test_parser.py index c6e1d5f..c00df2d 100644 --- a/tests/test_parser.py +++ b/tests/test_parser.py @@ -5,14 +5,31 @@ the prose before it, the list of entries inside it, and the prose after it. The section ends at the next ``##`` (level-2) heading or end of file. Blank lines and bullet markers (``- `` or ``* ``) are stripped from entries. -v2 (notes) parser scope: find pandoc-style markdown footnote definitions -``[^marker]: text`` anywhere in the document. Single-line definitions only; -multi-line continuation is deferred to a later chunk. +v2 (notes) parser scope: two complementary functions for finding footnote +definitions in two distinct source formats: + +- ``find_notes`` finds pandoc-style ``[^marker]: text`` definitions + (caret-prefixed marker, colon after the closing bracket). +- ``find_numbered_notes`` finds ``[N] text`` definitions (square bracket + numeric marker, no caret, no colon, just a space before the body) — + the format produced by docx-to-text conversion of footnoted Word docs. + +Both functions return ``NotesParseResult`` with a list of ``NoteDefinition`` +records carrying the marker, body text, line number, and the original +prefix string (so reassembly can round-trip the source's marker syntax). + +Single-line definitions only in both formats; multi-line continuation is +deferred. """ import pytest -from cmos.parser import NoBibliographyError, find_notes, split_bibliography +from cmos.parser import ( + NoBibliographyError, + find_notes, + find_numbered_notes, + split_bibliography, +) def test_three_entries_under_bibliography_heading(): @@ -159,3 +176,119 @@ Some prose that cites it.[^1] result = find_notes(draft) assert len(result.definitions) == 1 assert result.definitions[0].text == "A definition before the reference." + + +def test_find_notes_records_original_prefix_for_pandoc_format(): + """Reassembly needs the literal prefix string so the round-trip + preserves the source's marker syntax. For pandoc this is `[^N]: `.""" + draft = "[^1]: text.\n[^smith2020]: text.\n" + result = find_notes(draft) + assert result.definitions[0].original_prefix == "[^1]: " + assert result.definitions[1].original_prefix == "[^smith2020]: " + + +# --- find_numbered_notes ([N] text format, no caret, no colon) ------------- + + +def test_find_numbered_notes_single_definition(): + draft = """\ +Some prose. + +[1] Charles Yu, Interior Chinatown. Pantheon, 2020. p. 45. +""" + result = find_numbered_notes(draft) + assert len(result.definitions) == 1 + assert result.definitions[0].marker == "1" + assert result.definitions[0].text == "Charles Yu, Interior Chinatown. Pantheon, 2020. p. 45." + + +def test_find_numbered_notes_multiple_sequential(): + draft = """\ +[1] First. +[2] Second. +[3] Third. +""" + result = find_numbered_notes(draft) + assert len(result.definitions) == 3 + assert [d.marker for d in result.definitions] == ["1", "2", "3"] + assert [d.text for d in result.definitions] == ["First.", "Second.", "Third."] + + +def test_find_numbered_notes_multi_digit_markers(): + """The Anti-Communist draft has 107 notes, including [107]. Multi-digit + markers must work.""" + draft = "[1] one.\n[42] forty-two.\n[107] one-oh-seven.\n" + result = find_numbered_notes(draft) + assert [d.marker for d in result.definitions] == ["1", "42", "107"] + + +def test_find_numbered_notes_returns_empty_when_no_definitions(): + draft = "Just prose with no numbered notes.\n" + result = find_numbered_notes(draft) + assert result.definitions == [] + + +def test_find_numbered_notes_records_line_number(): + draft = """\ +Line zero. +Line one. + +[1] Note on line three. +""" + result = find_numbered_notes(draft) + assert result.definitions[0].line_number == 3 + + +def test_find_numbered_notes_strips_trailing_whitespace_from_text(): + draft = "[1] Note text with trailing spaces. \n" + result = find_numbered_notes(draft) + assert result.definitions[0].text == "Note text with trailing spaces." + + +def test_find_numbered_notes_ignores_pandoc_format(): + """`[^1]: text` is pandoc format, not numbered. find_numbered_notes + must NOT match it (or the two parsers would step on each other when + used in fallback fashion). The `^` after `[` disqualifies it.""" + draft = "[^1]: pandoc-style note.\n" + result = find_numbered_notes(draft) + assert result.definitions == [] + + +def test_find_numbered_notes_ignores_non_numeric_markers(): + """A marker like [Smith 2020] or [foo] is NOT a numbered footnote + definition — it could be many other things (citation reference, link + label, etc.). Only digit markers count.""" + draft = """\ +[Smith 2020] Some text — looks like an author-date reference. +[foo] some kind of label. +[1] Actual numbered note. +""" + result = find_numbered_notes(draft) + assert len(result.definitions) == 1 + assert result.definitions[0].marker == "1" + + +def test_find_numbered_notes_records_original_prefix(): + """For numbered format the prefix is `[N] ` — square bracket, number, + closing bracket, single space. Reassembly will use this verbatim.""" + draft = "[1] text.\n[107] text.\n" + result = find_numbered_notes(draft) + assert result.definitions[0].original_prefix == "[1] " + assert result.definitions[1].original_prefix == "[107] " + + +def test_find_numbered_notes_handles_anti_communist_draft_format(): + """Smoke test mirroring the actual Anti-Communist Formations draft: + notes follow a `## Notes` heading, separated by `________________` rule, + one definition per line, sequentially numbered.""" + draft = """\ +## Notes +________________ +[1] First reference. https://example.org/path +[2] Second reference, abbreviated. +[3] Rosen, 7. +""" + result = find_numbered_notes(draft) + assert len(result.definitions) == 3 + assert result.definitions[0].text == "First reference. https://example.org/path" + assert result.definitions[2].text == "Rosen, 7."