Files
cmos/tests/test_cli.py
T
cmos dev 2bd1bdf0d4 cli: parallelize formatter calls with thread pool
reformat_draft now uses concurrent.futures.ThreadPoolExecutor with a
default of 8 workers. OpenAI SDK calls are synchronous but network-
bound, so threads release the GIL during I/O and give real speedup.
ThreadPoolExecutor.map preserves input order regardless of completion
order, so output is deterministic.

Empirical: HML draft (134 entries) went from ~25 min serial to 1m55s
with concurrency=12. ~12x speedup; further increases hit OpenAI rate
limits.

CLI gains a --concurrency flag (default 8) for tuning per draft size /
rate limit headroom. concurrency=1 forces serial execution for
debugging. New test asserts that order is preserved when concurrent
calls finish out of input order (uses a sleep-by-index fake formatter).
2026-04-11 00:42:56 -04:00

104 lines
3.0 KiB
Python

"""Tests for src/cmos/cli.py — end-to-end draft reformatting.
The CLI orchestrates parser → formatter → reassemble. Tests inject a fake
formatter so no API calls happen; the goal is to pin the glue logic, not the
LLM behavior.
"""
from pathlib import Path
import pytest
from cmos.cli import reformat_draft
from cmos.parser import NoBibliographyError
def _fake_formatter(messy: str) -> str:
# Deterministic fake: return a cleaned-up marker string so the output is
# recognizable.
return f"FORMATTED({messy.strip()})"
def test_reformat_draft_rewrites_only_bibliography_section():
draft = """\
# My Paper
Some prose with citations.
## Bibliography
yu, charles. interior chinatown. 2020.
kwon, hyeyoung. inclusion work. 2022.
"""
output = reformat_draft(draft, formatter=_fake_formatter)
assert "# My Paper" in output
assert "Some prose with citations." in output
assert "FORMATTED(yu, charles. interior chinatown. 2020.)" in output
assert "FORMATTED(kwon, hyeyoung. inclusion work. 2022.)" in output
# Original messy lines must not also survive.
assert "yu, charles. interior chinatown. 2020." not in output.replace(
"FORMATTED(yu, charles. interior chinatown. 2020.)", ""
)
def test_reformat_draft_preserves_sections_after_bibliography():
draft = """\
## Bibliography
one entry.
## Appendix
appendix text.
"""
output = reformat_draft(draft, formatter=_fake_formatter)
assert "FORMATTED(one entry.)" in output
assert "## Appendix" in output
assert "appendix text." in output
def test_reformat_draft_raises_when_no_bibliography():
with pytest.raises(NoBibliographyError):
reformat_draft("## Intro\n\nno bib here.\n", formatter=_fake_formatter)
def test_reformat_draft_bibliography_heading_preserved():
draft = "## Bibliography\n\nentry.\n"
output = reformat_draft(draft, formatter=_fake_formatter)
assert "## Bibliography" in output
def test_reformat_draft_preserves_order_under_concurrency():
# With concurrent execution the formatter is called on all entries in
# parallel; the CLI must reassemble them in input order regardless of
# which call finishes first. This test uses a fake formatter that
# sleeps based on the entry content so earlier entries finish AFTER
# later ones if order were naively tied to completion.
import time
def slow_fake(messy: str) -> str:
# Entries with lower index sleep longer so they finish last.
n = int(messy.split()[-1])
time.sleep(0.05 * (5 - n))
return f"FORMATTED({messy})"
draft = """\
## Bibliography
entry 0
entry 1
entry 2
entry 3
entry 4
"""
output = reformat_draft(draft, formatter=slow_fake, concurrency=4)
# Entries must appear in input order.
lines = [l for l in output.splitlines() if l.startswith("FORMATTED(")]
assert lines == [
"FORMATTED(entry 0)",
"FORMATTED(entry 1)",
"FORMATTED(entry 2)",
"FORMATTED(entry 3)",
"FORMATTED(entry 4)",
]