"""Tests for src/cmos/note_formatter.py — the v2 inner-loop iterable artifact. Mirrors tests/test_formatter.py in shape and discipline. The note formatter is iterated independently of the bibliography formatter (Path B); these tests pin the stable seams (caller injection, retry loop, prompt smoke checks) and leave LLM output shape testing to the v2 exemplar canary. No real API calls in this file. Tests use injected fake callers. """ from cmos.note_formatter import ( MODEL, SYSTEM_PROMPT_NOTES, build_user_message, format_note_entry, ) def test_default_model_is_gpt5(): # Same default as v1. Override with OPENAI_MODEL=... in .env. assert MODEL == "gpt-5" def test_system_prompt_mentions_cmos_18_note_form(): assert "CMOS" in SYSTEM_PROMPT_NOTES assert "18" in SYSTEM_PROMPT_NOTES assert "NOTE" in SYSTEM_PROMPT_NOTES def test_system_prompt_mentions_no_inversion(): # The single most distinctive difference from bibliography form: notes # use normal-order author names ("First Last"), not inverted. lower = SYSTEM_PROMPT_NOTES.lower() assert "normal" in lower or "not inverted" in lower def test_system_prompt_mentions_comma_separation(): # The second most distinctive difference: commas between elements, # not periods. lower = SYSTEM_PROMPT_NOTES.lower() assert "comma" in lower def test_system_prompt_excludes_leading_number(): # The note number prefix ("1. ", "2. ") is supplied externally; the # formatter must not include it. The prompt must say so. lower = SYSTEM_PROMPT_NOTES.lower() assert "leading number" in lower or "do not include" in lower def test_system_prompt_mentions_no_ibid(): # CMOS 18 deprecates Ibid., same as v1. assert "Ibid" in SYSTEM_PROMPT_NOTES or "ibid" in SYSTEM_PROMPT_NOTES.lower() def test_build_user_message_contains_the_messy_input(): msg = build_user_message("yu, charles. interior chinatown. 2020, 45") assert "yu, charles. interior chinatown. 2020, 45" in msg def test_note_formatter_uses_injected_caller(): """Same caller-injection seam as v1. Tests can substitute a fake without touching the network.""" recorded: dict = {} def fake_caller(system: str, user: str) -> str: recorded["system"] = system recorded["user"] = user return "Charles Yu, *Interior Chinatown* (Pantheon Books, 2020), 45." output = format_note_entry( "yu, charles. interior chinatown. New York: Pantheon Books, 2020, p 45.", caller=fake_caller, ) assert output == "Charles Yu, *Interior Chinatown* (Pantheon Books, 2020), 45." assert recorded["system"] == SYSTEM_PROMPT_NOTES assert "yu, charles" in recorded["user"] def test_note_formatter_strips_whitespace_from_caller_output(): def fake_caller(system: str, user: str) -> str: return " Charles Yu, *Interior Chinatown* (Pantheon Books, 2020), 45. \n" output = format_note_entry("anything", caller=fake_caller) assert output == "Charles Yu, *Interior Chinatown* (Pantheon Books, 2020), 45." def test_note_formatter_retries_on_validator_failure(): """If the runtime validator rejects the first attempt (e.g., italics dropped), the formatter retries. Same retry semantics as v1.""" attempts = [] def flaky_caller(system: str, user: str) -> str: attempts.append(len(attempts) + 1) if len(attempts) == 1: # First attempt: italics dropped — validator should fail. return "Charles Yu, Interior Chinatown (Pantheon Books, 2020), 45." return "Charles Yu, *Interior Chinatown* (Pantheon Books, 2020), 45." output = format_note_entry("messy yu", caller=flaky_caller) assert "*Interior Chinatown*" in output assert len(attempts) == 2 def test_note_formatter_returns_last_attempt_after_max_retries(): """If every attempt fails validation, return the last attempt and don't loop forever. Same as v1 — the failure surfaces via scoring or human review, not via a runtime exception.""" attempts = [] def always_bad(system: str, user: str) -> str: attempts.append(1) return "Charles Yu, Interior Chinatown (Pantheon Books, 2020), 45." output = format_note_entry("messy", caller=always_bad, max_retries=2) assert len(attempts) == 3 # 1 initial + 2 retries assert "*" not in output