Initial phase-1 baseline of the karpathy/autoresearch-style loop. The formatter module is the inner-loop artifact; parser and linter are infra. The linter carries a LINTER_VERSION hash (v0.2.0) that will force a re-baseline on any rule change. Components: - harness/diff.py: case-sensitive field-level substring diff - harness/score.py: three-axis scoring (field, linter, canary exact) - src/cmos/linter.py: 9 CMOS 18 structural rules, each Purdue/CMOS cited - src/cmos/parser.py: locate ## Bibliography section, split entries - src/cmos/formatter.py: prompt + OpenAI call with caller injection - src/cmos/cli.py: cmos format path/to/draft.md - scripts/run_loop.py: loop runner with --fake mode for no-API runs - exemplars/: 3 canary seed exemplars (book, journal w/DOI, web), sourced from chicagomanualofstyle.org quick guide Tests: 48 passing. Fake-mode baseline scalar = 0.000 on the 3 seed exemplars (identity caller fails the linter on every rule). This is the floor the real GPT-5 formatter needs to improve from.
54 lines
1.9 KiB
Python
54 lines
1.9 KiB
Python
"""Tests for harness/diff.py — field-level structured diff.
|
|
|
|
The diff checks, for each (field, expected_value) in a canonical record,
|
|
whether the candidate formatted string contains that value as a case-sensitive
|
|
substring. This is intentionally simpler than parsing — the linter catches
|
|
structural issues; the diff catches "did the right facts survive?"
|
|
"""
|
|
|
|
from harness.diff import field_diff
|
|
|
|
|
|
def test_all_fields_present_in_candidate_match():
|
|
canonical = {
|
|
"author": "Smith, Jane",
|
|
"title": "The History of Nothing",
|
|
"publisher": "University of Chicago Press",
|
|
"year": 2023,
|
|
}
|
|
candidate = "Smith, Jane. *The History of Nothing*. University of Chicago Press, 2023."
|
|
result = field_diff(candidate, canonical)
|
|
assert result.matched == ["author", "title", "publisher", "year"]
|
|
assert result.missing == []
|
|
assert result.rate == 1.0
|
|
|
|
|
|
def test_missing_publisher_is_reported():
|
|
canonical = {
|
|
"author": "Smith, Jane",
|
|
"title": "The History of Nothing",
|
|
"publisher": "University of Chicago Press",
|
|
"year": 2023,
|
|
}
|
|
# Publisher dropped.
|
|
candidate = "Smith, Jane. *The History of Nothing*. 2023."
|
|
result = field_diff(candidate, canonical)
|
|
assert "publisher" in result.missing
|
|
assert "author" in result.matched
|
|
assert result.rate == 0.75
|
|
|
|
|
|
def test_title_case_mismatch_counts_as_missing():
|
|
# CMOS 18 requires headline-style caps. If the formatter emits sentence
|
|
# case, the title should not match its canonical headline-case form.
|
|
canonical = {"title": "The History of Nothing"}
|
|
candidate = "Smith, Jane. *The history of nothing*. 2023."
|
|
result = field_diff(candidate, canonical)
|
|
assert result.missing == ["title"]
|
|
assert result.rate == 0.0
|
|
|
|
|
|
def test_empty_canonical_is_rate_one():
|
|
# Degenerate case — no fields to check means nothing is wrong.
|
|
assert field_diff("whatever", {}).rate == 1.0
|