From b811b5080af3f2bc87943dea3358bb9e45f8ecff Mon Sep 17 00:00:00 2001 From: Matt McKay Date: Fri, 28 Aug 2026 12:29:33 +1000 Subject: [PATCH] Prose-gate unpaired additions and deletions (#14) MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An unpaired line with no target-script content — code, maths, markup or metadata arriving or leaving — now routes to code-or-markup instead of addition/deletion, mirroring what categorise() already does for non-prose paired lines, so the omission taxonomy counts prose only. The total pair count is unchanged (categories shift only), which keeps the regression's pinned corpus pair count green by construction. At programming.zh-cn @ b727528: additions 103 -> 30, deletions 80 -> 38, code-or-markup 25 -> 140, total 255 unchanged; composition and engine strata byte-identical. The fixture gains one purely additive and one purely deletive editor-shaped commit so the routing is asserted without depending on hunk pairing. Co-Authored-By: Claude Fable 5 --- CHANGELOG.md | 4 ++++ docs/method.md | 2 +- src/textstrata/scan.py | 8 ++++++-- tests/test_baseline.py | 29 ++++++++++++++++++++++++++++- 4 files changed, 39 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index 82d32f2..098cb77 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -23,3 +23,7 @@ stamped with `engine_model`/`engine_tool_version` from the state-file history (the record written by or immediately after the commit), `run.json` carries per-version totals (`engine_strata`), and `overwrites.json` entries are stamped the same way; syncs with no record read `unrecorded`. +- Unpaired additions and deletions are prose-gated ([#14](https://github.com/QuantEcon/textstrata/issues/14)): + an unpaired line with no target-script content routes to `code-or-markup` instead of + `addition`/`deletion`, mirroring the paired-line rule, so the omission taxonomy counts prose only. + The total pair count is unchanged — only categories shift. diff --git a/docs/method.md b/docs/method.md index 1340441..8972642 100644 --- a/docs/method.md +++ b/docs/method.md @@ -66,7 +66,7 @@ stay distinguishable. - **Squash merges** hide human work done inside a machine-drafted PR. `ai-initial` means *as landed*; human effort is a lower bound. - **Last-toucher blame** credits a whole line to whoever changed one character of it. Human shares are an upper bound at line granularity; churn is therefore also reported in changed characters (`chars_changed` per pair, `prose_chars_added`/`prose_chars_deleted` per commit, `prose_char_churn_by_tier` per document), where a one-character fix counts as one character. Counts come from `SequenceMatcher` opcodes over the paired lines' raw text; unpaired additions and deletions count the full line. -- **Line pairing** inside rewritten paragraphs is heuristic (similarity-matched within a hunk). Category counts are indicative. +- **Line pairing** inside rewritten paragraphs is heuristic (similarity-matched within a hunk). Category counts are indicative. Unpaired lines with no target-script content — code, maths, markup, metadata arriving or leaving — are categorised `code-or-markup`, not `addition`/`deletion`, so the omission taxonomy counts prose only. - **Identity** is resolved by e-mail and GitHub noreply handle only; display names are ignored. Unresolved authors fall to `ai-assisted` and should be reviewed in `commits.jsonl`. - **Pre-engine history** has no recorded engine version and lands in the `unrecorded` stratum; do not read its rates as the shipping engine's. diff --git a/src/textstrata/scan.py b/src/textstrata/scan.py index 80e4f3e..0d81f77 100644 --- a/src/textstrata/scan.py +++ b/src/textstrata/scan.py @@ -318,18 +318,22 @@ def tier_for(f: str, sha: str) -> str: "before": o, "after": n}) if tier in HUMAN_TIERS and cat in ("terminology", "punctuation-width", "fluency"): mine_substitutions(o, n, prose, subs, sub_examples) + # an unpaired line with no target-script content is code, math or + # markup arriving or leaving, not an omission-taxonomy edit for n in adds: if n.strip(): + cat = "addition" if prose.is_prose(n) else "code-or-markup" pairs_out.append({"document": f, "sha": c.sha[:8], "date": c.date[:10], "tier": tier, "pr": pr.group(1) if pr else None, - "category": "addition", "taxonomy": "omission", + "category": cat, "taxonomy": CATEGORY_MAP[cat], "similarity": 0.0, "chars_changed": len(n), "before": "", "after": n}) for _ln, o in dels: if o.strip(): + cat = "deletion" if prose.is_prose(o) else "code-or-markup" pairs_out.append({"document": f, "sha": c.sha[:8], "date": c.date[:10], "tier": tier, "pr": pr.group(1) if pr else None, - "category": "deletion", "taxonomy": "omission", + "category": cat, "taxonomy": CATEGORY_MAP[cat], "similarity": 0.0, "chars_changed": len(o), "before": o, "after": ""}) d.prose_churn_by_tier = dict(churn) diff --git a/tests/test_baseline.py b/tests/test_baseline.py index f98ed7e..62fdf2f 100644 --- a/tests/test_baseline.py +++ b/tests/test_baseline.py @@ -58,10 +58,20 @@ def engine_repo(tmp_path): (repo / "lectures" / "c.md").write_text("# 讲座丙\n\n没有状态文件的文稿。\n", encoding="utf-8") shas["c_doc"] = commit(repo, "Add c.md by hand", "2026-05-01T09:00:00Z") # a later sync touches a.md and its state file: must not move a.md's moment - (repo / "lectures" / "a.md").write_text("# 讲座甲\n\n这是机器重新同步的稿子。\n", encoding="utf-8") + (repo / "lectures" / "a.md").write_text( + "# 讲座甲\n\n这是机器重新同步的稿子。\n\ny = 2\n", encoding="utf-8") (repo / ".translate" / "state" / "a.md.yml").write_text( "mode: UPDATE\nmodel: alpha-2\ntool-version: 0.2.0\n", encoding="utf-8") shas["sync"] = commit(repo, "[translation-sync] resync a.md", "2026-06-01T09:00:00Z") + # two editor-shaped commits (non-bot, non-sync, off-roster -> ai-assisted), + # one purely additive and one purely deletive, so every line is unpaired: + # a prose line and a code line arrive, then a code line leaves + (repo / "lectures" / "a.md").write_text( + "# 讲座甲\n\n这是机器重新同步的稿子。\n\ny = 2\n\n新增的一段人写文字。\n\nx = 1\n", encoding="utf-8") + shas["edit"] = commit(repo, "Edit a.md by hand", "2026-07-01T09:00:00Z") + (repo / "lectures" / "a.md").write_text( + "# 讲座甲\n\n这是机器重新同步的稿子。\n\n新增的一段人写文字。\n\nx = 1\n", encoding="utf-8") + shas["edit2"] = commit(repo, "Drop the scratch code line", "2026-07-02T09:00:00Z") return repo, shas @@ -148,3 +158,20 @@ def test_engine_strata_in_scan(engine_repo, tmp_path): # per-version totals, ordered by first appearance assert [(s["model"], s["tool_version"], s["sync_commits"]) for s in run["engine_strata"]] == [ ("beta-1", "0.1.5", 1), ("alpha-2", "0.2.0", 1)] + + +def test_unpaired_lines_prose_gated(engine_repo, tmp_path): + repo, shas = engine_repo + cfg = make_cfg(repo) + cfg.machine.sync = [r"\[translation-sync\]"] + with open(os.devnull, "w") as devnull: + scan(cfg, tmp_path / "out", log=devnull) + pairs = [json.loads(ln) for ln in (tmp_path / "out" / "pairs.jsonl").open(encoding="utf-8")] + mine = [p for p in pairs if p["sha"] in (shas["edit"][:8], shas["edit2"][:8])] + got = {(p["category"], p["after"] or p["before"]) for p in mine} + # a prose line arriving is an addition; code arriving or leaving is code-or-markup + assert ("addition", "新增的一段人写文字。") in got + assert ("code-or-markup", "x = 1") in got + assert ("code-or-markup", "y = 2") in got + assert all(p["taxonomy"] == "formatting" for p in mine if p["category"] == "code-or-markup") + assert not any(p["category"] == "deletion" for p in mine)