Files
othmanadi--planning-with-files/tests/test_resolver_parity.py
T
OthmanAdi b49ea488fb feat: make the Stop hook and session recovery fire everywhere, add long-run injection controls
The Stop scalar never dispatched on macOS or Linux (the PowerShell branch
was always selected because check-complete.ps1 ships on every platform)
and was a silent no-op anywhere CLAUDE_SKILL_DIR was unset (the :- install
path fallback can never substitute because the probed variable is always a
non-empty string). Both the legacy completion advisory and the v3
completion gate were dead in those environments. The scalar now selects
targets by file existence and dispatches by platform, PowerShell only
under MINGW/MSYS/CYGWIN, sh elsewhere. Windows output is unchanged.
Patched in all 14 SKILL.md variants that carry the scalar.

session-catchup probed a ~/.claude/projects name that Claude Code never
writes for POSIX paths (leading dash stripped) or paths containing an
underscore (replaced with a dash), so recovery after /clear silently
found nothing there. The mapper now probes the exact spelling first,
keeps both legacy spellings for stores created by older versions, and
settles ambiguity via the cwd recorded in the newest session file.
Propagated to every shipped copy including the older-generation root,
.hermes, and .mastracode scripts.

Also: the attestation SHA cache is keyed on the absolute plan path so two
projects can no longer share a slot and report a false PLAN TAMPERED;
resolve-plan-dir.ps1 reaches parity with the sh resolver (slug filter,
task_plan.md requirement in the newest-dir scan, fail-closed containment);
ledger-append.sh no longer truncates summaries mid-codepoint (UTF-8-safe
trim via iconv with a pure-sh fallback).

New long-run features, all opt-in or additive: structure-aware injection
(PWF_INJECT=smart or an inject-smart .mode token) keeps the active phase,
phase counts, and the last three decisions in the window late in long
plans; a Next Step section in both plan templates plus a sixth reboot
question; session-catchup annotates tool results (ok or FAILED with the
first error line) instead of only listing attempts.

Infrastructure: macos-latest joins the CI matrix, a BSD-userland
simulation harness runs the script fleet without realpath, readlink,
flock, or sha256sum on the Linux leg, and .gitattributes pins LF for
scripts. SEO surfaces: llms.txt rewritten as a Q&A page, three
problem-query docs pages, plugin.json and CITATION.cff keyword hygiene.

Suite grows from 217 to 301 passed, 11 skipped, on Windows and both
existing CI legs; default hook output stays byte-identical to v2.43
(legacy invariant proven by the existing invariant tests).
2026-07-21 21:17:19 +02:00

166 lines
5.8 KiB
Python

"""resolve-plan-dir.sh vs resolve-plan-dir.ps1 parity (v3.8.0).
The ps1 mirror lagged the sh resolver: no slug validation on any branch, no
task_plan.md requirement in the newest-dir scan (a sessions/ dir could win),
and containment failed OPEN on canonicalization failure. These tests run both
resolvers over the same fixture trees and assert identical resolution. pwsh
ships on both ubuntu and windows CI runners, so the parity holds on both legs.
"""
from __future__ import annotations
import os
import shutil
import subprocess
import time
import unittest
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[1]
SH_RESOLVER = REPO_ROOT / "skills" / "planning-with-files" / "scripts" / "resolve-plan-dir.sh"
PS1_RESOLVER = REPO_ROOT / "skills" / "planning-with-files" / "scripts" / "resolve-plan-dir.ps1"
def pwsh_exe() -> str | None:
return shutil.which("pwsh") or shutil.which("powershell")
def have_sh() -> bool:
return shutil.which("sh") is not None
def run_sh(cwd: Path, env_extra: dict | None = None) -> str:
env = os.environ.copy()
env.pop("PLAN_ID", None)
if env_extra:
env.update(env_extra)
result = subprocess.run(
["sh", str(SH_RESOLVER)],
cwd=str(cwd), env=env, capture_output=True, text=True, timeout=60,
)
return result.stdout.strip()
def run_ps1(cwd: Path, env_extra: dict | None = None) -> str:
env = os.environ.copy()
env.pop("PLAN_ID", None)
if env_extra:
env.update(env_extra)
exe = pwsh_exe()
assert exe is not None
result = subprocess.run(
[exe, "-NoProfile", "-ExecutionPolicy", "Bypass",
"-File", str(PS1_RESOLVER)],
cwd=str(cwd), env=env, capture_output=True, text=True, timeout=120,
)
return result.stdout.strip()
def canon(cwd: Path, out: str) -> str | None:
"""Both resolvers may emit relative or absolute paths; compare resolved.
On Windows the sh resolver emits Git Bash POSIX spellings (/tmp/...,
/c/...); cygpath translates them to the Windows form before comparison.
"""
if not out:
return None
if os.name == "nt" and out.startswith("/"):
cygpath = shutil.which("cygpath")
if cygpath:
translated = subprocess.run(
[cygpath, "-w", out], capture_output=True, text=True, timeout=30
).stdout.strip()
if translated:
out = translated
p = Path(out)
if not p.is_absolute():
p = cwd / p
try:
return str(p.resolve()).lower()
except OSError:
return str(p).lower()
@unittest.skipUnless(have_sh(), "requires a POSIX sh")
@unittest.skipUnless(pwsh_exe(), "requires PowerShell")
class ResolverParityTests(unittest.TestCase):
def setUp(self):
import tempfile
self.tempdir = tempfile.TemporaryDirectory(prefix="pwf-parity-")
self.tmp = Path(self.tempdir.name)
def tearDown(self):
self.tempdir.cleanup()
def _plan(self, slug: str, mtime_offset: int = 0) -> Path:
d = self.tmp / ".planning" / slug
d.mkdir(parents=True, exist_ok=True)
(d / "task_plan.md").write_text("# plan\n", encoding="utf-8")
if mtime_offset:
t = time.time() + mtime_offset
os.utime(d, (t, t))
return d
def assert_parity(self, env_extra: dict | None = None, expect_slug: str | None = None):
sh_out = canon(self.tmp, run_sh(self.tmp, env_extra))
ps_out = canon(self.tmp, run_ps1(self.tmp, env_extra))
self.assertEqual(sh_out, ps_out, "sh and ps1 resolved differently")
if expect_slug is None:
self.assertIsNone(sh_out)
else:
assert sh_out is not None, "resolver returned nothing"
self.assertTrue(
sh_out.endswith(expect_slug.lower()),
f"expected {expect_slug}, got {sh_out}",
)
def test_plan_id_env_valid(self):
self._plan("2026-07-21-alpha")
self._plan("2026-07-21-beta", mtime_offset=60)
self.assert_parity({"PLAN_ID": "2026-07-21-alpha"}, "2026-07-21-alpha")
def test_plan_id_env_invalid_slug_falls_through(self):
self._plan("2026-07-21-alpha")
evil = self.tmp / ".planning" / ".." / "outside"
evil.mkdir(parents=True, exist_ok=True)
self.assert_parity({"PLAN_ID": "../outside"}, "2026-07-21-alpha")
def test_active_plan_pointer(self):
self._plan("2026-07-21-alpha", mtime_offset=60)
self._plan("2026-07-21-beta")
(self.tmp / ".planning" / ".active_plan").write_text(
"2026-07-21-beta\n", encoding="utf-8"
)
self.assert_parity(None, "2026-07-21-beta")
def test_active_plan_invalid_slug_falls_through_to_newest(self):
self._plan("2026-07-21-alpha", mtime_offset=60)
(self.tmp / ".planning" / ".active_plan").write_text(
"../../etc\n", encoding="utf-8"
)
self.assert_parity(None, "2026-07-21-alpha")
def test_newest_scan_skips_dir_without_task_plan(self):
self._plan("2026-07-21-real")
sessions = self.tmp / ".planning" / "sessions"
sessions.mkdir(parents=True)
(sessions / "log.jsonl").write_text("{}\n", encoding="utf-8")
t = time.time() + 120
os.utime(sessions, (t, t))
self.assert_parity(None, "2026-07-21-real")
def test_newest_scan_skips_hidden_dirs(self):
self._plan("2026-07-21-real")
hidden = self.tmp / ".planning" / ".cache"
hidden.mkdir(parents=True)
(hidden / "task_plan.md").write_text("# not a plan\n", encoding="utf-8")
t = time.time() + 120
os.utime(hidden, (t, t))
self.assert_parity(None, "2026-07-21-real")
def test_no_planning_dir_resolves_empty(self):
self.assert_parity(None, None)
if __name__ == "__main__":
unittest.main()