diff --git a/cli.py b/cli.py index 3e0f37237544..db6288a38f68 100644 --- a/cli.py +++ b/cli.py @@ -1131,23 +1131,11 @@ def _run_state_db_auto_maintenance(session_db) -> None: def _run_checkpoint_auto_maintenance() -> None: - """Call ``maybe_auto_prune_checkpoints`` per the ``checkpoints:`` config. Never raises.""" - try: - from hermes_cli.config import load_config as _load_full_config - cfg = (_load_full_config().get("checkpoints") or {}) - if not cfg.get("auto_prune", False): - return - from tools.checkpoint_manager import maybe_auto_prune_checkpoints - # delete_orphans stays False: a missing workdir at startup is ambiguous (unmounted - # volume / VPN down); orphans are only reclaimed by `hermes checkpoints prune`. - maybe_auto_prune_checkpoints( - retention_days=int(cfg.get("retention_days", 7)), - min_interval_hours=int(cfg.get("min_interval_hours", 24)), - delete_orphans=False, - max_total_size_mb=int(cfg.get("max_total_size_mb", 500)), - ) - except Exception as exc: - logger.debug("checkpoint auto-maintenance skipped: %s", exc) + """Checkpoint store retention on a daemon thread: its ``git gc`` can block for tens of seconds + on a large store, which used to stall the prompt once a day. ``auto_prune_from_config`` owns the + config gate and the 24h marker and never raises.""" + from tools.checkpoint_manager import auto_prune_from_config + threading.Thread(target=auto_prune_from_config, name="checkpoint-auto-prune", daemon=True).start() _ACCENT_ANSI_DEFAULT = "\033[1;38;2;255;215;0m" # #FFD700 bold fallback diff --git a/gateway/run.py b/gateway/run.py index 67e88e823f1f..989494b0ba81 100644 --- a/gateway/run.py +++ b/gateway/run.py @@ -3621,22 +3621,10 @@ class GatewayRunner( sessions_dir=self.config.sessions_dir) except Exception as exc: logger.debug("state.db auto-maintenance skipped: %s", exc) - - # Stale checkpoint repo cleanup; opt-in via checkpoints.auto_prune, idempotent via .last_prune. - try: - from hermes_cli.config import load_config as _load_full_config - _ckpt_cfg = (_load_full_config().get("checkpoints") or {}) - if _ckpt_cfg.get("auto_prune", False): - from tools.checkpoint_manager import maybe_auto_prune_checkpoints - # delete_orphans never honoured unattended: a missing workdir is ambiguous (deleted vs. - # unmounted share); orphan cleanup is only via explicit `hermes checkpoints prune`. - maybe_auto_prune_checkpoints( - retention_days=int(_ckpt_cfg.get("retention_days", 7)), - min_interval_hours=int(_ckpt_cfg.get("min_interval_hours", 24)), - delete_orphans=False, - max_total_size_mb=int(_ckpt_cfg.get("max_total_size_mb", 500))) - except Exception as exc: - logger.debug("checkpoint auto-maintenance skipped: %s", exc) + # Checkpoint store pruning is a housekeeping chore (``_housekeeping_checkpoint_prune``), not a + # constructor step: its ``git gc`` repacks the whole store (tens of seconds on a GB store) and + # here it ran before the control socket, adapters and the code_sha stamp — so the first + # restart of the day (the ``hermes update`` one) looked hung and failed fleet verification. def _init_registries_and_clocks(self) -> None: """Pairing stores, hook registry, voice modes, background-task set, liveness and idle clocks.""" @@ -4556,6 +4544,14 @@ def _housekeeping_memory_trim() -> None: trim_memory(reason="messaging gateway housekeeping") +def _housekeeping_checkpoint_prune() -> None: + """Checkpoint store retention + size cap on a live timer; ``auto_prune_from_config`` gates on + ``checkpoints.auto_prune`` and the 24h ``.last_prune`` marker. Off the startup path because its + ``git gc`` can block for tens of seconds on a large store.""" + from tools.checkpoint_manager import auto_prune_from_config + auto_prune_from_config() + + def _drain_restart_safe_cron_deliveries(adapters, loop, runner=None) -> None: """Drain each profile's worker queue through its matching live adapters. A credential-less satellite profile (empty adapter map) drains through the primary's adapters routed by its own profile routes.""" @@ -4612,7 +4608,9 @@ def _start_gateway_housekeeping( (60, "Auto-archive tick", _housekeeping_auto_archive), (1, "Deferred FTS retry tick", _housekeeping_deferred_fts_retry), (1, "gateway housekeeping memory trim", _housekeeping_memory_trim), - (1, "MCP config reconcile", _mcp_config_reconciler(runner))] + (1, "MCP config reconcile", _mcp_config_reconciler(runner)), + # Last: a real prune can hold this thread for a while; every other chore of the tick runs first. + (1, "Checkpoint prune tick", _housekeeping_checkpoint_prune)] logger.info("Gateway housekeeping started (interval=%ds)", interval) tick_count = 0 diff --git a/hermes_cli/config_defaults.py b/hermes_cli/config_defaults.py index 9fd80d07aa9a..0c8a6f363ca5 100644 --- a/hermes_cli/config_defaults.py +++ b/hermes_cli/config_defaults.py @@ -459,9 +459,11 @@ DEFAULT_CONFIG = { "max_total_size_mb": 500, # Skip files larger than this (MB) when staging (datasets, model weights). 0 = no filter. "max_file_size_mb": 10, - # Startup sweep (at most once per min_interval_hours): deletes projects whose last_touch is - # older than retention_days, GCs the shared store, enforces max_total_size_mb, deletes - # legacy-* archives older than retention_days. It NEVER deletes orphans (workdir missing on + # Background sweep (CLI helper thread / gateway housekeeping tick, at most once per + # min_interval_hours; never on the startup path — its git gc can block for tens of seconds): + # deletes projects whose last_touch is older than retention_days, GCs the shared store when + # refs moved, enforces max_total_size_mb, deletes legacy-* archives older than retention_days. + # It NEVER deletes orphans (workdir missing on # disk) — a missing workdir may just be an unmounted volume/VPN, and an unattended sweep # must not guess. Orphans: `hermes checkpoints prune` (`--keep-orphans` to skip). "auto_prune": True, diff --git a/tests/gateway/test_checkpoint_prune_housekeeping.py b/tests/gateway/test_checkpoint_prune_housekeeping.py new file mode 100644 index 000000000000..813dc9e0e56c --- /dev/null +++ b/tests/gateway/test_checkpoint_prune_housekeeping.py @@ -0,0 +1,31 @@ +"""Checkpoint store pruning rides the gateway housekeeping tick, not the constructor. + +Its ``git gc`` repacks the whole store (tens of seconds on a GB store); run at construction it +delayed the control socket, adapters and the code_sha stamp, so the first restart of the day (the +``hermes update`` one) looked hung and failed fleet verification. +""" + +import gateway.run as gateway_run + + +class _OneTickStopEvent: + def __init__(self): + self.waited = False + + def is_set(self): + return self.waited + + def wait(self, timeout=None): + self.waited = True + return True + + +def test_gateway_housekeeping_runs_the_checkpoint_prune(monkeypatch): + import tools.checkpoint_manager as cm + + calls = [] + monkeypatch.setattr(cm, "auto_prune_from_config", lambda: calls.append(True) or {"skipped": False}) + + gateway_run._start_gateway_housekeeping(_OneTickStopEvent(), interval=0) + + assert calls == [True] diff --git a/tests/tools/test_checkpoint_manager.py b/tests/tools/test_checkpoint_manager.py index bf041605e348..f36f97e11254 100644 --- a/tests/tools/test_checkpoint_manager.py +++ b/tests/tools/test_checkpoint_manager.py @@ -1050,6 +1050,57 @@ class TestPruneCheckpointsOrphanAllowlist: assert second_repo.exists() +class TestGcOnlyAfterStoreMutation: + """``git gc`` rewrites the whole pack (tens of seconds on a GB store). A checkpoint never runs + it — it rewrites refs and marks the store gc-pending — and the periodic prune runs it only when + a ref actually moved. A store over the cap with every ref at its one-snapshot floor used to gc on + every checkpoint AND on every daily prune, stalling tool calls and gateway startup.""" + + def test_checkpoint_never_gcs_and_hands_the_reclaim_to_prune(self, checkpoint_base, tmp_path, monkeypatch): + import tools.checkpoint_manager as cm + monkeypatch.setattr(cm, "CHECKPOINT_BASE", checkpoint_base) + gc_calls = [] + real_gc = cm._gc_store + monkeypatch.setattr(cm, "_gc_store", lambda store, wd: gc_calls.append(store) or real_gc(store, wd)) + work = tmp_path / "big" + work.mkdir() + (work / "blob.bin").write_bytes(os.urandom(2 * 1024 * 1024)) # incompressible → store > 1 MB cap + m = CheckpointManager(enabled=True, max_snapshots=50, max_total_size_mb=1) + store = checkpoint_base / "store" + + assert m.ensure_checkpoint(str(work), "first") is True + assert gc_calls == [] and not (store / ".gc-pending").exists() # at the floor: nothing to drop + + (work / "blob.bin").write_bytes(os.urandom(2 * 1024 * 1024)) + m.new_turn() + assert m.ensure_checkpoint(str(work), "second") is True + assert gc_calls == [] # the oldest snapshot was dropped, but the repack is not the tool call's + assert (store / ".gc-pending").exists() + + prune_checkpoints(retention_days=30, delete_orphans=False, checkpoint_base=checkpoint_base) + assert len(gc_calls) == 1 and not (store / ".gc-pending").exists() + + def test_prune_gcs_only_when_a_project_was_deleted(self, checkpoint_base, tmp_path, monkeypatch): + import tools.checkpoint_manager as cm + monkeypatch.setattr(cm, "CHECKPOINT_BASE", checkpoint_base) + gc_calls = [] + monkeypatch.setattr(cm, "_gc_store", lambda store, working_dir: gc_calls.append(store)) + work = tmp_path / "proj" + work.mkdir() + (work / "f").write_text("f") + CheckpointManager(enabled=True).ensure_checkpoint(str(work), "seed") + + assert prune_checkpoints(retention_days=30, delete_orphans=False, checkpoint_base=checkpoint_base)["deleted_stale"] == 0 + assert gc_calls == [] + + meta_path = checkpoint_base / "store" / "projects" / f"{_project_hash(str(work))}.json" + meta = json.loads(meta_path.read_text()) + meta["last_touch"] = time.time() - 60 * 86400 + meta_path.write_text(json.dumps(meta)) + assert prune_checkpoints(retention_days=30, delete_orphans=False, checkpoint_base=checkpoint_base)["deleted_stale"] == 1 + assert len(gc_calls) == 1 + + class TestMaybeAutoPruneCheckpoints: def test_prunes_once_then_skips_within_interval(self, tmp_path): base = tmp_path / "checkpoints" diff --git a/tools/checkpoint_manager.py b/tools/checkpoint_manager.py index 6df3db1c7212..7ef771ee42f6 100644 --- a/tools/checkpoint_manager.py +++ b/tools/checkpoint_manager.py @@ -335,11 +335,24 @@ def _rewrite_ref_to(store: Path, working_dir: str, ref: str, commits: List[str]) return True +_GC_PENDING_NAME = ".gc-pending" + + def _gc_store(store: Path, working_dir: str) -> None: - """Reclaim objects unreachable from the (rewritten/deleted) refs.""" + """Reclaim objects unreachable from the (rewritten/deleted) refs. A full repack — tens of + seconds on a GB store — so it belongs to the periodic prune, never to a checkpoint.""" _run_git(["reflog", "expire", "--expire=now", "--all"], store, working_dir) _run_git(["gc", "--prune=now", "--quiet"], store, working_dir, timeout=_GIT_TIMEOUT * 3) _repair_bare_repo_dirs(store) + _unlink_quiet(store / _GC_PENDING_NAME) + + +def _mark_gc_pending(store: Path) -> None: + """Refs were rewritten without a gc; the next ``prune_checkpoints`` reclaims the objects.""" + try: + (store / _GC_PENDING_NAME).touch() + except OSError as exc: + logger.debug("Could not mark checkpoint store gc-pending: %s", exc) def _drop_oldest_commit(store: Path, working_dir: str, ref: str) -> bool: @@ -349,18 +362,25 @@ def _drop_oldest_commit(store: Path, working_dir: str, ref: str) -> bool: return _rewrite_ref_to(store, working_dir, ref, _ref_commits_oldest_first(store, working_dir, ref)[1:]) +def _drop_one_snapshot_round(store: Path, working_dir: str) -> bool: + """Drop the oldest commit of every project ref that has more than one. True when any did.""" + return any([_drop_oldest_commit(store, working_dir, ref) for ref in _list_project_refs(store, working_dir)]) + + def _shrink_store_to_cap(store: Path, working_dir: str, cap_bytes: int) -> bool: - """Round-robin-drop the oldest commit per project ref until the store fits (bounded to 20 - rounds against pathological loops). False when there are no project refs.""" + """Drop the oldest snapshot per project, gc, re-measure, until the store fits (bounded to 20 + rounds). The gc per round is what makes the measurement move: without it the loop used to + drop 20 rounds of history against an unchanging pack size, flattening every ref to one + snapshot. True when at least one commit was dropped (the store was gc'd).""" + dropped = False for _ in range(20): if _dir_size_bytes(store) <= cap_bytes: break - refs = _list_project_refs(store, working_dir) - if not refs: - return False - if not any([_drop_oldest_commit(store, working_dir, ref) for ref in refs]): + if not _drop_one_snapshot_round(store, working_dir): break - return True + dropped = True + _gc_store(store, working_dir) + return dropped def _migrate_legacy_store(base: Path) -> Optional[Path]: @@ -883,24 +903,29 @@ class CheckpointManager: store, working_dir, index_file=index_file, allowed_returncodes={128}) def _prune(self, store: Path, working_dir: str, ref: str) -> None: - """Rewrite the ref to its last ``max_snapshots`` commits and gc (only limiting the - log view, as v1 did, let loose objects accumulate forever).""" + """Rewrite the ref to its last ``max_snapshots`` commits; the periodic prune reclaims the + objects (a gc here would hold the tool call for the whole repack).""" if _ref_commit_count(store, working_dir, ref) <= self.max_snapshots: return commits = _ref_commits_oldest_first(store, working_dir, ref) if _rewrite_ref_to(store, working_dir, ref, commits[-self.max_snapshots:]): - _gc_store(store, working_dir) + _mark_gc_pending(store) def _enforce_size_cap(self, store: Path) -> None: - """Drop oldest checkpoints across ALL projects until under ``max_total_size_mb``.""" + """Over ``max_total_size_mb``: drop ONE round of oldest snapshots and leave the reclaim to the + periodic prune. One round per checkpoint converges over turns; the full loop ran here once + and, measuring an unchanging pack, flattened every project to a single snapshot.""" cap_bytes = self.max_total_size_mb * _MB size = _dir_size_bytes(store) if cap_bytes > 0 else 0 if size <= cap_bytes: return - logger.info("Checkpoint store exceeded %d MB (actual %d MB) — pruning oldest", - self.max_total_size_mb, size // _MB) - if _shrink_store_to_cap(store, str(store.parent), cap_bytes): - _gc_store(store, str(store.parent)) + if _drop_one_snapshot_round(store, str(store.parent)): + _mark_gc_pending(store) + logger.info("Checkpoint store exceeded %d MB (actual %d MB) — dropped the oldest snapshot " + "per project; space is reclaimed by the next prune", self.max_total_size_mb, size // _MB) + else: + logger.debug("Checkpoint store exceeded %d MB (actual %d MB) with every project at one snapshot", + self.max_total_size_mb, size // _MB) def _step_failed(step: str, err: str) -> bool: @@ -1081,11 +1106,14 @@ def prune_checkpoints(retention_days: int = 7, delete_orphans: bool = True, chec _prune_pre_v2_repos(base, cutoff, delete_orphans, orphan_allowlist, result) store = _store_path(base) if _store_has_head(store): + # gc rewrites the whole pack — the entire cost of a prune on a large store — so it runs + # only when a ref moved: deleted here, or rewritten by a checkpoint that left it pending. + deleted_before = result["deleted_orphan"] + result["deleted_stale"] _prune_v2_projects(store, cutoff, delete_orphans, orphan_allowlist, result) - _gc_store(store, str(base)) + if result["deleted_orphan"] + result["deleted_stale"] > deleted_before or (store / _GC_PENDING_NAME).exists(): + _gc_store(store, str(base)) if max_total_size_mb > 0: _shrink_store_to_cap(store, str(base), max_total_size_mb * _MB) - _gc_store(store, str(base)) result["bytes_freed"] = max(result["bytes_freed"], size_before - _dir_size_bytes(base)) return result @@ -1110,12 +1138,14 @@ def maybe_auto_prune_checkpoints(retention_days: int = 7, min_interval_hours: in return out except (OSError, ValueError): pass # corrupt marker — treat as no prior run - result = out["result"] = prune_checkpoints(retention_days=retention_days, delete_orphans=delete_orphans, - checkpoint_base=base, max_total_size_mb=max_total_size_mb) + # Claim the interval before pruning: callers run on a periodic tick, and a prune that + # dies mid-way must cost one skipped day, not a git gc every tick. try: marker.write_text(str(now), encoding="utf-8") except OSError as exc: logger.debug("Could not write checkpoint prune marker: %s", exc) + result = out["result"] = prune_checkpoints(retention_days=retention_days, delete_orphans=delete_orphans, + checkpoint_base=base, max_total_size_mb=max_total_size_mb) total = result["deleted_orphan"] + result["deleted_stale"] if total > 0: @@ -1128,6 +1158,26 @@ def maybe_auto_prune_checkpoints(retention_days: int = 7, min_interval_hours: in return out +def auto_prune_from_config() -> Dict[str, object]: + """``maybe_auto_prune_checkpoints`` driven by the ``checkpoints:`` config section — the one + startup/housekeeping entry point for the CLI and the gateway. ``delete_orphans`` is never + honoured unattended: a missing workdir is ambiguous (deleted vs. unmounted share); orphan + cleanup is only via explicit ``hermes checkpoints prune``. Never raises.""" + try: + from hermes_cli.config import load_config + cfg = load_config().get("checkpoints") or {} + if not cfg.get("auto_prune", False): + return {"skipped": True} + return maybe_auto_prune_checkpoints( + retention_days=int(cfg.get("retention_days", 7)), + min_interval_hours=int(cfg.get("min_interval_hours", 24)), + delete_orphans=False, + max_total_size_mb=int(cfg.get("max_total_size_mb", 500))) + except Exception as exc: + logger.debug("checkpoint auto-maintenance skipped: %s", exc) + return {"skipped": True, "error": str(exc)} + + def store_status(checkpoint_base: Optional[Path] = None) -> Dict: """Summarise the shadow store: ``{"base", "store_size_bytes", "legacy_size_bytes", "total_size_bytes", "project_count", "projects", "pre_v2_projects", "legacy_archives"}``. diff --git a/website/docs/user-guide/checkpoints-and-rollback.md b/website/docs/user-guide/checkpoints-and-rollback.md index 1a14d7c1be33..a0433750fddd 100644 --- a/website/docs/user-guide/checkpoints-and-rollback.md +++ b/website/docs/user-guide/checkpoints-and-rollback.md @@ -94,14 +94,17 @@ checkpoints: max_total_size_mb: 500 # hard cap on total store size; oldest commits dropped max_file_size_mb: 10 # skip any single file larger than this - # Auto-maintenance (on by default): sweep ~/.hermes/checkpoints/ at startup - # and delete project entries whose last_touch is older than retention_days. - # Runs at most once per min_interval_hours, tracked via a .last_prune - # marker. This sweep never deletes "orphan" entries (working directory not - # found) — a missing workdir at startup is ambiguous (deleted project vs. - # an unmounted external volume / network share / VPN not yet up), so - # orphan cleanup is only ever done via the explicit - # `hermes checkpoints prune` command below, with a confirmation prompt. + # Auto-maintenance (on by default): sweep ~/.hermes/checkpoints/ in the + # background — the CLI on a helper thread right after launch, the gateway + # on its housekeeping tick — and delete project entries whose last_touch is + # older than retention_days. Runs at most once per min_interval_hours, + # tracked via a .last_prune marker. It never blocks the prompt or gateway + # startup: the `git gc` that reclaims space can take tens of seconds on a + # large store. This sweep never deletes "orphan" entries (working directory + # not found) — a missing workdir is ambiguous (deleted project vs. an + # unmounted external volume / network share / VPN not yet up), so orphan + # cleanup is only ever done via the explicit `hermes checkpoints prune` + # command below, with a confirmation prompt. auto_prune: true retention_days: 7 min_interval_hours: 24 @@ -235,8 +238,8 @@ Restore just one file from a checkpoint without affecting the rest of the direct - **Directory scope** — Hermes skips overly broad directories (root `/`, home `$HOME`). - **Repository size** — directories with more than 50,000 files are skipped. - **Per-file size cap** — files larger than `max_file_size_mb` (default 10 MB) are excluded from the snapshot. Prevents accidentally swallowing datasets, model weights, or generated media. -- **Total store size cap** — when the store exceeds `max_total_size_mb` (default 500 MB), the oldest commit per project is dropped round-robin until under the cap. -- **Real pruning** — `max_snapshots` is enforced by rewriting the per-project ref and running `git gc --prune=now` afterwards, so loose objects don't accumulate. +- **Total store size cap** — when the store exceeds `max_total_size_mb` (default 500 MB), each checkpoint drops the oldest commit of every project that still has more than one snapshot (one round per checkpoint), and the periodic prune repeats drop → gc → re-measure until the store fits. A project is never reduced below one snapshot, so a store of many large projects can legitimately sit above the cap. +- **Real pruning, off the hot path** — `max_snapshots` and the size cap are enforced by rewriting the per-project ref at checkpoint time (cheap); the store is then marked `.gc-pending` and the periodic prune runs `git gc --prune=now` once, so loose objects don't accumulate and a tool call never waits on a full repack. - **No-change snapshots** — if there are no changes since the last snapshot, the checkpoint is skipped. - **Non-fatal errors** — all errors inside the Checkpoint Manager are logged at debug level; your tools continue to run. @@ -249,6 +252,7 @@ Restore just one file from a checkpoint without affecting the rest of the direct │ ├── refs/hermes/ # per-project branch tip │ ├── indexes/ # per-project git index │ ├── projects/.json # workdir + created_at + last_touch + │ ├── .gc-pending # refs rewritten since the last gc; cleared by the next prune │ └── info/exclude ├── .last_prune # auto-prune idempotency marker └── legacy-/ # archived pre-v2 per-project shadow repos