mirror of
https://github.com/NousResearch/hermes-agent.git
synced 2026-09-28 06:45:17 +08:00
fix: checkpoint store gc never runs inside a tool call or gateway startup
Symptom: `hermes update` sat for ~40s after "Refreshing cua-driver" and ended with "Fleet version check returned no rows" (exit 1); the restarted gateway took 26s to reach "Starting Hermes Gateway" instead of the usual 3s. The gateway constructor was running `maybe_auto_prune_checkpoints` synchronously, before the control socket, adapters and the code_sha stamp, and on a 1.2 GB store its `git gc --prune=now` (a full repack) takes 20-28s — twice, because the size-cap shrink gc'd again even when it could drop nothing. The same defect sat on the tool-call path: `CheckpointManager._take` ran `_enforce_size_cap`, whose `_shrink_store_to_cap` returned True without dropping anything and triggered a 20-28s gc on the first file-mutating tool call of every turn once the store was over the cap. That loop also re-measured a pack size that cannot move without a gc, so a single over-cap checkpoint dropped 20 rounds of history and flattened every project to one snapshot. - `_take` never gcs: `_prune` and `_enforce_size_cap` rewrite refs (cheap), drop at most one snapshot round, and mark the store `.gc-pending`. - `prune_checkpoints` gcs only when a ref moved (project deleted, or the pending marker), and its cap loop is drop -> gc -> re-measure. - `maybe_auto_prune_checkpoints` claims the interval marker before the run so a failing prune costs one day, not a gc per housekeeping tick. - `auto_prune_from_config` is the one config-driven entry point; the gateway calls it from the housekeeping tick (last chore), the CLI from a daemon thread. Nothing on either startup path waits for git. Live A/B on a copy of a real 1.2 GB / 224-ref store: checkpoint 20.5s -> 1.2-1.6s (0 inline gc); the single repack (19.6s) now runs in the prune.
This commit is contained in:
@@ -1131,23 +1131,11 @@ def _run_state_db_auto_maintenance(session_db) -> None:
|
||||
|
||||
|
||||
def _run_checkpoint_auto_maintenance() -> None:
|
||||
"""Call ``maybe_auto_prune_checkpoints`` per the ``checkpoints:`` config. Never raises."""
|
||||
try:
|
||||
from hermes_cli.config import load_config as _load_full_config
|
||||
cfg = (_load_full_config().get("checkpoints") or {})
|
||||
if not cfg.get("auto_prune", False):
|
||||
return
|
||||
from tools.checkpoint_manager import maybe_auto_prune_checkpoints
|
||||
# delete_orphans stays False: a missing workdir at startup is ambiguous (unmounted
|
||||
# volume / VPN down); orphans are only reclaimed by `hermes checkpoints prune`.
|
||||
maybe_auto_prune_checkpoints(
|
||||
retention_days=int(cfg.get("retention_days", 7)),
|
||||
min_interval_hours=int(cfg.get("min_interval_hours", 24)),
|
||||
delete_orphans=False,
|
||||
max_total_size_mb=int(cfg.get("max_total_size_mb", 500)),
|
||||
)
|
||||
except Exception as exc:
|
||||
logger.debug("checkpoint auto-maintenance skipped: %s", exc)
|
||||
"""Checkpoint store retention on a daemon thread: its ``git gc`` can block for tens of seconds
|
||||
on a large store, which used to stall the prompt once a day. ``auto_prune_from_config`` owns the
|
||||
config gate and the 24h marker and never raises."""
|
||||
from tools.checkpoint_manager import auto_prune_from_config
|
||||
threading.Thread(target=auto_prune_from_config, name="checkpoint-auto-prune", daemon=True).start()
|
||||
|
||||
|
||||
_ACCENT_ANSI_DEFAULT = "\033[1;38;2;255;215;0m" # #FFD700 bold fallback
|
||||
|
||||
+15
-17
@@ -3621,22 +3621,10 @@ class GatewayRunner(
|
||||
sessions_dir=self.config.sessions_dir)
|
||||
except Exception as exc:
|
||||
logger.debug("state.db auto-maintenance skipped: %s", exc)
|
||||
|
||||
# Stale checkpoint repo cleanup; opt-in via checkpoints.auto_prune, idempotent via .last_prune.
|
||||
try:
|
||||
from hermes_cli.config import load_config as _load_full_config
|
||||
_ckpt_cfg = (_load_full_config().get("checkpoints") or {})
|
||||
if _ckpt_cfg.get("auto_prune", False):
|
||||
from tools.checkpoint_manager import maybe_auto_prune_checkpoints
|
||||
# delete_orphans never honoured unattended: a missing workdir is ambiguous (deleted vs.
|
||||
# unmounted share); orphan cleanup is only via explicit `hermes checkpoints prune`.
|
||||
maybe_auto_prune_checkpoints(
|
||||
retention_days=int(_ckpt_cfg.get("retention_days", 7)),
|
||||
min_interval_hours=int(_ckpt_cfg.get("min_interval_hours", 24)),
|
||||
delete_orphans=False,
|
||||
max_total_size_mb=int(_ckpt_cfg.get("max_total_size_mb", 500)))
|
||||
except Exception as exc:
|
||||
logger.debug("checkpoint auto-maintenance skipped: %s", exc)
|
||||
# Checkpoint store pruning is a housekeeping chore (``_housekeeping_checkpoint_prune``), not a
|
||||
# constructor step: its ``git gc`` repacks the whole store (tens of seconds on a GB store) and
|
||||
# here it ran before the control socket, adapters and the code_sha stamp — so the first
|
||||
# restart of the day (the ``hermes update`` one) looked hung and failed fleet verification.
|
||||
|
||||
def _init_registries_and_clocks(self) -> None:
|
||||
"""Pairing stores, hook registry, voice modes, background-task set, liveness and idle clocks."""
|
||||
@@ -4556,6 +4544,14 @@ def _housekeeping_memory_trim() -> None:
|
||||
trim_memory(reason="messaging gateway housekeeping")
|
||||
|
||||
|
||||
def _housekeeping_checkpoint_prune() -> None:
|
||||
"""Checkpoint store retention + size cap on a live timer; ``auto_prune_from_config`` gates on
|
||||
``checkpoints.auto_prune`` and the 24h ``.last_prune`` marker. Off the startup path because its
|
||||
``git gc`` can block for tens of seconds on a large store."""
|
||||
from tools.checkpoint_manager import auto_prune_from_config
|
||||
auto_prune_from_config()
|
||||
|
||||
|
||||
def _drain_restart_safe_cron_deliveries(adapters, loop, runner=None) -> None:
|
||||
"""Drain each profile's worker queue through its matching live adapters. A credential-less satellite
|
||||
profile (empty adapter map) drains through the primary's adapters routed by its own profile routes."""
|
||||
@@ -4612,7 +4608,9 @@ def _start_gateway_housekeeping(
|
||||
(60, "Auto-archive tick", _housekeeping_auto_archive),
|
||||
(1, "Deferred FTS retry tick", _housekeeping_deferred_fts_retry),
|
||||
(1, "gateway housekeeping memory trim", _housekeeping_memory_trim),
|
||||
(1, "MCP config reconcile", _mcp_config_reconciler(runner))]
|
||||
(1, "MCP config reconcile", _mcp_config_reconciler(runner)),
|
||||
# Last: a real prune can hold this thread for a while; every other chore of the tick runs first.
|
||||
(1, "Checkpoint prune tick", _housekeeping_checkpoint_prune)]
|
||||
|
||||
logger.info("Gateway housekeeping started (interval=%ds)", interval)
|
||||
tick_count = 0
|
||||
|
||||
@@ -459,9 +459,11 @@ DEFAULT_CONFIG = {
|
||||
"max_total_size_mb": 500,
|
||||
# Skip files larger than this (MB) when staging (datasets, model weights). 0 = no filter.
|
||||
"max_file_size_mb": 10,
|
||||
# Startup sweep (at most once per min_interval_hours): deletes projects whose last_touch is
|
||||
# older than retention_days, GCs the shared store, enforces max_total_size_mb, deletes
|
||||
# legacy-* archives older than retention_days. It NEVER deletes orphans (workdir missing on
|
||||
# Background sweep (CLI helper thread / gateway housekeeping tick, at most once per
|
||||
# min_interval_hours; never on the startup path — its git gc can block for tens of seconds):
|
||||
# deletes projects whose last_touch is older than retention_days, GCs the shared store when
|
||||
# refs moved, enforces max_total_size_mb, deletes legacy-* archives older than retention_days.
|
||||
# It NEVER deletes orphans (workdir missing on
|
||||
# disk) — a missing workdir may just be an unmounted volume/VPN, and an unattended sweep
|
||||
# must not guess. Orphans: `hermes checkpoints prune` (`--keep-orphans` to skip).
|
||||
"auto_prune": True,
|
||||
|
||||
@@ -0,0 +1,31 @@
|
||||
"""Checkpoint store pruning rides the gateway housekeeping tick, not the constructor.
|
||||
|
||||
Its ``git gc`` repacks the whole store (tens of seconds on a GB store); run at construction it
|
||||
delayed the control socket, adapters and the code_sha stamp, so the first restart of the day (the
|
||||
``hermes update`` one) looked hung and failed fleet verification.
|
||||
"""
|
||||
|
||||
import gateway.run as gateway_run
|
||||
|
||||
|
||||
class _OneTickStopEvent:
|
||||
def __init__(self):
|
||||
self.waited = False
|
||||
|
||||
def is_set(self):
|
||||
return self.waited
|
||||
|
||||
def wait(self, timeout=None):
|
||||
self.waited = True
|
||||
return True
|
||||
|
||||
|
||||
def test_gateway_housekeeping_runs_the_checkpoint_prune(monkeypatch):
|
||||
import tools.checkpoint_manager as cm
|
||||
|
||||
calls = []
|
||||
monkeypatch.setattr(cm, "auto_prune_from_config", lambda: calls.append(True) or {"skipped": False})
|
||||
|
||||
gateway_run._start_gateway_housekeeping(_OneTickStopEvent(), interval=0)
|
||||
|
||||
assert calls == [True]
|
||||
@@ -1050,6 +1050,57 @@ class TestPruneCheckpointsOrphanAllowlist:
|
||||
assert second_repo.exists()
|
||||
|
||||
|
||||
class TestGcOnlyAfterStoreMutation:
|
||||
"""``git gc`` rewrites the whole pack (tens of seconds on a GB store). A checkpoint never runs
|
||||
it — it rewrites refs and marks the store gc-pending — and the periodic prune runs it only when
|
||||
a ref actually moved. A store over the cap with every ref at its one-snapshot floor used to gc on
|
||||
every checkpoint AND on every daily prune, stalling tool calls and gateway startup."""
|
||||
|
||||
def test_checkpoint_never_gcs_and_hands_the_reclaim_to_prune(self, checkpoint_base, tmp_path, monkeypatch):
|
||||
import tools.checkpoint_manager as cm
|
||||
monkeypatch.setattr(cm, "CHECKPOINT_BASE", checkpoint_base)
|
||||
gc_calls = []
|
||||
real_gc = cm._gc_store
|
||||
monkeypatch.setattr(cm, "_gc_store", lambda store, wd: gc_calls.append(store) or real_gc(store, wd))
|
||||
work = tmp_path / "big"
|
||||
work.mkdir()
|
||||
(work / "blob.bin").write_bytes(os.urandom(2 * 1024 * 1024)) # incompressible → store > 1 MB cap
|
||||
m = CheckpointManager(enabled=True, max_snapshots=50, max_total_size_mb=1)
|
||||
store = checkpoint_base / "store"
|
||||
|
||||
assert m.ensure_checkpoint(str(work), "first") is True
|
||||
assert gc_calls == [] and not (store / ".gc-pending").exists() # at the floor: nothing to drop
|
||||
|
||||
(work / "blob.bin").write_bytes(os.urandom(2 * 1024 * 1024))
|
||||
m.new_turn()
|
||||
assert m.ensure_checkpoint(str(work), "second") is True
|
||||
assert gc_calls == [] # the oldest snapshot was dropped, but the repack is not the tool call's
|
||||
assert (store / ".gc-pending").exists()
|
||||
|
||||
prune_checkpoints(retention_days=30, delete_orphans=False, checkpoint_base=checkpoint_base)
|
||||
assert len(gc_calls) == 1 and not (store / ".gc-pending").exists()
|
||||
|
||||
def test_prune_gcs_only_when_a_project_was_deleted(self, checkpoint_base, tmp_path, monkeypatch):
|
||||
import tools.checkpoint_manager as cm
|
||||
monkeypatch.setattr(cm, "CHECKPOINT_BASE", checkpoint_base)
|
||||
gc_calls = []
|
||||
monkeypatch.setattr(cm, "_gc_store", lambda store, working_dir: gc_calls.append(store))
|
||||
work = tmp_path / "proj"
|
||||
work.mkdir()
|
||||
(work / "f").write_text("f")
|
||||
CheckpointManager(enabled=True).ensure_checkpoint(str(work), "seed")
|
||||
|
||||
assert prune_checkpoints(retention_days=30, delete_orphans=False, checkpoint_base=checkpoint_base)["deleted_stale"] == 0
|
||||
assert gc_calls == []
|
||||
|
||||
meta_path = checkpoint_base / "store" / "projects" / f"{_project_hash(str(work))}.json"
|
||||
meta = json.loads(meta_path.read_text())
|
||||
meta["last_touch"] = time.time() - 60 * 86400
|
||||
meta_path.write_text(json.dumps(meta))
|
||||
assert prune_checkpoints(retention_days=30, delete_orphans=False, checkpoint_base=checkpoint_base)["deleted_stale"] == 1
|
||||
assert len(gc_calls) == 1
|
||||
|
||||
|
||||
class TestMaybeAutoPruneCheckpoints:
|
||||
def test_prunes_once_then_skips_within_interval(self, tmp_path):
|
||||
base = tmp_path / "checkpoints"
|
||||
|
||||
+70
-20
@@ -335,11 +335,24 @@ def _rewrite_ref_to(store: Path, working_dir: str, ref: str, commits: List[str])
|
||||
return True
|
||||
|
||||
|
||||
_GC_PENDING_NAME = ".gc-pending"
|
||||
|
||||
|
||||
def _gc_store(store: Path, working_dir: str) -> None:
|
||||
"""Reclaim objects unreachable from the (rewritten/deleted) refs."""
|
||||
"""Reclaim objects unreachable from the (rewritten/deleted) refs. A full repack — tens of
|
||||
seconds on a GB store — so it belongs to the periodic prune, never to a checkpoint."""
|
||||
_run_git(["reflog", "expire", "--expire=now", "--all"], store, working_dir)
|
||||
_run_git(["gc", "--prune=now", "--quiet"], store, working_dir, timeout=_GIT_TIMEOUT * 3)
|
||||
_repair_bare_repo_dirs(store)
|
||||
_unlink_quiet(store / _GC_PENDING_NAME)
|
||||
|
||||
|
||||
def _mark_gc_pending(store: Path) -> None:
|
||||
"""Refs were rewritten without a gc; the next ``prune_checkpoints`` reclaims the objects."""
|
||||
try:
|
||||
(store / _GC_PENDING_NAME).touch()
|
||||
except OSError as exc:
|
||||
logger.debug("Could not mark checkpoint store gc-pending: %s", exc)
|
||||
|
||||
|
||||
def _drop_oldest_commit(store: Path, working_dir: str, ref: str) -> bool:
|
||||
@@ -349,18 +362,25 @@ def _drop_oldest_commit(store: Path, working_dir: str, ref: str) -> bool:
|
||||
return _rewrite_ref_to(store, working_dir, ref, _ref_commits_oldest_first(store, working_dir, ref)[1:])
|
||||
|
||||
|
||||
def _drop_one_snapshot_round(store: Path, working_dir: str) -> bool:
|
||||
"""Drop the oldest commit of every project ref that has more than one. True when any did."""
|
||||
return any([_drop_oldest_commit(store, working_dir, ref) for ref in _list_project_refs(store, working_dir)])
|
||||
|
||||
|
||||
def _shrink_store_to_cap(store: Path, working_dir: str, cap_bytes: int) -> bool:
|
||||
"""Round-robin-drop the oldest commit per project ref until the store fits (bounded to 20
|
||||
rounds against pathological loops). False when there are no project refs."""
|
||||
"""Drop the oldest snapshot per project, gc, re-measure, until the store fits (bounded to 20
|
||||
rounds). The gc per round is what makes the measurement move: without it the loop used to
|
||||
drop 20 rounds of history against an unchanging pack size, flattening every ref to one
|
||||
snapshot. True when at least one commit was dropped (the store was gc'd)."""
|
||||
dropped = False
|
||||
for _ in range(20):
|
||||
if _dir_size_bytes(store) <= cap_bytes:
|
||||
break
|
||||
refs = _list_project_refs(store, working_dir)
|
||||
if not refs:
|
||||
return False
|
||||
if not any([_drop_oldest_commit(store, working_dir, ref) for ref in refs]):
|
||||
if not _drop_one_snapshot_round(store, working_dir):
|
||||
break
|
||||
return True
|
||||
dropped = True
|
||||
_gc_store(store, working_dir)
|
||||
return dropped
|
||||
|
||||
|
||||
def _migrate_legacy_store(base: Path) -> Optional[Path]:
|
||||
@@ -883,24 +903,29 @@ class CheckpointManager:
|
||||
store, working_dir, index_file=index_file, allowed_returncodes={128})
|
||||
|
||||
def _prune(self, store: Path, working_dir: str, ref: str) -> None:
|
||||
"""Rewrite the ref to its last ``max_snapshots`` commits and gc (only limiting the
|
||||
log view, as v1 did, let loose objects accumulate forever)."""
|
||||
"""Rewrite the ref to its last ``max_snapshots`` commits; the periodic prune reclaims the
|
||||
objects (a gc here would hold the tool call for the whole repack)."""
|
||||
if _ref_commit_count(store, working_dir, ref) <= self.max_snapshots:
|
||||
return
|
||||
commits = _ref_commits_oldest_first(store, working_dir, ref)
|
||||
if _rewrite_ref_to(store, working_dir, ref, commits[-self.max_snapshots:]):
|
||||
_gc_store(store, working_dir)
|
||||
_mark_gc_pending(store)
|
||||
|
||||
def _enforce_size_cap(self, store: Path) -> None:
|
||||
"""Drop oldest checkpoints across ALL projects until under ``max_total_size_mb``."""
|
||||
"""Over ``max_total_size_mb``: drop ONE round of oldest snapshots and leave the reclaim to the
|
||||
periodic prune. One round per checkpoint converges over turns; the full loop ran here once
|
||||
and, measuring an unchanging pack, flattened every project to a single snapshot."""
|
||||
cap_bytes = self.max_total_size_mb * _MB
|
||||
size = _dir_size_bytes(store) if cap_bytes > 0 else 0
|
||||
if size <= cap_bytes:
|
||||
return
|
||||
logger.info("Checkpoint store exceeded %d MB (actual %d MB) — pruning oldest",
|
||||
self.max_total_size_mb, size // _MB)
|
||||
if _shrink_store_to_cap(store, str(store.parent), cap_bytes):
|
||||
_gc_store(store, str(store.parent))
|
||||
if _drop_one_snapshot_round(store, str(store.parent)):
|
||||
_mark_gc_pending(store)
|
||||
logger.info("Checkpoint store exceeded %d MB (actual %d MB) — dropped the oldest snapshot "
|
||||
"per project; space is reclaimed by the next prune", self.max_total_size_mb, size // _MB)
|
||||
else:
|
||||
logger.debug("Checkpoint store exceeded %d MB (actual %d MB) with every project at one snapshot",
|
||||
self.max_total_size_mb, size // _MB)
|
||||
|
||||
|
||||
def _step_failed(step: str, err: str) -> bool:
|
||||
@@ -1081,11 +1106,14 @@ def prune_checkpoints(retention_days: int = 7, delete_orphans: bool = True, chec
|
||||
_prune_pre_v2_repos(base, cutoff, delete_orphans, orphan_allowlist, result)
|
||||
store = _store_path(base)
|
||||
if _store_has_head(store):
|
||||
# gc rewrites the whole pack — the entire cost of a prune on a large store — so it runs
|
||||
# only when a ref moved: deleted here, or rewritten by a checkpoint that left it pending.
|
||||
deleted_before = result["deleted_orphan"] + result["deleted_stale"]
|
||||
_prune_v2_projects(store, cutoff, delete_orphans, orphan_allowlist, result)
|
||||
_gc_store(store, str(base))
|
||||
if result["deleted_orphan"] + result["deleted_stale"] > deleted_before or (store / _GC_PENDING_NAME).exists():
|
||||
_gc_store(store, str(base))
|
||||
if max_total_size_mb > 0:
|
||||
_shrink_store_to_cap(store, str(base), max_total_size_mb * _MB)
|
||||
_gc_store(store, str(base))
|
||||
|
||||
result["bytes_freed"] = max(result["bytes_freed"], size_before - _dir_size_bytes(base))
|
||||
return result
|
||||
@@ -1110,12 +1138,14 @@ def maybe_auto_prune_checkpoints(retention_days: int = 7, min_interval_hours: in
|
||||
return out
|
||||
except (OSError, ValueError):
|
||||
pass # corrupt marker — treat as no prior run
|
||||
result = out["result"] = prune_checkpoints(retention_days=retention_days, delete_orphans=delete_orphans,
|
||||
checkpoint_base=base, max_total_size_mb=max_total_size_mb)
|
||||
# Claim the interval before pruning: callers run on a periodic tick, and a prune that
|
||||
# dies mid-way must cost one skipped day, not a git gc every tick.
|
||||
try:
|
||||
marker.write_text(str(now), encoding="utf-8")
|
||||
except OSError as exc:
|
||||
logger.debug("Could not write checkpoint prune marker: %s", exc)
|
||||
result = out["result"] = prune_checkpoints(retention_days=retention_days, delete_orphans=delete_orphans,
|
||||
checkpoint_base=base, max_total_size_mb=max_total_size_mb)
|
||||
|
||||
total = result["deleted_orphan"] + result["deleted_stale"]
|
||||
if total > 0:
|
||||
@@ -1128,6 +1158,26 @@ def maybe_auto_prune_checkpoints(retention_days: int = 7, min_interval_hours: in
|
||||
return out
|
||||
|
||||
|
||||
def auto_prune_from_config() -> Dict[str, object]:
|
||||
"""``maybe_auto_prune_checkpoints`` driven by the ``checkpoints:`` config section — the one
|
||||
startup/housekeeping entry point for the CLI and the gateway. ``delete_orphans`` is never
|
||||
honoured unattended: a missing workdir is ambiguous (deleted vs. unmounted share); orphan
|
||||
cleanup is only via explicit ``hermes checkpoints prune``. Never raises."""
|
||||
try:
|
||||
from hermes_cli.config import load_config
|
||||
cfg = load_config().get("checkpoints") or {}
|
||||
if not cfg.get("auto_prune", False):
|
||||
return {"skipped": True}
|
||||
return maybe_auto_prune_checkpoints(
|
||||
retention_days=int(cfg.get("retention_days", 7)),
|
||||
min_interval_hours=int(cfg.get("min_interval_hours", 24)),
|
||||
delete_orphans=False,
|
||||
max_total_size_mb=int(cfg.get("max_total_size_mb", 500)))
|
||||
except Exception as exc:
|
||||
logger.debug("checkpoint auto-maintenance skipped: %s", exc)
|
||||
return {"skipped": True, "error": str(exc)}
|
||||
|
||||
|
||||
def store_status(checkpoint_base: Optional[Path] = None) -> Dict:
|
||||
"""Summarise the shadow store: ``{"base", "store_size_bytes", "legacy_size_bytes",
|
||||
"total_size_bytes", "project_count", "projects", "pre_v2_projects", "legacy_archives"}``.
|
||||
|
||||
@@ -94,14 +94,17 @@ checkpoints:
|
||||
max_total_size_mb: 500 # hard cap on total store size; oldest commits dropped
|
||||
max_file_size_mb: 10 # skip any single file larger than this
|
||||
|
||||
# Auto-maintenance (on by default): sweep ~/.hermes/checkpoints/ at startup
|
||||
# and delete project entries whose last_touch is older than retention_days.
|
||||
# Runs at most once per min_interval_hours, tracked via a .last_prune
|
||||
# marker. This sweep never deletes "orphan" entries (working directory not
|
||||
# found) — a missing workdir at startup is ambiguous (deleted project vs.
|
||||
# an unmounted external volume / network share / VPN not yet up), so
|
||||
# orphan cleanup is only ever done via the explicit
|
||||
# `hermes checkpoints prune` command below, with a confirmation prompt.
|
||||
# Auto-maintenance (on by default): sweep ~/.hermes/checkpoints/ in the
|
||||
# background — the CLI on a helper thread right after launch, the gateway
|
||||
# on its housekeeping tick — and delete project entries whose last_touch is
|
||||
# older than retention_days. Runs at most once per min_interval_hours,
|
||||
# tracked via a .last_prune marker. It never blocks the prompt or gateway
|
||||
# startup: the `git gc` that reclaims space can take tens of seconds on a
|
||||
# large store. This sweep never deletes "orphan" entries (working directory
|
||||
# not found) — a missing workdir is ambiguous (deleted project vs. an
|
||||
# unmounted external volume / network share / VPN not yet up), so orphan
|
||||
# cleanup is only ever done via the explicit `hermes checkpoints prune`
|
||||
# command below, with a confirmation prompt.
|
||||
auto_prune: true
|
||||
retention_days: 7
|
||||
min_interval_hours: 24
|
||||
@@ -235,8 +238,8 @@ Restore just one file from a checkpoint without affecting the rest of the direct
|
||||
- **Directory scope** — Hermes skips overly broad directories (root `/`, home `$HOME`).
|
||||
- **Repository size** — directories with more than 50,000 files are skipped.
|
||||
- **Per-file size cap** — files larger than `max_file_size_mb` (default 10 MB) are excluded from the snapshot. Prevents accidentally swallowing datasets, model weights, or generated media.
|
||||
- **Total store size cap** — when the store exceeds `max_total_size_mb` (default 500 MB), the oldest commit per project is dropped round-robin until under the cap.
|
||||
- **Real pruning** — `max_snapshots` is enforced by rewriting the per-project ref and running `git gc --prune=now` afterwards, so loose objects don't accumulate.
|
||||
- **Total store size cap** — when the store exceeds `max_total_size_mb` (default 500 MB), each checkpoint drops the oldest commit of every project that still has more than one snapshot (one round per checkpoint), and the periodic prune repeats drop → gc → re-measure until the store fits. A project is never reduced below one snapshot, so a store of many large projects can legitimately sit above the cap.
|
||||
- **Real pruning, off the hot path** — `max_snapshots` and the size cap are enforced by rewriting the per-project ref at checkpoint time (cheap); the store is then marked `.gc-pending` and the periodic prune runs `git gc --prune=now` once, so loose objects don't accumulate and a tool call never waits on a full repack.
|
||||
- **No-change snapshots** — if there are no changes since the last snapshot, the checkpoint is skipped.
|
||||
- **Non-fatal errors** — all errors inside the Checkpoint Manager are logged at debug level; your tools continue to run.
|
||||
|
||||
@@ -249,6 +252,7 @@ Restore just one file from a checkpoint without affecting the rest of the direct
|
||||
│ ├── refs/hermes/<hash> # per-project branch tip
|
||||
│ ├── indexes/<hash> # per-project git index
|
||||
│ ├── projects/<hash>.json # workdir + created_at + last_touch
|
||||
│ ├── .gc-pending # refs rewritten since the last gc; cleared by the next prune
|
||||
│ └── info/exclude
|
||||
├── .last_prune # auto-prune idempotency marker
|
||||
└── legacy-<ts>/ # archived pre-v2 per-project shadow repos
|
||||
|
||||
Reference in New Issue
Block a user