Merge remote-tracking branch 'origin/main' into tests/prune-low-value

# Conflicts:
#	tests/run_agent/test_conversation_fallback_state.py
This commit is contained in:
Teknium 2026-07-29 15:24:14 -07:00
commit 7c5a98d888
No known key found for this signature in database
31 changed files with 5683 additions and 87 deletions

View File

@ -455,8 +455,7 @@ def init_agent(
command: str = None,
args: list[str] | None = None,
model: str = "",
max_iterations: int = 500, # Default tool-calling iterations (shared with subagents)
tool_delay: float = 1.0,
max_iterations: int = 90, # Default tool-calling iterations (shared with subagents)
enabled_toolsets: List[str] = None,
disabled_toolsets: List[str] = None,
save_trajectories: bool = False,
@ -529,8 +528,7 @@ def init_agent(
requested_provider (str): Original provider identity before runtime canonicalization
api_mode (str): API mode override: "chat_completions" or "codex_responses"
model (str): Model name to use (default: "anthropic/claude-opus-4.6")
max_iterations (int): Maximum number of tool calling iterations (default: 500)
tool_delay (float): Delay between tool calls in seconds (default: 1.0)
max_iterations (int): Maximum number of tool calling iterations (default: 90)
enabled_toolsets (List[str]): Only enable tools from these toolsets (optional)
disabled_toolsets (List[str]): Disable tools from these toolsets (optional)
save_trajectories (bool): Whether to save conversation trajectories to JSONL files (default: False)
@ -576,7 +574,6 @@ def init_agent(
# Shared iteration budget — parent creates, children inherit.
# Consumed by every LLM turn across parent + all subagents.
agent.iteration_budget = iteration_budget or IterationBudget(max_iterations)
agent.tool_delay = tool_delay
agent.save_trajectories = save_trajectories
agent.verbose_logging = verbose_logging
agent.quiet_mode = quiet_mode

View File

@ -19,13 +19,18 @@ from typing import Optional
from agent.runtime_cwd import resolve_agent_cwd
from agent.skill_utils import (
EXCLUDED_SKILL_DIRS,
ORG_ACTIVE_MARKER,
ORG_MIRROR_DIR_NAME,
ORG_PROVENANCE_FILE,
SKILL_SUPPORT_DIRS,
extract_skill_conditions,
extract_skill_description,
get_all_skills_dirs,
get_disabled_skill_names,
iter_skill_index_files,
org_id_of_path,
parse_frontmatter,
read_active_org_id,
skill_matches_environment,
skill_matches_platform,
skill_matches_platform_list,
@ -1340,7 +1345,9 @@ def drain_truncation_warnings() -> list:
_SKILLS_PROMPT_CACHE_MAX = 8
_SKILLS_PROMPT_CACHE: OrderedDict[tuple, str] = OrderedDict()
_SKILLS_PROMPT_CACHE_LOCK = threading.Lock()
_SKILLS_SNAPSHOT_VERSION = 1
# v2: entries gained org provenance fields (org_id/org_author/rel_dir) for M2
# org-shared skills; older snapshots are discarded and rebuilt.
_SKILLS_SNAPSHOT_VERSION = 2
def _skills_prompt_snapshot_path() -> Path:
@ -1359,13 +1366,32 @@ def clear_skills_system_prompt_cache(*, clear_snapshot: bool = False) -> None:
def _build_skills_manifest(skills_dir: Path) -> dict[str, list[int]]:
"""Build an mtime/size manifest of all SKILL.md and DESCRIPTION.md files."""
"""Build an mtime/size manifest of all SKILL.md and DESCRIPTION.md files.
Org mirrors (M2): only the ACTIVE org's mirror participates, and the
``.active_org`` marker itself is included so switching/leaving an org
invalidates the snapshot even when no SKILL.md changed.
"""
manifest: dict[str, list[int]] = {}
skills_dir_str = str(skills_dir)
base = os.path.join(skills_dir_str, "")
prefix_len = len(base)
active_org = read_active_org_id(skills_dir)
org_root = os.path.join(skills_dir_str, ORG_MIRROR_DIR_NAME)
marker_path = os.path.join(org_root, ORG_ACTIVE_MARKER)
try:
st = os.stat(marker_path)
manifest[ORG_MIRROR_DIR_NAME + "/" + ORG_ACTIVE_MARKER] = [
int(st.st_mtime), int(st.st_size),
]
except OSError:
pass
for root, dirs, files in os.walk(skills_dir_str, followlinks=True):
has_skill_md = "SKILL.md" in files
if root == skills_dir_str and ORG_MIRROR_DIR_NAME in dirs and active_org is None:
dirs.remove(ORG_MIRROR_DIR_NAME)
elif root == org_root:
dirs[:] = [d for d in dirs if d == active_org]
dirs[:] = [
d
for d in dirs
@ -1430,6 +1456,15 @@ def _build_snapshot_entry(
"""Build a serialisable metadata dict for one skill."""
rel_path = skill_file.relative_to(skills_dir)
parts = rel_path.parts
# M2 org mirror: strip the `_org/<org_id>/` prefix so category/name derive
# from the path WITHIN the mirror (same shape the org tree was built
# from), and record provenance for labeling + fail-loud collisions.
org_id: str | None = None
if len(parts) >= 3 and parts[0] == ORG_MIRROR_DIR_NAME:
org_id = parts[1]
parts = parts[2:]
if len(parts) >= 2:
skill_name = parts[-2]
category = "/".join(parts[:-2]) if len(parts) > 2 else parts[0]
@ -1441,7 +1476,7 @@ def _build_snapshot_entry(
if isinstance(platforms, str):
platforms = [platforms]
return {
entry = {
"skill_name": skill_name,
"category": category,
"frontmatter_name": str(frontmatter.get("name", skill_name)),
@ -1449,6 +1484,22 @@ def _build_snapshot_entry(
"platforms": [str(p).strip() for p in platforms if str(p).strip()],
"conditions": extract_skill_conditions(frontmatter),
}
if org_id:
entry["org_id"] = org_id
# Author from the pull-time provenance sidecar (token-verified at
# push by the plane's author_mismatch guard). Best-effort.
try:
import json as _json
prov_path = (
skills_dir / ORG_MIRROR_DIR_NAME / org_id / ORG_PROVENANCE_FILE
)
prov = _json.loads(prov_path.read_text(encoding="utf-8"))
device = str(prov.get("author_device") or "")
entry["org_author"] = device or str(prov.get("author_user_id") or "")
except Exception:
entry["org_author"] = ""
return entry
# =========================================================================
@ -1584,6 +1635,10 @@ def build_skills_system_prompt(
skills_by_category: dict[str, list[tuple[str, str]]] = {}
category_descriptions: dict[str, str] = {}
# Unified visible-entry list (both paths) so the org labeling +
# fail-loud collision pass below runs identically for snapshot and scan.
visible_entries: list[dict] = []
skill_entries: list[dict] = []
if snapshot is not None:
# Fast path: use pre-parsed metadata from disk
@ -1591,7 +1646,6 @@ def build_skills_system_prompt(
if not isinstance(entry, dict):
continue
skill_name = entry.get("skill_name") or ""
category = entry.get("category") or "general"
frontmatter_name = entry.get("frontmatter_name") or skill_name
platforms = entry.get("platforms") or []
if not skill_matches_platform_list(platforms):
@ -1604,16 +1658,13 @@ def build_skills_system_prompt(
available_toolsets,
):
continue
skills_by_category.setdefault(category, []).append(
(frontmatter_name, entry.get("description", ""))
)
visible_entries.append(entry)
category_descriptions = {
str(k): str(v)
for k, v in (snapshot.get("category_descriptions") or {}).items()
}
else:
# Cold path: full filesystem scan + write snapshot for next time
skill_entries: list[dict] = []
for skill_file in iter_skill_index_files(skills_dir, "SKILL.md"):
is_compatible, frontmatter, desc = _parse_skill_file(skill_file)
entry = _build_snapshot_entry(skill_file, skills_dir, frontmatter, desc)
@ -1629,10 +1680,38 @@ def build_skills_system_prompt(
available_toolsets,
):
continue
skills_by_category.setdefault(entry["category"], []).append(
(entry["frontmatter_name"], entry["description"])
)
visible_entries.append(entry)
# ── M2 org labeling + FAIL-LOUD collisions ─────────────────────────
# An org skill lists with an explicit provenance tag. When a personal and
# an org skill share a name, NEITHER silently wins: both list qualified
# (personal keeps the bare name is the wrong default — silent divergence
# from the org set; org winning silently shadows the user's own work) —
# so both entries carry a [name collision] flag and skill_view refuses
# the ambiguous bare name (its existing multi-candidate guard).
name_owners: dict[str, set[str]] = {}
for entry in visible_entries:
fm = entry.get("frontmatter_name") or entry.get("skill_name") or ""
kind = "org" if entry.get("org_id") else "personal"
name_owners.setdefault(fm, set()).add(kind)
for entry in visible_entries:
fm = entry.get("frontmatter_name") or entry.get("skill_name") or ""
desc = entry.get("description", "")
org_id = entry.get("org_id")
collided = len(name_owners.get(fm, set())) > 1
if org_id:
author = entry.get("org_author") or ""
tag = f"[org-shared{': by ' + author if author else ''}]"
desc = f"{tag} {desc}".strip()
category = f"org:{org_id}"
else:
category = entry.get("category") or "general"
if collided:
desc = f"[name collision — also exists {'personally' if org_id else 'in your org'}; load via category path] {desc}".strip()
skills_by_category.setdefault(category, []).append((fm, desc))
if snapshot is None:
# (continuation of the cold path below: category descriptions + write)
# Read category-level DESCRIPTION.md files
for desc_file in iter_skill_index_files(skills_dir, "DESCRIPTION.md"):
try:

View File

@ -49,6 +49,55 @@ EXCLUDED_SKILL_DIRS = frozenset(
# archive workflow preserves a complete old skill package under references/.
SKILL_SUPPORT_DIRS = frozenset(("references", "templates", "assets", "scripts"))
# ── Org-shared skills (sync contract) ───────────────────────────
# Org mirrors live under ~/.hermes/skills/_org/<org_id>/. Resolution is
# TOKEN-GATED via a marker file the sync client writes after verifying the
# token (skills_sync_client.pull_org_skills): only the marked org's mirror is
# scanned. No marker ⇒ no org skills load. The marker is plain data (org_id
# string) so this module stays import-light; the VERIFICATION lives in the
# sync client, which is the only writer. Offline grace: the marker persists,
# so already-pulled org skills keep working without connectivity; a VERIFIED
# org change (or personal-org token) rewrites/removes it.
ORG_MIRROR_DIR_NAME = "_org"
ORG_ACTIVE_MARKER = ".active_org"
ORG_PROVENANCE_FILE = ".org-provenance.json"
# Records the fingerprint of each skill exactly as upstream sent it, so a
# later local edit is detectable and an org pull can refuse to clobber it.
ORG_BASELINE_FILE = ".org-baseline.json"
def read_active_org_id(skills_dir: Path) -> Optional[str]:
"""The org id whose mirror may resolve, or None (no org skills load)."""
try:
marker = skills_dir / ORG_MIRROR_DIR_NAME / ORG_ACTIVE_MARKER
if not marker.exists():
return None
val = marker.read_text(encoding="utf-8").strip()
return val or None
except OSError:
return None
def is_org_mirror_path(path, skills_dir: Path) -> bool:
"""True when *path* is inside the org mirror (``_org/``)."""
try:
rel = Path(path).resolve().relative_to(Path(skills_dir).resolve())
except (OSError, ValueError):
return False
return bool(rel.parts) and rel.parts[0] == ORG_MIRROR_DIR_NAME
def org_id_of_path(path, skills_dir: Path) -> Optional[str]:
"""The ``<org_id>`` segment for a path under ``_org/<org_id>/...``."""
try:
rel = Path(path).resolve().relative_to(Path(skills_dir).resolve())
except (OSError, ValueError):
return None
if len(rel.parts) >= 2 and rel.parts[0] == ORG_MIRROR_DIR_NAME:
return rel.parts[1]
return None
def is_excluded_skill_path(path, *, root: Optional[Path] = None) -> bool:
"""True if *path* should be skipped by active skill scanners.
@ -817,11 +866,24 @@ def iter_skill_index_files(skills_dir: Path, filename: str):
scripts) can contain arbitrary markdown and even archived package
``SKILL.md`` files, but they are progressive-disclosure data loaded through
``skill_view(..., file_path=...)`` rather than active skill roots.
M2 org mirrors (``_org/``): TOKEN-GATED resolution. Only the active org's
subdir (per the sync-client-written ``.active_org`` marker) is walked;
every other ``_org/<id>/`` (stale mirror from a previous org, or no
marker at all) is pruned leave an org and its skills stop resolving,
without any manual cleanup.
"""
skills_dir_str = str(skills_dir)
active_org = read_active_org_id(skills_dir)
org_root = os.path.join(skills_dir_str, ORG_MIRROR_DIR_NAME)
matches: list[str] = []
for root, dirs, files in os.walk(skills_dir_str, followlinks=True):
has_skill_md = "SKILL.md" in files
if root == skills_dir_str and ORG_MIRROR_DIR_NAME in dirs and active_org is None:
dirs.remove(ORG_MIRROR_DIR_NAME)
elif root == org_root:
# Inside _org/: descend ONLY into the active org's mirror.
dirs[:] = [d for d in dirs if d == active_org]
dirs[:] = [
d
for d in dirs

View File

@ -1979,9 +1979,6 @@ def execute_tool_calls_sequential(agent, assistant_message, messages: list, effe
return
break
if agent.tool_delay > 0 and i < len(assistant_message.tool_calls):
time.sleep(agent.tool_delay)
# ── Per-turn aggregate budget enforcement ─────────────────────────
num_tools_seq = len(assistant_message.tool_calls)
if finalize and num_tools_seq > 0:

20
cli.py
View File

@ -14650,6 +14650,26 @@ class HermesCLI(CLIAgentSetupMixin, CLICommandsMixin, CLIBillingMixin):
)
except Exception:
pass
# Skill sync — best-effort periodic pull, piggy-backing on the
# curator tick. Inert unless the access gate is open and a sync base
# URL is configured; swallows all errors so it never blocks startup.
try:
from tools.skills_sync_client import maybe_pull_skills
maybe_pull_skills()
except Exception:
pass
# Org-shared skills — pull the organisation's approved set into the
# read-only mirror. Gated on real org membership: resolve_org_identity
# requires an org role on the token, which is only issued for
# multi-member organisations, so a solo account never reaches the
# network here. Fail-quiet, exactly like the personal pull above.
try:
from tools.skills_sync_client import maybe_pull_org_skills
maybe_pull_org_skills()
except Exception:
pass
if self.preloaded_skills and not self._startup_skills_line_shown:
skills_label = ", ".join(self.preloaded_skills)
self._console_print(

View File

@ -711,7 +711,56 @@ per-gateway secret and the same host as `/relay/provision`.
---
## 8. Versioning policy
## 8. Gateway-side platform behavior controls (enterprise)
Enterprise deployments configure fronted-platform behavior on the GATEWAY
side, under `platforms.relay.extra.<platform>` — a supported subset of that
platform's native options. The native platform block (e.g. `platforms.slack`)
is not read on the relay lane; the connector receives the *outcome* of these
controls as frame metadata (§4) and executes mechanically — it holds no
platform behavior policy of its own.
```yaml
platforms:
relay:
extra:
slack:
reply_in_thread: true # default
```
Resolution: nested `extra.<platform>` object wins → legacy flat key on
`extra` honored as fallback → default. Source of truth:
`RelayAdapter._effective_reply_in_thread` (`gateway/relay/adapter.py`).
Values coerce exactly as the native Slack adapter's do — `1/true/yes/on`
(case-insensitive, whitespace-trimmed) are ON, anything else is OFF — so a
YAML-quoted `"false"` turns a knob off rather than being read as a truthy
string.
Current controls (Slack):
| Key | Default | Effect |
| --- | --- | --- |
| `reply_in_thread` | `true` | `true`: thread-per-message — each top-level DM message anchors its own thread (status, progress, prompts, final reply all carry that `metadata.thread_id`). `false`: flat rolling DM — send-lane frames carry NO thread anchor (stripped, not omitted), one shared session per DM. |
| `dm_top_level_threads_as_sessions` | `true` | Native-parity escape hatch (mirrors `platforms.slack.extra.dm_top_level_threads_as_sessions`). `true`: in thread-per-message mode each top-level DM message keys its own session, so concurrent messages run in parallel. `false`: threaded reply placement is kept but the session stamp is skipped — one rolling DM session (legacy steer/queue posture). No effect in flat mode, which always keeps the single rolling session. |
Typing/status frames always carry the triggering-ts anchor when one is known
(liveliness is unconditional, both modes): Slack's status line is
thread-scoped, and in flat mode the send-side anchor strip guarantees the
status anchor can never leak into reply placement. Semantics of the native
key: see `website/docs/user-guide/messaging/slack.md`.
Thread-anchor resolution applies to EVERY send lane — text (`send`) and media
(`send_media`) alike — through one choke point
(`RelayAdapter._apply_slack_thread_anchor`). Media frames egress via the same
connector-side Slack sender, which threads on `metadata.thread_id` only, so an
attachment resolves its anchor identically to a text reply: promoted into
metadata in thread-per-message mode, stripped in flat mode.
Changes take effect on gateway restart; no connector involvement.
---
## 9. Versioning policy
- `contract_version` is an int; bump **only** for additive changes during the
experimental phase (new optional fields, new `op`s).

View File

@ -72,6 +72,20 @@ class RelayAdapter(BasePlatformAdapter):
# recipient's author binding; we re-attach this user_id as
# metadata.user_id on the outbound action so it can. See _capture_scope.
self._dm_user_by_chat: Dict[str, str] = {}
# chat_id -> chat_type (e.g. "dm", "channel", "group") learned from the
# inbound event. Used to reproduce native Slack's synthetic-DM-thread
# suppression on the relay lane: a DM streaming reply carries
# reply_to=<triggering message ts> as its edit anchor, but the connector
# maps a raw reply_to to a Slack thread_ts — so a plain DM reply would be
# threaded UNDER the user's message (and lose progressive edit streaming)
# instead of posting flat at the DM root. Native SlackAdapter drops that
# synthetic reply_to in _resolve_thread_ts; the relay lane needs the same
# disambiguation, and it needs the chat_type to know a chat is a DM.
self._chat_type_by_chat: Dict[str, str] = {}
# chat_id -> last triggering message ts (Slack). The typing/status
# lane's synthetic thread anchor in thread-per-message mode;
# see _capture_scope and send_typing.
self._last_inbound_ts_by_chat: Dict[str, str] = {}
# chat_id -> the UNDERLYING platform (e.g. "discord", "telegram") this
# chat belongs to (Phase 1.5 multi-platform-per-agent). One relay adapter
# fronts N platforms on one WS; an outbound reply must egress through the
@ -123,6 +137,27 @@ class RelayAdapter(BasePlatformAdapter):
def message_len_fn(self) -> Callable[[str], int]:
return _LEN_FNS.get(self.descriptor.len_unit, len)
@property
def supports_status_text(self) -> bool: # type: ignore[override]
"""Whether the fronted platform renders a TEXT status line.
Native parity (rich status text): Slack's typing surface is the
assistant status line ("Finding answers…" next to the bot name), a
text-rendering indicator. When the relay fronts Slack, advertise it so
run.py's live-status lane feeds per-tool phrases via
``set_status_text()`` exactly the wiring the native SlackAdapter
gets (``supports_status_text = True``). Other fronted platforms keep
textless typing bubbles and must NOT receive phrase traffic.
Property (not class attr) because ONE RelayAdapter class fronts many
platforms; the answer depends on the handshaked descriptor. On a
multi-platform relay this scalar reflects the PRIMARY identity's
platform (same convention as the scalar ``descriptor``); per-chat
egress paths that need chat-accurate capabilities use
``_descriptor_for_chat`` below.
"""
return self.descriptor.platform == Platform.SLACK.value
# ── per-chat capability resolution (Phase 1.5 multi-platform) ─────────
def _descriptor_for_chat(self, chat_id: str) -> CapabilityDescriptor:
"""The capability descriptor governing a specific chat.
@ -279,6 +314,7 @@ class RelayAdapter(BasePlatformAdapter):
async def _on_inbound(self, event) -> None:
"""Bridge a connector-delivered MessageEvent into the normal adapter path."""
self._capture_scope(event)
self._stamp_slack_session_thread(event)
# Phase 3: a structured prompt answer resolves its waiting primitive
# (approval/confirm/clarify) and is CONSUMED — it must not also
# dispatch as a chat message. Unknown/expired prompt ids fall through
@ -288,6 +324,112 @@ class RelayAdapter(BasePlatformAdapter):
await self._localize_inbound_media(event)
await self.handle_message(event)
def _relay_slack_extra(self) -> Dict[str, Any]:
"""The Slack-behavior subset of the RELAY platform config.
Enterprise knob shape (Hermes-config directed, relay-namespaced):
platforms:
relay:
extra:
slack: # supported subset of native Slack fields
reply_in_thread: true
The native ``platforms.slack`` block keeps meaning "native adapter
settings"; relay-fronted Slack reads its subset here. Legacy fallback:
a flat key on the relay extra (``extra.reply_in_thread``) still wins
when no ``slack`` object exists, preserving current staging configs.
"""
extra = getattr(self.config, "extra", None) or {}
sub = extra.get("slack")
return sub if isinstance(sub, dict) else extra
@staticmethod
def _coerce_flag(raw: Any, default: bool) -> bool:
"""Coerce an operator-supplied boolean exactly as native Slack does.
Native SlackAdapter reads its behavior flags with
``str(raw).strip().lower() in {"1","true","yes","on"}``, so a
YAML-quoted ``"false"`` a shape operators write routinely turns
the flag OFF. A bare ``bool()`` would read that same string as True
(non-empty string), silently ignoring the off switch. These knobs are
documented as native-parity mirrors, so they must coerce identically
or the parity claim only holds for unquoted YAML booleans.
"""
if raw is None:
return default
if isinstance(raw, bool):
return raw
return str(raw).strip().lower() in {"1", "true", "yes", "on"}
def _effective_reply_in_thread(self) -> bool:
"""Resolve the thread-per-message vs flat-DM mode for fronted Slack."""
try:
return self._coerce_flag(
self._relay_slack_extra().get("reply_in_thread"), True
)
except Exception: # noqa: BLE001 - config shape is operator-owned
return True
def _dm_top_level_threads_as_sessions(self) -> bool:
"""Native-parity escape hatch: per-message DM sessions on/off.
Mirrors native SlackAdapter._dm_top_level_threads_as_sessions
(platforms.slack.extra.dm_top_level_threads_as_sessions). Default
True: in thread-per-message mode each top-level DM message keys its
own session (parallel turns). Set
platforms.relay.extra.slack.dm_top_level_threads_as_sessions: false
to keep threaded reply PLACEMENT but ONE rolling DM session the
legacy steer/queue posture, decoupled from reply_in_thread.
"""
try:
return self._coerce_flag(
self._relay_slack_extra().get("dm_top_level_threads_as_sessions"),
True,
)
except Exception: # noqa: BLE001 - config shape is operator-owned
return True
def _stamp_slack_session_thread(self, event) -> None:
"""Native session-keying parity for fronted Slack DMs.
Native SlackAdapter's inbound handler stamps ``thread_ts =
event.thread_ts or ts`` every TOP-LEVEL message carries its own ts
as ``source.thread_id``, so build_session_key appends it and each
top-level message gets a FRESH session (per-message threads
per-message sessions; a 2nd message runs parallel instead of steering
the in-flight turn). The connector normalizes a top-level message
with thread_id=null, so without this stamp every top-level DM
collapses into ONE session key and message 2 pre-empts message 1
("Redirected current run", 2026-07-27 report).
Only in thread-per-message mode: flat mode keeps the shared rolling
DM session on purpose (steer/queue there is the intended UX). Never
overwrites a real thread_id (an in-thread reply must keep resolving
to its thread's session).
"""
try:
src = getattr(event, "source", None)
if not src:
return
platform = getattr(src, "platform", None)
if getattr(platform, "value", platform) != Platform.SLACK.value:
return
if getattr(src, "thread_id", None):
return # real thread — its session key is already correct
message_id = getattr(event, "message_id", None) or getattr(
src, "message_id", None
)
if not message_id:
return
if not self._effective_reply_in_thread():
return
if not self._dm_top_level_threads_as_sessions():
return # opt-out: threaded replies, one rolling session
src.thread_id = str(message_id)
except Exception: # noqa: BLE001 - session stamping must never break inbound
logger.debug("slack session-thread stamp failed", exc_info=True)
async def _localize_inbound_media(self, event) -> None:
"""Download connector re-hosted attachments to local temp paths.
@ -380,10 +522,32 @@ class RelayAdapter(BasePlatformAdapter):
scope = getattr(src, "scope_id", None)
if scope:
self._scope_by_chat[str(chat)] = str(scope)
# Remember the chat_type so send() can suppress the synthetic-DM
# thread anchor on Slack (native _resolve_thread_ts parity). send()
# only receives a chat_id, so it needs this per-chat cache to know a
# chat is a DM.
chat_type = getattr(src, "chat_type", None)
if chat_type:
self._chat_type_by_chat[str(chat)] = str(chat_type)
# Triggering message ts: the typing/status lane's metadata
# (base.py _thread_metadata_for_source) carries NO thread anchor
# for a top-level DM, but in thread-per-message mode the status
# must target the per-message thread (its root = this ts). Cache
# it per chat so send_typing can synthesize the anchor, mirroring
# native send_typing's _resolve_thread_ts(metadata.message_id).
# NOTE: message_id lives on the EVENT (MessageEvent), not the
# source — fall back to source for defensive coverage.
message_id = getattr(event, "message_id", None) or getattr(
src, "message_id", None
)
if message_id:
self._last_inbound_ts_by_chat[str(chat)] = str(message_id)
except Exception: # noqa: BLE001 - scope tracking must never break inbound
pass
def _with_scope(self, chat_id: str, metadata: Optional[Dict[str, Any]]) -> Dict[str, Any]:
def _with_scope(
self, chat_id: str, metadata: Optional[Dict[str, Any]]
) -> Dict[str, Any]:
"""Ensure the outbound metadata carries the discriminator(s) the connector's
egress guard needs to resolve the owning tenant.
@ -542,16 +706,26 @@ class RelayAdapter(BasePlatformAdapter):
else:
text = ""
member = payload.get("member") or {}
user = (member.get("user") if isinstance(member, dict) else None) or payload.get("user") or {}
user = (
(member.get("user") if isinstance(member, dict) else None)
or payload.get("user")
or {}
)
channel_id = str(payload.get("channel_id") or "")
guild_id = payload.get("guild_id") # real Discord interaction wire field
source = SessionSource(
platform=Platform.RELAY,
chat_id=channel_id,
chat_type="channel" if guild_id else "dm",
user_id=str(user.get("id")) if isinstance(user, dict) and user.get("id") else None,
user_name=str(user.get("username")) if isinstance(user, dict) and user.get("username") else None,
scope_id=str(guild_id) if guild_id else None, # Discord guild → generic scope slot
user_id=str(user.get("id"))
if isinstance(user, dict) and user.get("id")
else None,
user_name=str(user.get("username"))
if isinstance(user, dict) and user.get("username")
else None,
scope_id=str(guild_id)
if guild_id
else None, # Discord guild → generic scope slot
message_id=str(payload.get("id")) if payload.get("id") else None,
)
event = MessageEvent(text=text, message_type=message_type, source=source)
@ -567,7 +741,9 @@ class RelayAdapter(BasePlatformAdapter):
prompt_id, option_id = decoded
msg = payload.get("message") or {}
prompt_message_id = (
str(msg.get("id")) if isinstance(msg, dict) and msg.get("id") else None
str(msg.get("id"))
if isinstance(msg, dict) and msg.get("id")
else None
)
event.prompt_response = {
"prompt_id": prompt_id,
@ -620,7 +796,9 @@ class RelayAdapter(BasePlatformAdapter):
sub_name = str(opt.get("name") or "").strip()
if sub_name:
parts.append(sub_name)
parts.extend(RelayAdapter._render_interaction_options(opt.get("options")))
parts.extend(
RelayAdapter._render_interaction_options(opt.get("options"))
)
else:
value = opt.get("value")
if value is not None and str(value).strip():
@ -744,12 +922,19 @@ class RelayAdapter(BasePlatformAdapter):
)
if self._transport is None:
return SendResult(success=False, error="no transport")
# Native _resolve_thread_ts parity: a Slack DM reply must post flat at
# the DM root, not threaded under the triggering message. One shared
# helper resolves the anchor for EVERY egress lane (see
# _apply_slack_thread_anchor) so the text and media lanes cannot drift.
effective_reply_to = self._apply_slack_thread_anchor(
chat_id, reply_to, send_metadata
)
result = await self._transport.send_outbound(
{
"op": "send",
"chat_id": chat_id,
"content": content,
"reply_to": reply_to,
"reply_to": effective_reply_to,
"metadata": self._with_scope(chat_id, send_metadata),
},
platform=self._platform_by_chat.get(str(chat_id)),
@ -760,6 +945,144 @@ class RelayAdapter(BasePlatformAdapter):
error=result.get("error"),
)
def _resolve_reply_to_for_send(
self,
chat_id: str,
reply_to: Optional[str],
metadata: Optional[Dict[str, Any]],
) -> Optional[str]:
"""Suppress the synthetic-DM thread anchor for a Slack DM reply.
A DM turn's streaming reply is sent with ``reply_to`` = the triggering
message's ts (the stream consumer's ``initial_reply_to_id``, used as the
edit anchor and, on threading platforms, the reply target). The
connector's slackRestSender maps a raw ``reply_to`` to a Slack
``thread_ts``, so a plain DM reply would be posted THREADED under the
user's message instead of flat at the DM root — and a threaded first
send loses the progressive edit-streaming the user sees in a real
thread (the reported symptom: DM/home replies arrive flat, no
progressive edits).
Native Slack Hermes already suppresses this synthetic DM thread anchor:
``SlackAdapter._resolve_thread_ts`` returns ``None`` for a top-level /
DM message when ``reply_in_thread`` is off. The relay lane has no such
disambiguation, so we reproduce it here. run.py already encodes the
real-thread decision in ``metadata["thread_id"]`` (it is set only when
progress threading is active a real thread, or channel autoThread);
for a DM with no real thread that key is absent. So the rule is:
Slack DM + no real ``thread_id`` in metadata drop ``reply_to``.
This posts the reply flat at the DM root and lets the consumer edit its
own first-send ts streaming works exactly as in a thread. It does NOT:
* reintroduce a synthetic DM thread_id (#18859 / the /sethome
landmine) it removes an anchor, never adds one;
* regress real-thread streaming a real thread carries a distinct
``thread_id`` in metadata, so the guard leaves ``reply_to`` alone;
* regress channel autoThread a channel/group top-level reply carries
``thread_id`` (the message's own ts) in metadata when threading is
on, so it is left alone; and a non-DM chat is never matched here.
"""
if reply_to is None:
return None
if self._platform_by_chat.get(str(chat_id)) != Platform.SLACK.value:
return reply_to
if self._chat_type_by_chat.get(str(chat_id)) != "dm":
return reply_to
md = metadata or {}
if md.get("thread_id") or md.get("thread_ts"):
# A real thread was resolved by run.py — honour it.
return reply_to
# Mode gate (native _resolve_thread_ts parity). The final-reply lane
# (gateway/platforms/base.py) builds metadata from source.thread_id
# ONLY — for a top-level DM that is None, so in thread-per-message
# mode the triggering-ts reply_to here is the final reply's ONLY
# threading signal (run.py's synthetic root feeds just the
# progress/status lane). Dropping it unconditionally exiled the final
# message to the DM root while progress stayed threaded (2026-07-27
# report, same class as the prompt-placement bug). Native SlackAdapter only
# suppresses the anchor when reply_in_thread=false; mirror that.
reply_in_thread = self._effective_reply_in_thread()
if reply_in_thread:
# Thread-per-message: the triggering ts is the thread anchor.
return reply_to
# Flat mode: synthetic DM self-anchor — post flat at the DM root.
return None
def _apply_slack_thread_anchor(
self,
chat_id: str,
reply_to: Optional[str],
metadata: Dict[str, Any],
*,
mirror_key: str = "reply_to_message_id",
) -> Optional[str]:
"""Resolve the outbound Slack thread anchor for ONE egress frame.
The single choke point every send lane goes through text (``send``)
and media (``send_media``) alike. It does three things that must always
happen together, and previously only happened on the text lane:
1. Mode gate: ``_resolve_reply_to_for_send`` drops the synthetic DM
self-anchor in flat mode, keeps it in thread-per-message mode.
2. Mirror strip: when the anchor is dropped, remove the mirrored
``metadata.reply_to_message_id`` too, so the connector cannot
thread on the copy we forgot about.
3. Anchor promotion: the connector's Slack sender THREADS ON METADATA
ONLY ``threadTs()`` reads ``metadata.thread_id``/``thread_ts``
and never looks at the frame's ``reply_to``. A surviving anchor is
promoted into ``metadata.thread_id`` or the message silently lands
in the home channel instead of the per-message thread.
``metadata`` is mutated in place; the effective ``reply_to`` is
returned. Non-Slack and non-DM chats are untouched by (1), and (3) is
Slack-only, so other fronted platforms keep their existing behaviour.
"""
effective_reply_to = self._resolve_reply_to_for_send(
chat_id, reply_to, metadata
)
if effective_reply_to is None and reply_to is not None:
metadata.pop(mirror_key, None)
if (
effective_reply_to is not None
and self._platform_by_chat.get(str(chat_id)) == Platform.SLACK.value
and not (metadata.get("thread_id") or metadata.get("thread_ts"))
):
metadata["thread_id"] = str(effective_reply_to)
return effective_reply_to
def _with_status_thread_anchor(
self, chat_id: str, metadata: Optional[Dict[str, Any]]
) -> Dict[str, Any]:
"""Copy ``metadata`` with the typing/status thread anchor applied.
Slack's status line is THREAD-scoped: the connector's typing case
no-ops without a thread anchor, and the typing lane's metadata
(base.py ``_thread_metadata_for_source``) carries none for a top-level
DM (``source.thread_id`` is None). Synthesize it from the per-chat
inbound-ts cache, exactly as native ``send_typing`` resolves
``thread_ts`` from ``metadata.message_id``.
Unconditional across both modes: in flat mode the send lane strips its
own anchors (see ``_apply_slack_thread_anchor``), so the status anchor
cannot leak into reply placement, and ``setStatus`` clears without
leaving a message artifact.
Shared by ``send_typing`` and ``stop_typing`` the clear MUST target
the same thread the heartbeat set, or the status line sticks until
Slack's own timeout. Keeping one implementation is what guarantees it.
"""
md = dict(metadata or {})
if (
not (md.get("thread_id") or md.get("thread_ts"))
and self._platform_by_chat.get(str(chat_id)) == Platform.SLACK.value
and self._chat_type_by_chat.get(str(chat_id)) == "dm"
):
anchor = self._last_inbound_ts_by_chat.get(str(chat_id))
if anchor:
md["thread_id"] = anchor
return md
async def edit_message(
self,
chat_id: str,
@ -817,13 +1140,39 @@ class RelayAdapter(BasePlatformAdapter):
"""
if self._transport is None:
return
# Thread anchor for the status surface. Slack's status line
# ("is thinking…" in the thread's replies footer — works with plain
# chat:write, confirmed on native no-assistant bots) is THREAD-only:
# the connector's typing case no-ops without a thread_ts. But the
# typing lane's metadata (base.py _thread_metadata_for_source) has no
# anchor for a top-level DM — source.thread_id is None — so every
# heartbeat was silently dropped. In thread-per-message mode the
# turn's thread root IS the triggering message ts (run.py's synthetic
# root); synthesize it here from the per-chat inbound cache, exactly
# like native send_typing resolves thread_ts from metadata.message_id.
# Flat mode (reply_in_thread=false) keeps the no-anchor no-op: there
# is no thread and must not be one (#18859).
md = self._with_status_thread_anchor(chat_id, metadata)
# Rich status parity: run.py's live-status lane stashes the
# current per-tool phrase via set_status_text() (base class store).
# Carry it as the typing frame's content so the connector's Slack
# sender renders it on assistant.threads.setStatus — the same phrase
# the native adapter shows ("is running pytest…", "Finding answers…").
# Absent (None/empty) => omit content; the connector falls back to its
# default "is typing…" heartbeat, preserving pre-phrase behaviour on
# every platform. Never send empty-string content here: on Slack that
# is the explicit CLEAR request reserved for stop_typing.
frame: Dict[str, Any] = {
"op": "typing",
"chat_id": chat_id,
"metadata": self._with_scope(chat_id, md),
}
phrase = getattr(self, "_status_text", {}).get(str(chat_id))
if phrase:
frame["content"] = str(phrase)
try:
await self._transport.send_outbound(
{
"op": "typing",
"chat_id": chat_id,
"metadata": self._with_scope(chat_id, metadata),
},
frame,
platform=self._platform_by_chat.get(str(chat_id)),
)
except Exception: # noqa: BLE001 - typing is cosmetic, never breaks a turn
@ -852,13 +1201,17 @@ class RelayAdapter(BasePlatformAdapter):
platform = self._platform_by_chat.get(str(chat_id))
if platform != Platform.SLACK.value:
return
# Clear must target the SAME thread the heartbeat set, or the clear
# frame no-ops threadless and the status line sticks until Slack's own
# timeout. Shared helper with send_typing so the two guards cannot drift.
md = self._with_status_thread_anchor(chat_id, metadata)
try:
await self._transport.send_outbound(
{
"op": "typing",
"chat_id": chat_id,
"content": "",
"metadata": self._with_scope(chat_id, metadata),
"metadata": self._with_scope(chat_id, md),
},
platform=platform,
)
@ -983,14 +1336,23 @@ class RelayAdapter(BasePlatformAdapter):
if not uploaded:
return None
source_url = uploaded
# Same Slack thread-anchor contract as the text lane (send). Media
# frames egress through the connector's Slack sender too, so an
# unresolved anchor threads an image under the user's DM message in
# flat mode, and loses the per-message thread entirely in thread mode
# (threadTs() reads metadata only). Route through the shared helper.
media_metadata: Dict[str, Any] = dict(metadata or {})
effective_reply_to = self._apply_slack_thread_anchor(
chat_id, reply_to, media_metadata
)
action: Dict[str, Any] = {
"op": "send_media",
"chat_id": chat_id,
"media_kind": media_kind,
"source_url": source_url,
"content": caption or "",
"reply_to": reply_to,
"metadata": self._with_scope(chat_id, metadata),
"reply_to": effective_reply_to,
"metadata": self._with_scope(chat_id, media_metadata),
}
if filename:
action["filename"] = filename
@ -1065,8 +1427,12 @@ class RelayAdapter(BasePlatformAdapter):
if result is not None:
return result
return await super().send_image_file(
chat_id, image_path, caption=caption, reply_to=reply_to,
metadata=metadata, **kwargs,
chat_id,
image_path,
caption=caption,
reply_to=reply_to,
metadata=metadata,
**kwargs,
)
async def send_voice(
@ -1091,8 +1457,12 @@ class RelayAdapter(BasePlatformAdapter):
if result is not None:
return result
return await super().send_voice(
chat_id, audio_path, caption=caption, reply_to=reply_to,
metadata=metadata, **kwargs,
chat_id,
audio_path,
caption=caption,
reply_to=reply_to,
metadata=metadata,
**kwargs,
)
async def send_video(
@ -1117,8 +1487,12 @@ class RelayAdapter(BasePlatformAdapter):
if result is not None:
return result
return await super().send_video(
chat_id, video_path, caption=caption, reply_to=reply_to,
metadata=metadata, **kwargs,
chat_id,
video_path,
caption=caption,
reply_to=reply_to,
metadata=metadata,
**kwargs,
)
async def send_document(
@ -1145,13 +1519,20 @@ class RelayAdapter(BasePlatformAdapter):
if result is not None:
return result
return await super().send_document(
chat_id, file_path, caption=caption, file_name=file_name,
reply_to=reply_to, metadata=metadata, **kwargs,
chat_id,
file_path,
caption=caption,
file_name=file_name,
reply_to=reply_to,
metadata=metadata,
**kwargs,
)
# ── Phase 3 interactive: prompt + react ──────────────────────────────
def _mint_prompt(self, kind: str, state: Dict[str, Any], timeout_s: float = 3600.0) -> str:
def _mint_prompt(
self, kind: str, state: Dict[str, Any], timeout_s: float = 3600.0
) -> str:
"""Register a pending prompt and return its 8-hex id.
``state`` carries what the resolver needs when the answer comes back
@ -1171,7 +1552,9 @@ class RelayAdapter(BasePlatformAdapter):
# Opportunistic sweep so abandoned prompts can't accumulate: drop
# anything already expired (cheap — dict is small by construction).
now = time.time()
for stale in [k for k, v in self._pending_prompts.items() if v.get("expires_at", 0) < now]:
for stale in [
k for k, v in self._pending_prompts.items() if v.get("expires_at", 0) < now
]:
self._pending_prompts.pop(stale, None)
return prompt_id
@ -1207,6 +1590,20 @@ class RelayAdapter(BasePlatformAdapter):
"""
if self._transport is None or not self.descriptor.supports_op("prompt"):
return None
# An interactive prompt (approval / clarify / slash-confirm) is emitted
# mid-turn in reply to the triggering inbound event, so `metadata` carries
# that event's thread context (run.py _thread_metadata_for_source stamps
# metadata.thread_id — for a Slack DM the triggering message's own ts,
# used only as a session-keying fallback). Forwarding it makes the
# connector thread the prompt card UNDER the triggering message instead
# of posting it flat at the DM root (the reported bug). Native Slack
# Hermes suppresses this synthetic DM thread anchor; drop it here for the
# same Slack-DM-with-no-real-thread case, matching _resolve_reply_to_for_send.
# Prompt metadata is forwarded VERBATIM. The threading mode is decided
# in exactly one place — run.py's _resolve_progress_thread_id (flat mode
# suppresses the synthetic self-anchor there; thread mode stamps the
# turn's thread). Boundary pinned by test_run_py_suppresses_self_anchor*.
prompt_metadata = metadata
action: Dict[str, Any] = {
"op": "prompt",
"chat_id": chat_id,
@ -1214,8 +1611,10 @@ class RelayAdapter(BasePlatformAdapter):
"prompt_kind": prompt_kind,
"prompt_id": prompt_id,
"options": options,
"reply_to": reply_to,
"metadata": self._with_scope(chat_id, metadata),
"reply_to": self._resolve_reply_to_for_send(
chat_id, reply_to, prompt_metadata
),
"metadata": self._with_scope(chat_id, prompt_metadata),
}
if timeout_s is not None:
action["timeout_s"] = int(timeout_s)
@ -1260,12 +1659,15 @@ class RelayAdapter(BasePlatformAdapter):
buttontext fallback takes over (same contract as a native adapter's
failed button send).
"""
options: list = [{"id": "once", "label": "Allow Once", "style": "success"}]
options: list = [{"id": "once", "label": "Allow Once", "style": "primary"}]
if not smart_denied and allow_session:
options.append({"id": "session", "label": "✅ Session", "style": "primary"})
options.append({"id": "session", "label": "Allow Session"})
if allow_permanent:
options.append({"id": "always", "label": "✅ Always", "style": "primary"})
options.append({"id": "deny", "label": "❌ Deny", "style": "danger"})
options.append({
"id": "always",
"label": "Always Allow",
})
options.append({"id": "deny", "label": "Deny", "style": "danger"})
cmd_preview = command if len(command) <= 1500 else command[:1500] + "..."
text = (
@ -1274,7 +1676,9 @@ class RelayAdapter(BasePlatformAdapter):
f"Reason: {description}"
)
if smart_denied:
text += "\n\n**Smart DENY:** owner override applies to this one operation only."
text += (
"\n\n**Smart DENY:** owner override applies to this one operation only."
)
prompt_id = self._mint_prompt(
"exec_approval",
@ -1310,9 +1714,9 @@ class RelayAdapter(BasePlatformAdapter):
gateway's text-intercept flow when the prompt lane is unavailable.
"""
options = [
{"id": "once", "label": "Approve Once", "style": "success"},
{"id": "always", "label": "🔒 Always Approve", "style": "primary"},
{"id": "cancel", "label": "Cancel", "style": "danger"},
{"id": "once", "label": "Approve Once", "style": "primary"},
{"id": "always", "label": "Always Approve"},
{"id": "cancel", "label": "Cancel", "style": "danger"},
]
text = f"**{title}**\n\n{message}" if title else message
prompt_id = self._mint_prompt(
@ -1422,7 +1826,11 @@ class RelayAdapter(BasePlatformAdapter):
if kind == "exec_approval":
from tools.approval import resolve_gateway_approval
choice = option_id if option_id in {"once", "session", "always", "deny"} else "deny"
choice = (
option_id
if option_id in {"once", "session", "always", "deny"}
else "deny"
)
count = resolve_gateway_approval(session_key, choice)
label = {
"once": "✅ Approved once",
@ -1435,13 +1843,17 @@ class RelayAdapter(BasePlatformAdapter):
# Acknowledge in-channel (the connector's prompt message can't
# be edited cross-platform yet — edit support varies; a short
# confirmation preserves the audit trail the native edit gives).
await self.send(chat_id, label, metadata=self._prompt_reply_metadata(event))
await self.send(
chat_id, label, metadata=self._prompt_reply_metadata(event)
)
if count:
self.resume_typing_for_chat(chat_id)
elif kind == "slash_confirm":
from tools import slash_confirm as slash_confirm_mod
choice = option_id if option_id in {"once", "always", "cancel"} else "cancel"
choice = (
option_id if option_id in {"once", "always", "cancel"} else "cancel"
)
result_text = await slash_confirm_mod.resolve(
session_key, str(state.get("confirm_id") or ""), choice
)
@ -1450,13 +1862,20 @@ class RelayAdapter(BasePlatformAdapter):
"always": "🔒 Always approve",
"cancel": "❌ Cancelled",
}.get(choice, "Resolved")
await self.send(chat_id, label, metadata=self._prompt_reply_metadata(event))
await self.send(
chat_id, label, metadata=self._prompt_reply_metadata(event)
)
if result_text:
await self.send(
chat_id, str(result_text), metadata=self._prompt_reply_metadata(event)
chat_id,
str(result_text),
metadata=self._prompt_reply_metadata(event),
)
elif kind == "clarify":
from tools.clarify_gateway import mark_awaiting_text, resolve_gateway_clarify
from tools.clarify_gateway import (
mark_awaiting_text,
resolve_gateway_clarify,
)
clarify_id = str(state.get("clarify_id") or "")
if option_id == "other":

View File

@ -21924,11 +21924,22 @@ class GatewayRunner(GatewayAuthorizationMixin, GatewayKanbanWatchersMixin, Gatew
_slack_adapter_for_progress = self._adapter_for_source(source)
if _slack_adapter_for_progress is not None:
try:
_progress_reply_in_thread = bool(
_slack_adapter_for_progress.config.extra.get(
"reply_in_thread", True
)
# Relay lane: the adapter owns mode resolution (nested
# platforms.relay.extra.slack subset with flat-key
# fallback). Native lane: read the flat extra as before.
_mode_fn = getattr(
_slack_adapter_for_progress,
"_effective_reply_in_thread",
None,
)
if callable(_mode_fn):
_progress_reply_in_thread = bool(_mode_fn())
else:
_progress_reply_in_thread = bool(
_slack_adapter_for_progress.config.extra.get(
"reply_in_thread", True
)
)
except Exception:
_progress_reply_in_thread = True
_progress_thread_id = _resolve_progress_thread_id(
@ -24707,6 +24718,23 @@ def _start_gateway_housekeeping(stop_event: threading.Event, adapters=None, loop
except Exception as e:
logger.debug("Curator tick error: %s", e)
# Skill Sync — best-effort periodic pull on the same cadence.
# Inert unless the access gate is open and a sync base URL is
# configured; never raises.
try:
from tools.skills_sync_client import maybe_pull_skills
maybe_pull_skills()
except Exception as e:
logger.debug("Sync pull tick error: %s", e)
# Org-shared skills. Gated on real org membership (the token must
# carry an org role), so a solo account never reaches the network.
try:
from tools.skills_sync_client import maybe_pull_org_skills
maybe_pull_org_skills()
except Exception as e:
logger.debug("Org sync pull tick error: %s", e)
# Stale-session auto-archive — a live timer, so gateways that stay up
# for weeks keep sweeping on schedule (the startup hook fires once).
# maybe_auto_archive() is gated by sessions.min_interval_hours in

View File

@ -433,6 +433,7 @@ import functools as _functools
from hermes_cli.sessions_cmd import cmd_sessions # noqa: F401
from hermes_cli.subcommands._shared import add_accept_hooks_flag as _add_accept_hooks_flag
from hermes_cli.subcommands.cron import build_cron_parser
from hermes_cli.subcommands.sync import build_sync_parser
from hermes_cli.subcommands.gateway import build_gateway_parser
from hermes_cli.subcommands.profile import build_profile_parser
from hermes_cli.subcommands.model import build_model_parser
@ -4536,6 +4537,201 @@ def cmd_cron(args):
cron_command(args)
def cmd_sync(args):
"""Skill Sync — personal sync across devices, plus sharing with your org."""
import json as _json
sub = getattr(args, "sync_command", None)
if sub in {None, ""}:
print(
"usage: hermes sync "
"<status|pull|push|now|enable|disable|device|propose>\n"
"\n"
"Your skills, across your devices:\n"
" status Show what is synced, and from where\n"
" pull Pull your synced skills\n"
" push Push your opted-in skills\n"
" now Reconcile now: pull then push\n"
" enable <skill> Include a skill in your sync\n"
" disable <skill> Exclude a skill from your sync\n"
" device [--name N] Show or set this device's label\n"
"\n"
"Shared with your team:\n"
" propose <skill> Share a skill with your organisation",
file=sys.stderr,
)
return 1
if sub == "device":
from tools import skills_sync_client as ssc
name = getattr(args, "device_name", None)
if name is not None:
try:
stored = ssc.set_device_name(name)
except ValueError as e:
print(f"error: {e}", file=sys.stderr)
return 1
print(f"device label set to '{stored}'.")
print(
"New commits from this device will use this label; existing "
"commits keep their previous one.",
file=sys.stderr,
)
return 0
# No --name: print the current (creating a default on first use).
print(ssc.stable_device_id())
return 0
if sub == "propose":
from tools import skills_sync_client as ssc
name = args.name
try:
result = ssc.propose_skill(name, message=args.message)
except ssc.SyncInertError as e:
print(f"cannot share this skill: {e}", file=sys.stderr)
return 1
except ssc.SyncError as e:
print(f"could not share '{name}': {e}", file=sys.stderr)
return 1
if result.get("proposal_pending"):
print(
f"Shared '{name}' with your organisation — an admin needs to "
f"approve it (proposal #{result.get('proposal_id')}). It is "
f"not live for the team until then."
)
else:
print(f"Added '{name}' to your organisation's shared skills.")
return 0
if sub in {"enable", "disable"}:
from tools.skill_usage import set_sync, is_curation_eligible
skill = args.skill
if not is_curation_eligible(skill):
print(
f"'{skill}' is not sync-eligible (bundled, hub-installed, "
f"external, or not found). Only agent-created / user-authored "
f"skills under ~/.hermes/skills/ can sync.",
file=sys.stderr,
)
return 1
set_sync(skill, sub == "enable")
print(f"sync {'enabled' if sub == 'enable' else 'disabled'} for '{skill}'.")
return 0
from tools import skills_sync_client as ssc
if sub == "status":
status = ssc.sync_status()
print(_json.dumps(status, indent=2, ensure_ascii=False))
if status.get("org_available"):
n = len(status.get("org_skills") or [])
modified = status.get("org_skills_modified") or []
print(
f"\nOrg skills: {n} shared skill(s) from your organisation "
f"(your role: {status.get('org_role')}). They load alongside "
f"your own, labeled by origin, and you can edit them.",
file=sys.stderr,
)
if modified:
print(
f" {len(modified)} with local edits not yet shared: "
f"{', '.join(modified)}\n"
f" Share them back with `hermes sync propose <skill>`. "
f"Org updates will not overwrite them.",
file=sys.stderr,
)
elif status.get("logged_in"):
print(
"\nOrg skills: not applicable — this account isn't a member "
"of a shared organisation.",
file=sys.stderr,
)
if not status.get("logged_in"):
print("\nNot logged into Nous Portal — sync is inert.", file=sys.stderr)
elif not status.get("nous_admin"):
print(
"\nSync is not enabled for your account yet.",
file=sys.stderr,
)
elif not status.get("feature_enabled"):
print(
"\nSync feature is off for this instance (set HERMES_SYNC_ENABLED=1 "
"or config.yaml sync.enabled: true). Sync is inert.",
file=sys.stderr,
)
elif not status.get("base_url"):
print(
"\nNo sync base URL configured (config.yaml sync.base_url or "
"HERMES_SYNC_BASE_URL). Sync is inert.",
file=sys.stderr,
)
return 0
# pull / push / now — enforce the gate up front with a clear message.
try:
identity = ssc.resolve_identity()
except ssc.SyncInertError as e:
print(f"sync inert: {e}", file=sys.stderr)
return 1
if not identity.get("nous_admin"):
print(
"sync unavailable: not enabled for your account yet.",
file=sys.stderr,
)
return 1
if not ssc.resolve_sync_base_url():
print(
"sync inert: no sync base URL configured (config.yaml sync.base_url "
"or HERMES_SYNC_BASE_URL).",
file=sys.stderr,
)
return 1
try:
if sub == "pull":
result = ssc.pull_skills(identity=identity)
# Refresh the org mirror too when this account belongs to an
# organisation (no-op otherwise), so one pull covers both.
org_result = ssc.maybe_pull_org_skills()
if org_result:
n = len(org_result.get("updated") or [])
print(
f"org: refreshed {n} shared skill(s) from your "
f"organisation.",
file=sys.stderr,
)
clashes = org_result.get("conflicted") or []
if clashes:
print(
f"org: {len(clashes)} skill(s) have BOTH local edits "
f"and org updates, so they were left as-is: "
f"{', '.join(clashes)}\n"
f" Your local version is intact. Review it, then "
f"either propose it or delete the local copy and pull "
f"again to take the org version.",
file=sys.stderr,
)
elif sub == "push":
result = ssc.push_skills(identity=identity, message="hermes sync push")
elif sub == "now":
pull_res = ssc.pull_skills(identity=identity)
push_res = ssc.push_skills(identity=identity, message="hermes sync now")
result = {"pull": pull_res, "push": push_res}
else:
print(f"Unknown sync subcommand: {sub}", file=sys.stderr)
return 1
except ssc.SyncError as e:
print(f"sync failed: {e}", file=sys.stderr)
return 1
print(_json.dumps(result, indent=2, ensure_ascii=False))
return 0
def cmd_webhook(args):
"""Webhook subscription management."""
from hermes_cli.webhook import webhook_command
@ -10232,7 +10428,7 @@ _BUILTIN_SUBCOMMANDS = frozenset(
"project", "proxy",
"prompt-size",
"send", "sessions", "setup",
"skin", "skills", "slack", "status", "tools", "uninstall", "update",
"skin", "skills", "slack", "status", "sync", "tools", "uninstall", "update",
"version", "webhook", "whatsapp", "whatsapp-cloud", "chat", "secrets", "security",
# Help-ish invocations — plugin commands not being listed in
# top-level --help is an acceptable trade-off for skipping an
@ -11096,6 +11292,7 @@ def main():
# cron command (parser built in hermes_cli/subcommands/cron.py)
# =========================================================================
build_cron_parser(subparsers, cmd_cron=cmd_cron)
build_sync_parser(subparsers, cmd_sync=cmd_sync)
# =========================================================================
# webhook command (parser built in hermes_cli/subcommands/webhook.py)

View File

@ -312,4 +312,5 @@ def build_skills_parser(subparsers, *, cmd_skills: Callable) -> None:
"config",
help="Interactive skill configuration — enable/disable individual skills",
)
skills_parser.set_defaults(func=cmd_skills)

View File

@ -0,0 +1,99 @@
"""``hermes sync`` subcommand parser — Skill Sync.
Cloned from ``hermes_cli/subcommands/cron.py`` same injected-handler shape
(``func=cmd_sync``) so this module does not import ``main`` (cycle avoidance).
Skill Sync covers two surfaces, both under this one command for launch:
Personal your own skills, across your own devices:
hermes sync status show gate/opt-in/head state
hermes sync pull pull and materialize opted-in skills
hermes sync push push opted-in skills
hermes sync now reconcile: pull then push
hermes sync enable <skill> opt a skill into sync
hermes sync disable <skill> opt a skill out of sync
hermes sync device [--name] show or set this device's label
Organisation skills shared with your team:
hermes sync propose <skill> share a skill with your organisation
Sync is INERT unless the resolved Nous token carries the access-gate claim
AND a sync base URL is configured. The commands report that state rather than
failing opaquely.
"""
from __future__ import annotations
import argparse
from typing import Callable
def build_sync_parser(subparsers, *, cmd_sync: Callable) -> None:
"""Attach the ``sync`` subcommand (and its sub-actions) to ``subparsers``."""
sync_parser = subparsers.add_parser(
"sync",
help="Skill Sync — sync your skills across devices and with your team",
description=(
"Skill Sync keeps your skills with you. Personal sync moves your "
"own skills between your devices; if you belong to an "
"organisation, you also get its shared skills and can propose "
"your own back to the team."
),
epilog=(
"Examples:\n"
" hermes sync status what is synced, and from where\n"
" hermes sync enable my-skill include a skill in your sync\n"
" hermes sync now pull, then push\n"
" hermes sync propose my-skill share a skill with your team\n"
),
formatter_class=argparse.RawDescriptionHelpFormatter,
)
sync_sub = sync_parser.add_subparsers(dest="sync_command")
sync_sub.add_parser("status", help="Show what is synced, and from where")
sync_sub.add_parser(
"pull", help="Pull your synced skills (and your organisation's)"
)
sync_sub.add_parser("push", help="Push your opted-in skills")
sync_sub.add_parser("now", help="Reconcile now: pull then push")
enable = sync_sub.add_parser("enable", help="Include a skill in your sync")
enable.add_argument("skill", help="Skill name (frontmatter name / directory name)")
disable = sync_sub.add_parser("disable", help="Exclude a skill from your sync")
disable.add_argument("skill", help="Skill name (frontmatter name / directory name)")
device = sync_sub.add_parser(
"device",
help="Show or set this device's label (shown in the sync console)",
)
device.add_argument(
"--name",
dest="device_name",
default=None,
help="Set a human-friendly label for this device (e.g. \"Ben's Laptop\"). "
"Omit to print the current label.",
)
# Org-shared skills. A member's submission becomes a proposal an admin
# reviews; an admin's merges straight into the shared set. Accounts that
# aren't in a shared organisation are told so plainly.
propose = sync_sub.add_parser(
"propose",
help="Share a skill with your organisation",
description=(
"Submit one of your skills to your organisation's shared set. If "
"you are an admin it is added directly; otherwise it becomes a "
"proposal for an admin to review. Accounts that aren't part of a "
"shared organisation don't have this workflow."
),
)
propose.add_argument("name", help="Skill name to share")
propose.add_argument(
"-m",
"--message",
default=None,
help="Optional message describing the change",
)
sync_parser.set_defaults(func=cmd_sync)

View File

@ -45,6 +45,7 @@ import tempfile
import time
import threading
import uuid
import warnings
from typing import List, Dict, Any, Optional, Callable
# NOTE: `from openai import OpenAI` is deliberately NOT at module top — the
# SDK pulls ~240 ms of imports. We expose `OpenAI` as a thin proxy object
@ -439,8 +440,8 @@ class AIAgent:
command: str = None,
args: list[str] | None = None,
model: str = "",
max_iterations: int = 500, # Default tool-calling iterations (shared with subagents)
tool_delay: float = 1.0,
max_iterations: int = 90, # Default tool-calling iterations (shared with subagents)
tool_delay: float = None, # Deprecated: accepted for compatibility, ignored
enabled_toolsets: List[str] = None,
disabled_toolsets: List[str] = None,
save_trajectories: bool = False,
@ -504,6 +505,13 @@ class AIAgent:
requested_provider: str = None,
):
"""Forwarder — see ``agent.agent_init.init_agent``."""
if tool_delay is not None:
warnings.warn(
"tool_delay is deprecated and ignored; sequential tool calls "
"no longer sleep between executions.",
DeprecationWarning,
stacklevel=2,
)
from agent.agent_init import init_agent
init_agent(
self,
@ -518,7 +526,6 @@ class AIAgent:
args=args,
model=model,
max_iterations=max_iterations,
tool_delay=tool_delay,
enabled_toolsets=enabled_toolsets,
disabled_toolsets=disabled_toolsets,
save_trajectories=save_trajectories,

View File

@ -0,0 +1,389 @@
"""M2 org-skill namespace: token-gated resolution, provenance, collisions.
Covers the design agreed 2026-07-23 (bare-name first-class org skills):
1. TOKEN-GATED discovery only the `.active_org`-marked mirror resolves;
stale mirrors and marker-less trees never load.
2. Fail-loud collisions a personal/org name clash lists BOTH sides flagged;
skill_view's existing multi-candidate guard refuses the bare name.
3. Load-time provenance header org skill content announces org + author.
4. Org mirrors are read-only (skill_manage guards) and curation-exempt.
"""
import json
import pytest
from agent import skill_utils as sku
from agent.prompt_builder import _build_snapshot_entry
def _mk_skill(root, rel, name=None, body="# body\n"):
d = root
for part in rel.split("/"):
d = d / part
d.mkdir(parents=True, exist_ok=True)
(d / "SKILL.md").write_text(
f"---\nname: {name or rel.split('/')[-1]}\ndescription: d\n---\n{body}",
encoding="utf-8",
)
return d
def _mark_active(skills, org_id):
org_root = skills / sku.ORG_MIRROR_DIR_NAME
org_root.mkdir(parents=True, exist_ok=True)
(org_root / sku.ORG_ACTIVE_MARKER).write_text(org_id, encoding="utf-8")
class TestTokenGatedDiscovery:
def test_no_marker_no_org_skills(self, tmp_path):
skills = tmp_path / "skills"
_mk_skill(skills, "personal-a")
_mk_skill(skills, f"{sku.ORG_MIRROR_DIR_NAME}/org-1/shared-x", name="shared-x")
found = [p.parent.name for p in sku.iter_skill_index_files(skills, "SKILL.md")]
assert "personal-a" in found
assert "shared-x" not in found # unmarked mirror never resolves
def test_marker_gates_to_active_org_only(self, tmp_path):
skills = tmp_path / "skills"
_mk_skill(skills, f"{sku.ORG_MIRROR_DIR_NAME}/org-1/shared-x", name="shared-x")
_mk_skill(skills, f"{sku.ORG_MIRROR_DIR_NAME}/org-OLD/stale-y", name="stale-y")
_mark_active(skills, "org-1")
found = [p.parent.name for p in sku.iter_skill_index_files(skills, "SKILL.md")]
assert "shared-x" in found
assert "stale-y" not in found # stale mirror pruned at resolution
def test_switching_org_flips_resolution(self, tmp_path):
skills = tmp_path / "skills"
_mk_skill(skills, f"{sku.ORG_MIRROR_DIR_NAME}/org-1/shared-x", name="shared-x")
_mk_skill(skills, f"{sku.ORG_MIRROR_DIR_NAME}/org-2/other-z", name="other-z")
_mark_active(skills, "org-2")
found = [p.parent.name for p in sku.iter_skill_index_files(skills, "SKILL.md")]
assert found and "other-z" in found and "shared-x" not in found
def test_helpers(self, tmp_path):
skills = tmp_path / "skills"
d = _mk_skill(skills, f"{sku.ORG_MIRROR_DIR_NAME}/org-9/cat/sk", name="sk")
assert sku.is_org_mirror_path(d, skills) is True
assert sku.org_id_of_path(d, skills) == "org-9"
p = _mk_skill(skills, "plain")
assert sku.is_org_mirror_path(p, skills) is False
assert sku.read_active_org_id(skills) is None
_mark_active(skills, "org-9")
assert sku.read_active_org_id(skills) == "org-9"
class TestSnapshotEntryProvenance:
def test_org_entry_strips_prefix_and_carries_provenance(self, tmp_path):
skills = tmp_path / "skills"
d = _mk_skill(
skills, f"{sku.ORG_MIRROR_DIR_NAME}/org-1/devops/beta", name="beta"
)
(skills / sku.ORG_MIRROR_DIR_NAME / "org-1" / sku.ORG_PROVENANCE_FILE).write_text(
json.dumps(
{"author_device": "bens-macbook-a1b2c3", "author_user_id": "u1"}
),
encoding="utf-8",
)
entry = _build_snapshot_entry(d / "SKILL.md", skills, {"name": "beta"}, "d")
assert entry["org_id"] == "org-1"
assert entry["org_author"] == "bens-macbook-a1b2c3"
# Category derives from the path WITHIN the mirror, not _org/org-1/...
assert entry["category"] == "devops"
assert entry["skill_name"] == "beta"
def test_personal_entry_unchanged(self, tmp_path):
skills = tmp_path / "skills"
d = _mk_skill(skills, "devops/beta", name="beta")
entry = _build_snapshot_entry(d / "SKILL.md", skills, {"name": "beta"}, "d")
assert "org_id" not in entry
assert entry["category"] == "devops"
class TestListingCollisionsAndLabels:
def _render(self, tmp_path, monkeypatch):
from agent import prompt_builder as pb
skills = tmp_path / "skills"
skills.mkdir(parents=True, exist_ok=True)
monkeypatch.setattr(pb, "get_skills_dir", lambda: skills, raising=True)
monkeypatch.setattr(
pb, "get_all_skills_dirs", lambda: [skills], raising=True
)
monkeypatch.setattr(pb, "get_disabled_skill_names", lambda *a, **k: set())
monkeypatch.setattr(
pb, "_skills_prompt_snapshot_path", lambda: tmp_path / "snap.json"
)
pb.clear_skills_system_prompt_cache()
return skills, pb
def test_org_skill_listed_with_provenance_tag(self, tmp_path, monkeypatch):
skills, pb = self._render(tmp_path, monkeypatch)
_mk_skill(skills, "personal-a")
_mk_skill(skills, f"{sku.ORG_MIRROR_DIR_NAME}/org-1/shared-x", name="shared-x")
(skills / sku.ORG_MIRROR_DIR_NAME / "org-1" / sku.ORG_PROVENANCE_FILE).write_text(
json.dumps({"author_device": "bens-macbook"}), encoding="utf-8"
)
_mark_active(skills, "org-1")
out = pb.build_skills_system_prompt()
assert "org:org-1" in out
assert "[org-shared: by bens-macbook]" in out
assert "personal-a" in out
def test_collision_flags_both_sides(self, tmp_path, monkeypatch):
skills, pb = self._render(tmp_path, monkeypatch)
_mk_skill(skills, "k8s-debug", body="personal version\n")
_mk_skill(
skills, f"{sku.ORG_MIRROR_DIR_NAME}/org-1/k8s-debug", name="k8s-debug"
)
_mark_active(skills, "org-1")
out = pb.build_skills_system_prompt()
# BOTH entries flagged — neither silently wins.
assert out.count("[name collision") == 2
def test_no_collision_flag_when_unique(self, tmp_path, monkeypatch):
skills, pb = self._render(tmp_path, monkeypatch)
_mk_skill(skills, "personal-a")
_mk_skill(skills, f"{sku.ORG_MIRROR_DIR_NAME}/org-1/shared-x", name="shared-x")
_mark_active(skills, "org-1")
out = pb.build_skills_system_prompt()
assert "[name collision" not in out
class TestOrgSkillsAreEditableInPlace:
"""The learning loop must work ON shared skills, not around them.
Refusing edits to `_org/` froze exactly the skills the most people use:
the agent is instructed to patch a skill the moment it finds a gap, and
"fork it to a personal skill first" is not something an agent does
mid-task. So edits land in place; org updates never clobber them; the
user (or auto-propose) shares them back.
"""
def _org_skill(self, tmp_path, monkeypatch):
from tools import skill_manager_tool as smt
from agent import skill_utils as _sku
skills = tmp_path / "skills"
d = _mk_skill(
skills, f"{sku.ORG_MIRROR_DIR_NAME}/org-1/shared-x", name="shared-x"
)
_mark_active(skills, "org-1")
monkeypatch.setattr(smt, "_skills_dir", lambda: skills)
monkeypatch.setattr(
_sku, "get_all_skills_dirs", lambda: [skills], raising=True
)
return smt, skills, d
def test_patch_is_allowed_and_applied(self, tmp_path, monkeypatch):
smt, _skills, d = self._org_skill(tmp_path, monkeypatch)
result = smt._patch_skill("shared-x", "body", "improved")
assert result["success"] is True, result.get("error")
assert "improved" in (d / "SKILL.md").read_text(encoding="utf-8")
def test_edit_tells_the_user_how_to_share_it_back(self, tmp_path, monkeypatch):
smt, _skills, _d = self._org_skill(tmp_path, monkeypatch)
result = smt._patch_skill("shared-x", "body", "improved")
# Without auto-propose the edit stays local, and the tool result must
# say so AND name the command — otherwise the improvement is stranded.
assert "propose" in (result.get("org_sharing") or "")
def test_delete_is_still_refused(self, tmp_path, monkeypatch):
smt, _skills, d = self._org_skill(tmp_path, monkeypatch)
guard = smt._org_mirror_write_guard("shared-x", d, "delete")
assert guard is not None and guard["success"] is False
assert "admin" in guard["error"]
def test_curation_is_allowed(self, tmp_path, monkeypatch):
from tools import skill_usage as su
skills = tmp_path / "skills"
d = _mk_skill(
skills, f"{sku.ORG_MIRROR_DIR_NAME}/org-1/shared-x", name="shared-x"
)
monkeypatch.setattr(su, "_skills_dir", lambda: skills)
# The curator must be able to improve shared skills — they are the
# highest-leverage ones in the system.
assert su.is_curation_eligible("shared-x", d) is True
class TestOrgPullIsWiredIn:
"""Guards the integration gap that unit tests structurally cannot catch.
The org pull functions were fully implemented and unit-tested while having
ZERO runtime callers, so org skills never loaded for anyone. Testing the
functions directly could never surface that. These tests assert the CALL
SITES exist, so the feature can't silently become dead code again.
"""
def test_session_startup_calls_maybe_pull_org_skills(self):
import pathlib
cli_src = (
pathlib.Path(__file__).resolve().parents[2] / "cli.py"
).read_text(encoding="utf-8")
assert "maybe_pull_org_skills" in cli_src, (
"cli.py session startup must call maybe_pull_org_skills() — "
"without a call site the org mirror is never populated and org "
"skills never load (the function being importable is not enough)."
)
# It must sit alongside the personal pull, not replace it.
assert "maybe_pull_skills" in cli_src
def test_sync_pull_command_refreshes_org_mirror(self):
import pathlib
main_src = (
pathlib.Path(__file__).resolve().parents[2]
/ "hermes_cli"
/ "main.py"
).read_text(encoding="utf-8")
assert "maybe_pull_org_skills" in main_src, (
"`hermes sync pull` must also refresh the org mirror."
)
def test_sync_status_exposes_org_state(self):
from tools import skills_sync_client as ssc
status = ssc.sync_status()
# These keys must always be present so a user can tell whether the org
# workflow applies to them, rather than it being invisible.
for key in ("org_available", "org_id", "org_role", "org_skills"):
assert key in status, f"sync status must expose {key!r}"
def test_no_internal_jargon_in_user_facing_strings(self):
"""User-visible help/errors must not leak internal design coordinates."""
import pathlib
import re
root = pathlib.Path(__file__).resolve().parents[2]
targets = [
root / "hermes_cli" / "subcommands" / "sync.py",
root / "hermes_cli" / "subcommands" / "skills.py",
]
banned = re.compile(
r"\(M[12]\)|\bHSP\b|HSP/1|§[0-9]|DEV-PHASE|hsp-1-contract"
)
for path in targets:
for i, line in enumerate(path.read_text(encoding="utf-8").split("\n"), 1):
if "help=" in line or "description=" in line:
assert not banned.search(line), (
f"{path.name}:{i} leaks internal jargon to users: {line.strip()}"
)
class TestSkillSyncIsOneCommand:
"""Every Skill Sync verb lives under `hermes sync` for launch.
The surface is deliberately encapsulated: one command to learn, one to
document, and top-level `sync` stays free of skill-management verbs that
belong elsewhere. `propose` in particular used to sit under `hermes
skills`, which split one feature across two commands.
"""
def _src(self, *parts):
import pathlib
return (
pathlib.Path(__file__).resolve().parents[2].joinpath(*parts)
).read_text(encoding="utf-8")
def test_propose_is_a_sync_subcommand(self):
sync_src = self._src("hermes_cli", "subcommands", "sync.py")
assert '"propose"' in sync_src, (
"`propose` must be a `hermes sync` subcommand."
)
def test_propose_is_not_under_skills(self):
skills_src = self._src("hermes_cli", "subcommands", "skills.py")
assert '"propose"' not in skills_src, (
"`propose` must NOT remain under `hermes skills` — Skill Sync is "
"one command for launch."
)
def test_sync_usage_lists_propose(self):
main_src = self._src("hermes_cli", "main.py")
usage_start = main_src.index("usage: hermes sync ")
usage_block = main_src[usage_start : usage_start + 1400]
assert "propose" in usage_block, (
"`hermes sync` usage must list the propose verb."
)
class TestLocalEditsSurviveOrgUpdates:
"""Ben's requirement: local edits are never silently overwritten.
An org pull materializes the shared set. Before this, it `rmtree`'d each
skill dir and re-wrote it, so any local improvement vanished on the next
session start with no warning. Now a locally-modified skill is skipped
and reported as a conflict for the user to resolve deliberately.
"""
def _mirror(self, tmp_path, monkeypatch, body="original\n"):
from tools import skills_sync_client as ssc
skills = tmp_path / "skills"
d = _mk_skill(
skills,
f"{sku.ORG_MIRROR_DIR_NAME}/org-1/shared-x",
name="shared-x",
body=body,
)
_mark_active(skills, "org-1")
monkeypatch.setattr(ssc, "_skills_dir", lambda: skills)
monkeypatch.setattr(
ssc, "_org_dir", lambda: skills / sku.ORG_MIRROR_DIR_NAME
)
return ssc, skills, d
def test_unmodified_skill_is_not_flagged(self, tmp_path, monkeypatch):
ssc, _skills, d = self._mirror(tmp_path, monkeypatch)
ssc._write_org_baseline(
"org-1",
{"shared-x": {"fingerprint": ssc._skill_dir_fingerprint(d), "tree": "t1"}},
)
assert ssc.org_skill_is_locally_modified("shared-x", "org-1") is False
assert ssc.list_locally_modified_org_skills("org-1") == []
def test_edited_skill_is_detected(self, tmp_path, monkeypatch):
ssc, _skills, d = self._mirror(tmp_path, monkeypatch)
ssc._write_org_baseline(
"org-1",
{"shared-x": {"fingerprint": ssc._skill_dir_fingerprint(d), "tree": "t1"}},
)
(d / "SKILL.md").write_text("---\nname: shared-x\n---\nEDITED\n", encoding="utf-8")
assert ssc.org_skill_is_locally_modified("shared-x", "org-1") is True
assert ssc.list_locally_modified_org_skills("org-1") == ["shared-x"]
def test_missing_baseline_does_not_cry_wolf(self, tmp_path, monkeypatch):
ssc, _skills, _d = self._mirror(tmp_path, monkeypatch)
# Mirror pulled before baselines existed — must not be reported as
# modified (that would block every update with a phantom conflict).
assert ssc.org_skill_is_locally_modified("shared-x", "org-1") is False
def test_fingerprint_is_content_based_not_mtime(self, tmp_path, monkeypatch):
import os
import time
ssc, _skills, d = self._mirror(tmp_path, monkeypatch)
before = ssc._skill_dir_fingerprint(d)
time.sleep(0.01)
os.utime(d / "SKILL.md", None) # touch: mtime changes, content doesn't
assert ssc._skill_dir_fingerprint(d) == before
def test_auto_propose_defaults_off(self, monkeypatch):
from tools import skills_sync_client as ssc
monkeypatch.delenv("HERMES_SYNC_ORG_AUTO_PROPOSE", raising=False)
monkeypatch.setattr(
"hermes_cli.config.load_config", lambda: {}, raising=False
)
# Default must be OFF: silently pushing every agent edit to the whole
# organisation is not a safe default.
assert ssc.sync_org_auto_propose() is False
def test_auto_propose_can_be_enabled_by_env(self, monkeypatch):
from tools import skills_sync_client as ssc
monkeypatch.setenv("HERMES_SYNC_ORG_AUTO_PROPOSE", "1")
assert ssc.sync_org_auto_propose() is True

View File

@ -0,0 +1,403 @@
"""Slack relay: edit-based streaming of the reply must fire in a DM.
Reported symptom (live): agent responses stream progressively (edit-based) in a
Slack THREAD but arrive FLAT (single message, no progressive edits) in a Slack
DM/home over the relay.
Root cause: a DM turn's streaming reply is sent with
``reply_to = <triggering message ts>`` (the stream consumer's
``initial_reply_to_id`` its edit anchor). The connector's slackRestSender maps
a raw ``reply_to`` to a Slack ``thread_ts``, so a plain DM reply gets threaded
UNDER the user's message instead of posting flat at the DM root, and a threaded
first send loses the progressive edit streaming the user sees in a real thread.
Native ``SlackAdapter._resolve_thread_ts`` already suppresses this synthetic DM
thread anchor; the relay lane had no such disambiguation.
These are behaviour-contract tests: they assert how the outbound frame relates to
the chat type + thread metadata (the invariant the connector depends on), not a
snapshot. They drive the REAL ``RelayAdapter`` + ``GatewayStreamConsumer`` +
``StubConnector`` end to end.
"""
from __future__ import annotations
import pytest
from gateway.config import Platform, PlatformConfig
from gateway.platforms.base import MessageEvent, MessageType
from gateway.relay.adapter import RelayAdapter
from gateway.relay.descriptor import CONTRACT_VERSION, CapabilityDescriptor
from gateway.session import SessionSource
from gateway.stream_consumer import GatewayStreamConsumer, StreamConsumerConfig
from tests.gateway.relay.stub_connector import StubConnector
def _slack_desc(**kw) -> CapabilityDescriptor:
base = dict(
contract_version=CONTRACT_VERSION,
platform="slack",
label="Slack",
max_message_length=4000,
supports_draft_streaming=False,
supports_edit=True,
supports_threads=True,
markdown_dialect="mrkdwn",
len_unit="chars",
emoji="\U0001f4ac",
platform_hint="",
pii_safe=False,
)
base.update(kw)
return CapabilityDescriptor(**base)
def _wire(chat_id: str, chat_type: str, *, user_id="U1", scope_id=None):
"""A RelayAdapter fronting Slack, with inbound scope captured for chat_id."""
stub = StubConnector(_slack_desc())
adapter = RelayAdapter(PlatformConfig(), _slack_desc(), transport=stub)
src = SessionSource(
platform=Platform.SLACK,
chat_id=chat_id,
chat_type=chat_type,
user_id=user_id,
scope_id=scope_id,
)
adapter._capture_scope(
MessageEvent(text="hi", source=src, message_type=MessageType.TEXT)
)
return adapter, stub
# ---------------------------------------------------------------------------
# The pure disambiguation contract (RelayAdapter.send)
# ---------------------------------------------------------------------------
@pytest.mark.asyncio
async def test_slack_dm_reply_keeps_anchor_in_thread_per_message_mode():
"""Default mode (reply_in_thread=True, thread-per-message): the triggering
ts reply_to is the final reply's ONLY threading signal (base.py builds
metadata from source.thread_id, which is None for a top-level DM) it
must be KEPT so the final message lands in the per-message thread with
the progress bubbles (2026-07-27 mixed-placement report)."""
adapter, stub = _wire("D1", "dm")
await adapter.send("D1", "the answer", reply_to="1700.0001")
assert len(stub.sent) == 1
frame = stub.sent[0]
assert frame["op"] == "send"
assert frame["reply_to"] == "1700.0001", (
"thread-per-message: the triggering ts anchors the final reply"
)
# The connector's Slack sender threads on metadata.thread_id ONLY
# (threadTs() never reads the frame's reply_to), so the surviving anchor
# must be promoted into metadata for the send to actually thread.
assert (frame["metadata"] or {}).get("thread_id") == "1700.0001"
@pytest.mark.asyncio
async def test_slack_dm_reply_drops_synthetic_anchor_in_flat_mode():
"""Flat mode (reply_in_thread=False): the synthetic self-anchor is dropped
so the reply posts flat at the DM root (native _resolve_thread_ts parity)
and no synthetic thread is invented (#18859)."""
adapter, stub = _wire("D1", "dm")
adapter.config.extra = {"reply_in_thread": False}
await adapter.send("D1", "the answer", reply_to="1700.0001")
frame = stub.sent[0]
assert frame["reply_to"] is None
assert "thread_id" not in (frame["metadata"] or {})
assert "thread_ts" not in (frame["metadata"] or {})
@pytest.mark.asyncio
async def test_slack_dm_reply_with_real_thread_keeps_anchor():
"""A DM turn that IS inside a real thread (metadata carries a distinct
thread_id) must keep threading the guard only drops the synthetic anchor."""
adapter, stub = _wire("D1", "dm")
await adapter.send(
"D1", "in thread", reply_to="1700.0002", metadata={"thread_id": "1699.9000"}
)
frame = stub.sent[0]
assert frame["reply_to"] == "1700.0002"
assert frame["metadata"]["thread_id"] == "1699.9000"
@pytest.mark.asyncio
async def test_slack_channel_top_level_reply_keeps_autothread_anchor():
"""A channel top-level reply carries thread_id (its own ts) in metadata when
autoThread is on; the DM-only guard must not touch it."""
adapter, stub = _wire("C1", "channel", scope_id="T1")
await adapter.send(
"C1", "channel reply", reply_to="1700.0003", metadata={"thread_id": "1700.0003"}
)
frame = stub.sent[0]
assert frame["reply_to"] == "1700.0003"
assert frame["metadata"]["thread_id"] == "1700.0003"
@pytest.mark.asyncio
async def test_non_slack_dm_reply_unchanged():
"""The disambiguation is Slack-scoped: a non-Slack relay chat keeps reply_to
(its connector owns its own threading semantics)."""
stub = StubConnector(_slack_desc(platform="discord"))
adapter = RelayAdapter(
PlatformConfig(), _slack_desc(platform="discord"), transport=stub
)
src = SessionSource(
platform=Platform.DISCORD, chat_id="dc1", chat_type="dm", user_id="U1"
)
adapter._capture_scope(
MessageEvent(text="hi", source=src, message_type=MessageType.TEXT)
)
await adapter.send("dc1", "hi", reply_to="msg-9")
assert stub.sent[0]["reply_to"] == "msg-9"
# ---------------------------------------------------------------------------
# End-to-end: the stream consumer keeps edit-streaming in a DM
# ---------------------------------------------------------------------------
async def _drive_stream(adapter, chat_id, *, metadata, initial_reply_to_id, chat_type):
cfg = StreamConsumerConfig(
edit_interval=0.0,
buffer_threshold=1,
transport="edit",
chat_type=chat_type,
)
consumer = GatewayStreamConsumer(
adapter=adapter,
chat_id=chat_id,
config=cfg,
metadata=metadata,
initial_reply_to_id=initial_reply_to_id,
)
# Feed progressive deltas, then finalize — mirrors the live delta callback.
for chunk in ("Hel", "lo ", "world", ". Done."):
consumer.on_delta(chunk)
consumer.finish()
await consumer.run()
return consumer
@pytest.mark.asyncio
async def test_slack_dm_stream_consumer_edits_own_ts_not_flat():
"""A Slack DM turn (chat_type='dm', no thread, metadata None) still builds a
stream consumer that keeps edit support and emits progressive EDITs of the
reply message the flat-DM regression contract.
The connector returns a real message_id for the flat first send, so edit
support must stay on and at least one edit op must be emitted (progressive
streaming), identical to a thread. No synthetic thread is created.
Runs in EXPLICIT flat mode (reply_in_thread=False) that is the mode this
contract belongs to; the default thread-per-message path is covered by
test_slack_dm_stream_consumer_threads_in_thread_per_message_mode."""
adapter, stub = _wire("D1", "dm")
adapter.config.extra = {"reply_in_thread": False}
consumer = await _drive_stream(
adapter,
"D1",
metadata=None, # DM: _status_thread_metadata is None in run.py
initial_reply_to_id="1700.0001", # the triggering message ts
chat_type="dm",
)
ops = [f["op"] for f in stub.sent]
# First a flat send; edit support stays on so progressive edits CAN flow
# (exact intermediate-frame timing is covered by the stream_consumer unit
# suite — here we assert the DM regression contract: streaming is not
# self-disabled and every edit targets the reply's own ts).
assert ops[0] == "send"
# Edit support survived: message_id set, not the __no_edit__ sentinel.
assert consumer.message_id and consumer.message_id != "__no_edit__"
assert consumer._edit_supported is True
first_send = stub.sent[0]
# The reply posts FLAT at the DM root — no synthetic thread anchor.
assert first_send["reply_to"] is None
assert "thread_id" not in (first_send["metadata"] or {})
assert "thread_ts" not in (first_send["metadata"] or {})
# reply_to_message_id (the mirrored self-anchor) is stripped too.
assert "reply_to_message_id" not in (first_send["metadata"] or {})
# Any edits that flowed target the same first-send ts (editing its own
# message), never a synthetic thread.
edit_ids = {f["message_id"] for f in stub.sent if f["op"] == "edit"}
assert edit_ids <= {stub.next_send_result["message_id"]}
@pytest.mark.asyncio
async def test_slack_thread_stream_consumer_still_threads_and_streams():
"""Regression guard: a Slack THREAD turn keeps its real thread_id AND streams
(the DM fix must not change the thread path)."""
adapter, stub = _wire("C1", "channel", scope_id="T1")
consumer = await _drive_stream(
adapter,
"C1",
metadata={"thread_id": "1699.9000"},
initial_reply_to_id="1700.0002",
chat_type="channel",
)
ops = [f["op"] for f in stub.sent]
assert ops[0] == "send"
assert consumer._edit_supported is True
first_send = stub.sent[0]
# Thread preserved: the real thread_id rides along and reply_to is kept.
assert first_send["metadata"]["thread_id"] == "1699.9000"
assert first_send["reply_to"] == "1700.0002"
@pytest.mark.asyncio
async def test_slack_dm_stream_consumer_threads_in_thread_per_message_mode():
"""Default mode: the DM stream's first send keeps the triggering-ts anchor
so the streamed final reply lands in the per-message thread; edits still
target the reply's own ts."""
adapter, stub = _wire("D1", "dm")
consumer = await _drive_stream(
adapter,
"D1",
metadata=None,
initial_reply_to_id="1700.0001",
chat_type="dm",
)
first_send = stub.sent[0]
assert first_send["op"] == "send"
assert first_send["reply_to"] == "1700.0001"
assert consumer.message_id and consumer.message_id != "__no_edit__"
edit_ids = {f["message_id"] for f in stub.sent if f["op"] == "edit"}
assert edit_ids <= {stub.next_send_result["message_id"]}
# ---------------------------------------------------------------------------
# The media lane obeys the SAME thread-anchor contract as the text lane.
#
# send() and _send_media() both egress through the connector's Slack sender,
# so an anchor resolved on only one of them threads an image under the user's
# DM message in flat mode, and loses the per-message thread in thread mode
# (threadTs() reads metadata only). Both lanes route through
# _apply_slack_thread_anchor; these pin that they stay in agreement.
# ---------------------------------------------------------------------------
def _media_wire(chat_id: str, chat_type: str):
"""Like _wire, but the descriptor advertises the send_media op."""
desc = _slack_desc(supported_ops=("send", "edit", "typing", "send_media"))
stub = StubConnector(desc)
adapter = RelayAdapter(PlatformConfig(), desc, transport=stub)
src = SessionSource(
platform=Platform.SLACK,
chat_id=chat_id,
chat_type=chat_type,
user_id="U1",
)
adapter._capture_scope(
MessageEvent(text="hi", source=src, message_type=MessageType.TEXT)
)
return adapter, stub
@pytest.mark.asyncio
async def test_slack_dm_media_keeps_and_promotes_anchor_in_thread_mode():
"""Thread-per-message: an image must land in the per-message thread. The
connector threads on metadata.thread_id only, so the surviving anchor has
to be promoted there a bare reply_to would post to the home channel."""
adapter, stub = _media_wire("D1", "dm")
await adapter.send_image("D1", "https://example.com/x.png", reply_to="1700.0001")
frame = [f for f in stub.sent if f["op"] == "send_media"][-1]
assert frame["reply_to"] == "1700.0001"
assert (frame["metadata"] or {}).get("thread_id") == "1700.0001", (
"media frame must carry the anchor where the connector reads it"
)
@pytest.mark.asyncio
async def test_slack_dm_media_drops_synthetic_anchor_in_flat_mode():
"""Flat mode: the media frame drops the synthetic self-anchor exactly as
the text lane does, so the image posts flat at the DM root instead of
threading under the user's message, and invents no thread (#18859)."""
adapter, stub = _media_wire("D1", "dm")
adapter.config.extra = {"reply_in_thread": False}
await adapter.send_image("D1", "https://example.com/x.png", reply_to="1700.0001")
frame = [f for f in stub.sent if f["op"] == "send_media"][-1]
assert frame["reply_to"] is None
assert "thread_id" not in (frame["metadata"] or {})
assert "thread_ts" not in (frame["metadata"] or {})
@pytest.mark.asyncio
async def test_slack_channel_media_anchor_untouched():
"""Non-DM chats are outside the synthetic-anchor rule: a channel media
send keeps its reply_to unchanged."""
adapter, stub = _media_wire("C1", "channel")
await adapter.send_image("C1", "https://example.com/x.png", reply_to="1700.0009")
frame = [f for f in stub.sent if f["op"] == "send_media"][-1]
assert frame["reply_to"] == "1700.0009"
@pytest.mark.asyncio
async def test_media_caller_metadata_not_mutated():
"""The anchor promotion must not leak into the caller's dict — media
helpers are called in loops with a shared metadata mapping."""
adapter, stub = _media_wire("D1", "dm")
caller_md = {"user_id": "U1"}
await adapter.send_image(
"D1", "https://example.com/x.png", reply_to="1700.0001", metadata=caller_md
)
assert caller_md == {"user_id": "U1"}, "caller metadata was mutated in place"
# ---------------------------------------------------------------------------
# Operator flags coerce exactly as the native Slack adapter's do.
# ---------------------------------------------------------------------------
@pytest.mark.parametrize(
"raw,expected",
[
(False, False),
("false", False),
("False", False),
(" no ", False),
("off", False),
("0", False),
(True, True),
("true", True),
("yes", True),
("on", True),
("1", True),
],
)
def test_relay_slack_flags_coerce_like_native(raw, expected):
"""A YAML-quoted "false" must turn these knobs OFF, matching native's
str().strip().lower() predicate. A bare bool() would read any non-empty
string as True and silently ignore the operator's off switch."""
adapter, _stub = _wire("D1", "dm")
adapter.config.extra = {
"slack": {
"reply_in_thread": raw,
"dm_top_level_threads_as_sessions": raw,
}
}
assert adapter._effective_reply_in_thread() is expected
assert adapter._dm_top_level_threads_as_sessions() is expected
def test_relay_slack_flags_default_true_when_absent():
"""Both knobs default ON when the operator sets nothing."""
adapter, _stub = _wire("D1", "dm")
adapter.config.extra = {}
assert adapter._effective_reply_in_thread() is True
assert adapter._dm_top_level_threads_as_sessions() is True
# ---------------------------------------------------------------------------
# The status clear targets the same thread the heartbeat set.
# ---------------------------------------------------------------------------
@pytest.mark.asyncio
async def test_typing_and_clear_share_one_status_anchor():
"""send_typing and stop_typing resolve the anchor through one helper: a
clear that no-ops threadless leaves the status line stuck until Slack's
own timeout."""
adapter, stub = _wire("D1", "dm")
adapter._last_inbound_ts_by_chat["D1"] = "1700.0001"
await adapter.send_typing("D1")
await adapter.stop_typing("D1")
typing_frames = [f for f in stub.sent if f["op"] == "typing"]
assert len(typing_frames) == 2
anchors = [(f["metadata"] or {}).get("thread_id") for f in typing_frames]
assert anchors == ["1700.0001", "1700.0001"], (
"the clear must target the thread the heartbeat set"
)

View File

@ -0,0 +1,486 @@
"""Slack relay: interactive prompts follow the turn's thread stamp.
The threading MODE (flat DM vs thread-per-message) is decided in exactly ONE
place: run.py's ``_resolve_progress_thread_id``, which reads
``platforms.slack.extra.reply_in_thread`` and encodes the verdict into the
outbound ``metadata`` stamp:
* flat mode -> the synthetic self-anchor is suppressed in run.py, so prompt
metadata arrives with NO ``thread_id`` and the card posts at the DM root;
* thread-per-message (default) -> ``metadata.thread_id`` is stamped for the
whole turn; on the FIRST turn it legitimately equals the triggering
message's ts (the synthetic root IS the thread).
The prompt lane must TRUST that stamp, like ``_resolve_reply_to_for_send``
does. Re-deriving the mode here (the old unconditional
``thread_id == message_id`` strip) exiled the approval card and its
resolved-state swap to the DM root while progress bubbles honoured the thread
(the 2026-07-27 mixed-placement report).
These are behaviour-contract tests: they assert how the outbound ``prompt``
frame relates to the inherited thread metadata (the invariant the connector
depends on), not a snapshot. They drive the REAL ``RelayAdapter`` +
``StubConnector`` end to end.
"""
from __future__ import annotations
import pytest
from gateway.config import Platform, PlatformConfig
from gateway.platforms.base import MessageEvent, MessageType
from gateway.relay.adapter import RelayAdapter
from gateway.relay.descriptor import CONTRACT_VERSION, CapabilityDescriptor
from gateway.session import SessionSource
from tests.gateway.relay.stub_connector import StubConnector
FULL_OPS = ("send", "edit", "typing", "get_chat_info", "send_media", "prompt", "react")
def _slack_desc(**kw) -> CapabilityDescriptor:
base = dict(
contract_version=CONTRACT_VERSION,
platform="slack",
label="Slack",
max_message_length=4000,
supports_draft_streaming=False,
supports_edit=True,
supports_threads=True,
markdown_dialect="mrkdwn",
len_unit="chars",
supported_ops=FULL_OPS,
)
base.update(kw)
return CapabilityDescriptor(**base)
def _wire(
chat_id: str,
chat_type: str,
*,
user_id="U1",
scope_id=None,
platform=Platform.SLACK,
):
"""A RelayAdapter fronting Slack, with inbound scope + chat_type captured."""
stub = StubConnector(_slack_desc())
adapter = RelayAdapter(PlatformConfig(), _slack_desc(), transport=stub)
src = SessionSource(
platform=platform,
chat_id=chat_id,
chat_type=chat_type,
user_id=user_id,
scope_id=scope_id,
)
adapter._capture_scope(
MessageEvent(text="hi", source=src, message_type=MessageType.TEXT)
)
return adapter, stub
def _last_prompt(stub) -> dict:
prompts = [f for f in stub.sent if f["op"] == "prompt"]
assert prompts, "expected a prompt op on the wire"
return prompts[-1]
# ---------------------------------------------------------------------------
# Flat mode: run.py stamps NO thread_id -> the card posts at the DM root.
# ---------------------------------------------------------------------------
@pytest.mark.asyncio
async def test_exec_approval_flat_mode_posts_at_dm_root():
"""Flat-DM turn (reply_in_thread=false): run.py suppressed the synthetic
anchor upstream, so prompt metadata has no thread_id and none appears on
the wire the card posts at the DM root."""
adapter, stub = _wire("D1", "dm", scope_id="T1")
md = {"message_id": "1700000000.000100", "scope_id": "T1"}
result = await adapter.send_exec_approval(
"D1", "rm -rf /tmp/x", "sess:1", description="deletes files", metadata=md
)
assert result.success is True
frame = _last_prompt(stub)
meta = frame["metadata"] or {}
assert "thread_id" not in meta
assert "thread_ts" not in meta
# reply_to on the outbound action stays unset — a root-level post.
assert frame["reply_to"] is None
# Tenant scope is preserved untouched (egress routing must not break).
assert meta.get("scope_id") == "T1"
# ---------------------------------------------------------------------------
# Thread-per-message mode, end-to-end placement contract: run.py stamps the
# turn's thread (first turn: the triggering message's own ts) and the adapter
# forwards prompt metadata UNTOUCHED — no re-derivation, no strip. Mixed
# placement (progress threaded, card at root) was the 2026-07-27 regression.
# ---------------------------------------------------------------------------
@pytest.mark.asyncio
async def test_exec_approval_forwards_run_py_thread_stamp_untouched():
"""The adapter must forward run.py's thread stamp verbatim: the approval
card posts INTO the stamped thread. Any adapter-side re-derivation or
strip exiled the card to the home channel (2026-07-27 report)."""
adapter, stub = _wire("D1", "dm", scope_id="T1")
md = {
"thread_id": "1700000000.000100",
"message_id": "1700000000.000100",
"scope_id": "T1",
}
result = await adapter.send_exec_approval(
"D1", "rm -rf /tmp/x", "sess:1", description="deletes files", metadata=md
)
assert result.success is True
frame = _last_prompt(stub)
meta = frame["metadata"] or {}
assert meta.get("thread_id") == "1700000000.000100", (
"first-turn self-anchor is the thread root; the prompt must honour it"
)
assert meta.get("scope_id") == "T1"
@pytest.mark.asyncio
async def test_clarify_forwards_run_py_thread_stamp_untouched():
adapter, stub = _wire("D1", "dm", scope_id="T1")
md = {
"thread_id": "1700000000.000200",
"message_id": "1700000000.000200",
"scope_id": "T1",
}
result = await adapter.send_clarify(
"D1", "Which env?", ["prod", "staging"], "cl-1", "sess:1", metadata=md
)
assert result.success is True
frame = _last_prompt(stub)
meta = frame["metadata"] or {}
assert meta.get("thread_id") == "1700000000.000200"
assert meta.get("scope_id") == "T1"
@pytest.mark.asyncio
async def test_slash_confirm_forwards_run_py_thread_stamp_untouched():
"""The forward-untouched rule covers every prompt surface (single
_send_prompt choke point)."""
adapter, stub = _wire("D1", "dm")
md = {"thread_id": "1700000000.000300", "message_id": "1700000000.000300"}
await adapter.send_slash_confirm(
"D1", "Reload MCP", "invalidates cache", "s", "cf-1", metadata=md
)
frame = _last_prompt(stub)
assert (frame["metadata"] or {}).get("thread_id") == "1700000000.000300"
# ---------------------------------------------------------------------------
# Regression guards: a REAL thread and non-DM / non-Slack chats are untouched
# ---------------------------------------------------------------------------
@pytest.mark.asyncio
async def test_exec_approval_in_real_thread_keeps_thread_id():
"""A DM prompt raised inside a REAL thread (thread_id distinct from the
triggering message ts) stays in that thread."""
adapter, stub = _wire("D1", "dm", scope_id="T1")
md = {
"thread_id": "1699000000.999000",
"message_id": "1700000000.000100",
"scope_id": "T1",
}
await adapter.send_exec_approval("D1", "cmd", "s", metadata=md)
frame = _last_prompt(stub)
assert frame["metadata"]["thread_id"] == "1699000000.999000"
@pytest.mark.asyncio
async def test_channel_approval_keeps_thread_id():
"""A Slack CHANNEL prompt keeps its thread_id (autoThread / real thread)."""
adapter, stub = _wire("C1", "channel", scope_id="T1")
md = {
"thread_id": "1700000000.000400",
"message_id": "1700000000.000400",
"scope_id": "T1",
}
await adapter.send_exec_approval("C1", "cmd", "s", metadata=md)
frame = _last_prompt(stub)
assert frame["metadata"]["thread_id"] == "1700000000.000400"
@pytest.mark.asyncio
async def test_non_slack_dm_approval_keeps_thread_id():
"""A non-Slack relay DM keeps thread_id (its connector owns its own
threading semantics)."""
adapter, stub = _wire("dc1", "dm", platform=Platform.DISCORD)
md = {"thread_id": "9000", "message_id": "9000"}
await adapter.send_exec_approval("dc1", "cmd", "s", metadata=md)
frame = _last_prompt(stub)
assert frame["metadata"]["thread_id"] == "9000"
# ---------------------------------------------------------------------------
# Rich status: the relay advertises Slack's text status line and carries
# the live per-tool phrase on the typing frame (native set_status_text parity).
# ---------------------------------------------------------------------------
@pytest.mark.asyncio
async def test_slack_relay_advertises_status_text():
adapter, _stub = _wire("D1", "dm")
assert adapter.supports_status_text is True
@pytest.mark.asyncio
async def test_non_slack_relay_does_not_advertise_status_text():
stub = StubConnector(_slack_desc(platform="discord"))
adapter = RelayAdapter(
PlatformConfig(), _slack_desc(platform="discord"), transport=stub
)
assert adapter.supports_status_text is False
@pytest.mark.asyncio
async def test_typing_carries_live_status_phrase():
"""set_status_text() -> the next typing frame carries the phrase as
content; clearing it (None) reverts to a content-less heartbeat frame
(never an empty string, which is Slack's explicit clear)."""
adapter, stub = _wire("D1", "dm", scope_id="T1")
adapter.set_status_text("D1", "is running pytest…")
await adapter.send_typing("D1", metadata={"scope_id": "T1"})
typing = [f for f in stub.sent if f["op"] == "typing"]
assert typing and typing[-1].get("content") == "is running pytest…"
adapter.set_status_text("D1", None)
await adapter.send_typing("D1", metadata={"scope_id": "T1"})
typing = [f for f in stub.sent if f["op"] == "typing"]
assert "content" not in typing[-1], (
"cleared phrase must omit content (empty string means CLEAR on Slack)"
)
# ---------------------------------------------------------------------------
# Status thread anchor: typing frames synthesize the per-message thread
# root in thread-per-message mode (the status line is thread-only on Slack).
# ---------------------------------------------------------------------------
def _wire_with_ts(chat_id, chat_type, message_id, **kw):
adapter, stub = _wire(chat_id, chat_type, **kw)
src = SessionSource(
platform=Platform.SLACK, chat_id=chat_id, chat_type=chat_type,
user_id="U1", scope_id=kw.get("scope_id"),
)
ev = MessageEvent(
text="hi", source=src, message_type=MessageType.TEXT, message_id=message_id
)
adapter._capture_scope(ev)
return adapter, stub
@pytest.mark.asyncio
async def test_typing_synthesizes_thread_anchor_in_thread_mode():
"""Top-level DM turn, thread-per-message mode: the typing frame gains the
triggering ts as thread_id so the connector's setStatus targets the
per-message thread instead of no-oping threadless."""
adapter, stub = _wire_with_ts("D1", "dm", "1700.0042")
await adapter.send_typing("D1", metadata=None)
typing = [f for f in stub.sent if f["op"] == "typing"]
assert typing and typing[-1]["metadata"].get("thread_id") == "1700.0042"
@pytest.mark.asyncio
async def test_typing_flat_mode_status_anchors_to_trigger_ts_by_default():
"""Flat-DM liveliness: the STATUS still anchors to the triggering ts
(renders in the footer space, no message artifact) while replies stay
flat the send lane strips its anchors, so placement cannot inherit this."""
adapter, stub = _wire_with_ts("D1", "dm", "1700.0042")
adapter.config.extra = {"reply_in_thread": False}
await adapter.send_typing("D1", metadata=None)
typing = [f for f in stub.sent if f["op"] == "typing"]
assert typing and typing[-1]["metadata"].get("thread_id") == "1700.0042"
@pytest.mark.asyncio
async def test_typing_anchors_unconditionally_in_both_modes():
"""Liveliness is not a preference: the status anchors whenever an inbound
ts exists, regardless of reply_in_thread. Placement safety comes from the
send-side anchor strip, not from suppressing the status."""
for extra in ({}, {"slack": {"reply_in_thread": False}}):
adapter, stub = _wire_with_ts("D1", "dm", "1700.0042")
adapter.config.extra = extra
await adapter.send_typing("D1", metadata=None)
typing = [f for f in stub.sent if f["op"] == "typing"]
assert typing and typing[-1]["metadata"].get("thread_id") == "1700.0042"
@pytest.mark.asyncio
async def test_flat_mode_sends_stay_flat_with_status_anchor_active():
"""The liveliness anchor must NOT leak into reply placement: sends in
flat mode still strip the synthetic anchor (send-lane contract)."""
adapter, stub = _wire_with_ts("D1", "dm", "1700.0042")
adapter.config.extra = {"reply_in_thread": False}
await adapter.send_typing("D1", metadata=None)
await adapter.send("D1", "the answer", reply_to="1700.0042")
frame = [f for f in stub.sent if f["op"] == "send"][-1]
assert frame["reply_to"] is None
assert "thread_id" not in (frame["metadata"] or {})
@pytest.mark.asyncio
async def test_typing_honours_real_thread_anchor():
"""Metadata that already names a thread wins over the synthetic cache."""
adapter, stub = _wire_with_ts("D1", "dm", "1700.0042")
await adapter.send_typing("D1", metadata={"thread_id": "1699.9000"})
typing = [f for f in stub.sent if f["op"] == "typing"]
assert typing[-1]["metadata"]["thread_id"] == "1699.9000"
@pytest.mark.asyncio
async def test_stop_typing_clear_targets_same_synthesized_thread():
"""The clear frame targets the same synthesized thread as the heartbeat
(else the status line sticks)."""
adapter, stub = _wire_with_ts("D1", "dm", "1700.0042")
await adapter.send_typing("D1", metadata=None)
await adapter.stop_typing("D1", metadata=None)
clears = [
f for f in stub.sent if f["op"] == "typing" and f.get("content") == ""
]
assert clears and clears[-1]["metadata"].get("thread_id") == "1700.0042"
# ---------------------------------------------------------------------------
# Session keying: a top-level Slack DM message gets its own ts stamped as
# source.thread_id (native inbound parity) so each message keys a FRESH
# session in thread-per-message mode; flat mode and real threads untouched.
# ---------------------------------------------------------------------------
def _inbound_event(chat_id="D1", message_id="1700.0100", thread_id=None):
src = SessionSource(
platform=Platform.SLACK, chat_id=chat_id, chat_type="dm",
user_id="U1", scope_id="T1", thread_id=thread_id,
)
return MessageEvent(
text="hi", source=src, message_type=MessageType.TEXT,
message_id=message_id,
)
def test_top_level_dm_gets_session_thread_stamp():
adapter, _ = _wire("D1", "dm")
ev = _inbound_event(message_id="1700.0100")
adapter._stamp_slack_session_thread(ev)
assert ev.source.thread_id == "1700.0100"
def test_two_top_level_messages_key_distinct_sessions():
from gateway.session import build_session_key
adapter, _ = _wire("D1", "dm")
e1 = _inbound_event(message_id="1700.0100")
e2 = _inbound_event(message_id="1700.0200")
adapter._stamp_slack_session_thread(e1)
adapter._stamp_slack_session_thread(e2)
k1 = build_session_key(e1.source)
k2 = build_session_key(e2.source)
assert k1 != k2, "each top-level message must be its own session"
def test_real_thread_reply_keeps_its_thread_session():
adapter, _ = _wire("D1", "dm")
ev = _inbound_event(message_id="1700.0300", thread_id="1700.0100")
adapter._stamp_slack_session_thread(ev)
assert ev.source.thread_id == "1700.0100", (
"an in-thread reply must keep resolving to its thread's session"
)
def test_flat_mode_keeps_shared_dm_session():
adapter, _ = _wire("D1", "dm")
adapter.config.extra = {"reply_in_thread": False}
ev = _inbound_event(message_id="1700.0400")
adapter._stamp_slack_session_thread(ev)
assert ev.source.thread_id is None, (
"flat mode: shared rolling DM session (steer/queue) is intended UX"
)
def test_nested_relay_slack_config_subset_wins():
"""Enterprise knob shape: platforms.relay.extra.slack.reply_in_thread."""
adapter, _ = _wire("D1", "dm")
adapter.config.extra = {"slack": {"reply_in_thread": False}}
assert adapter._effective_reply_in_thread() is False
adapter.config.extra = {"slack": {"reply_in_thread": True}}
assert adapter._effective_reply_in_thread() is True
# Legacy flat key still honoured when no nested object exists.
adapter.config.extra = {"reply_in_thread": False}
assert adapter._effective_reply_in_thread() is False
# Default: thread-per-message.
adapter.config.extra = {}
assert adapter._effective_reply_in_thread() is True
# ---------------------------------------------------------------------------
# Cross-module boundary pin (review 2026-07-28): the adapter deliberately has
# NO prompt-side strip — flat-mode placement depends entirely on run.py's
# _resolve_progress_thread_id suppressing the synthetic self-anchor upstream.
# If that suppression regresses, prompt cards silently thread again. These
# tests pin the boundary in BOTH modes so the coupling is load-bearing.
# ---------------------------------------------------------------------------
def test_run_py_suppresses_self_anchor_in_flat_mode():
from gateway.run import _resolve_progress_thread_id
# Flat mode + synthetic self-anchor (thread_id == own message id) => None:
# prompt/progress metadata arrives at the adapter with NO thread anchor.
assert (
_resolve_progress_thread_id(
"slack", "1700.001", "1700.001", reply_in_thread=False
)
is None
)
# Flat mode + REAL thread (ids differ) => the real thread survives.
assert (
_resolve_progress_thread_id(
"slack", "1699.000", "1700.001", reply_in_thread=False
)
== "1699.000"
)
def test_run_py_keeps_self_anchor_in_thread_mode():
from gateway.run import _resolve_progress_thread_id
# Thread-per-message mode: the first-turn self-anchor IS the thread root
# and must flow through to the adapter unchanged.
assert (
_resolve_progress_thread_id(
"slack", "1700.001", "1700.001", reply_in_thread=True
)
== "1700.001"
)
# No source thread at all: Slack synthesizes the root from the message id.
assert (
_resolve_progress_thread_id("slack", None, "1700.001", reply_in_thread=True)
== "1700.001"
)
# ---------------------------------------------------------------------------
# Native parity escape hatch: platforms.relay.extra.slack.
# dm_top_level_threads_as_sessions=false keeps threaded replies but ONE
# rolling DM session (mirrors native SlackAdapter._dm_top_level_threads_as_sessions).
# Without the knob, reply_in_thread alone couples placement AND session
# keying — a posture native operators can express and relay ones could not.
# ---------------------------------------------------------------------------
@pytest.mark.asyncio
async def test_session_stamp_opt_out_keeps_rolling_dm_session():
adapter, stub = _wire("D1", "dm")
adapter.config.extra = {
"slack": {
"reply_in_thread": True,
"dm_top_level_threads_as_sessions": False,
}
}
event = _inbound_event("D1", message_id="1700.0001", thread_id=None)
adapter._stamp_slack_session_thread(event)
assert getattr(event.source, "thread_id", None) is None, (
"opt-out: top-level DM must NOT be stamped — one rolling session"
)
@pytest.mark.asyncio
async def test_session_stamp_default_remains_per_message():
adapter, stub = _wire("D1", "dm")
adapter.config.extra = {"slack": {"reply_in_thread": True}}
event = _inbound_event("D1", message_id="1700.0002", thread_id=None)
adapter._stamp_slack_session_thread(event)
assert getattr(event.source, "thread_id", None) == "1700.0002", (
"default (native parity): per-message sessions stay on"
)

View File

@ -38,7 +38,6 @@ class TestGeneric400Heuristic:
a.client = MagicMock()
a._cached_system_prompt = "You are helpful."
a._use_prompt_caching = False
a.tool_delay = 0
a.compression_enabled = False
return a

View File

@ -94,7 +94,6 @@ def agent():
a.client = MagicMock()
a._cached_system_prompt = "You are helpful."
a._use_prompt_caching = False
a.tool_delay = 0
# Default matches production (`compression.enabled` defaults to True).
# Overflow-recovery tests below verify that 413 / context-overflow
# errors DO trigger compression; the disabled-path behavior is

View File

@ -80,7 +80,6 @@ def test_substantive_tool_only_turn_invalidates_older_housekeeping_fallback():
agent._cached_system_prompt = "You are helpful."
agent._use_prompt_caching = False
agent.tool_delay = 0
agent.compression_enabled = False
agent.save_trajectories = False
agent.valid_tool_names = {"todo", "web_search"}

View File

@ -34,7 +34,6 @@ def _make_agent() -> AIAgent:
skip_memory=True,
)
agent.client = MagicMock()
agent.tool_delay = 0
agent._flush_messages_to_session_db = MagicMock()
return agent

View File

@ -72,7 +72,6 @@ def _make_agent() -> AIAgent:
a.client = MagicMock()
a._cached_system_prompt = "You are helpful."
a._use_prompt_caching = False
a.tool_delay = 0
a.compression_enabled = False
a.save_trajectories = False
return a

View File

@ -198,7 +198,6 @@ def loop_agent():
a.client = MagicMock()
a._cached_system_prompt = "You are helpful."
a._use_prompt_caching = False
a.tool_delay = 0
a.compression_enabled = False
a.save_trajectories = False
return a

View File

@ -627,6 +627,25 @@ class TestInit:
assert agent.api_mode == "anthropic_messages"
mock_anthropic.Anthropic.assert_called_once()
def test_tool_delay_kwarg_is_deprecated_noop(self):
"""tool_delay stays accepted for compatibility but warns and is ignored."""
with (
patch("run_agent.get_tool_definitions", return_value=[]),
patch("run_agent.check_toolset_requirements", return_value={}),
patch("run_agent.OpenAI"),
):
with pytest.warns(DeprecationWarning, match="tool_delay"):
a = AIAgent(
api_key="test-key-1234567890",
base_url="https://openrouter.ai/api/v1",
tool_delay=0,
quiet_mode=True,
skip_context_files=True,
skip_memory=True,
)
# The value is discarded — nothing downstream reads it anymore.
assert not hasattr(a, "tool_delay")
def test_prompt_caching_claude_openrouter(self):
"""Claude model via OpenRouter should enable prompt caching."""
with (
@ -1336,6 +1355,22 @@ class TestExecuteToolCalls:
assert messages[0]["role"] == "tool"
assert "search result" in messages[0]["content"]
def test_sequential_tool_calls_run_without_delay(self, agent):
"""Two sequential tool calls execute back-to-back with no sleep between them."""
tc1 = _mock_tool_call(name="web_search", arguments="{}", call_id="c1")
tc2 = _mock_tool_call(name="web_search", arguments="{}", call_id="c2")
mock_msg = _mock_assistant_msg(content="", tool_calls=[tc1, tc2])
messages = []
with (
patch("run_agent.handle_function_call", return_value="ok") as mock_hfc,
patch("agent.tool_executor.time.sleep") as mock_sleep,
):
agent._execute_tool_calls_sequential(mock_msg, messages, "task-1")
assert mock_hfc.call_count == 2
mock_sleep.assert_not_called()
tool_results = [m for m in messages if m["role"] == "tool"]
assert [m["tool_call_id"] for m in tool_results] == ["c1", "c2"]
def test_sequential_memory_remove_notifies_provider_with_tool_result(self, agent):
old_text = "stale preference entry"
tc = _mock_tool_call(
@ -2439,7 +2474,6 @@ class TestRunConversation:
"""Common setup for run_conversation tests."""
agent._cached_system_prompt = "You are helpful."
agent._use_prompt_caching = False
agent.tool_delay = 0
agent.compression_enabled = False
agent.save_trajectories = False
@ -3953,7 +3987,6 @@ class TestRetryExhaustion:
def _setup_agent(self, agent):
agent._cached_system_prompt = "You are helpful."
agent._use_prompt_caching = False
agent.tool_delay = 0
agent.compression_enabled = False
agent.save_trajectories = False
@ -5352,7 +5385,6 @@ class TestReasoningReplayForStrictProviders:
def _setup_agent(self, agent):
agent._cached_system_prompt = "You are helpful."
agent._use_prompt_caching = False
agent.tool_delay = 0
agent.compression_enabled = False
agent.save_trajectories = False

View File

@ -54,7 +54,6 @@ def _make_agent(*tool_names: str, max_iterations: int = 10, config: dict | None
agent.client = MagicMock()
agent._cached_system_prompt = "You are helpful."
agent._use_prompt_caching = False
agent.tool_delay = 0
agent.compression_enabled = False
agent.save_trajectories = False
return agent

View File

@ -74,7 +74,6 @@ def _make_agent():
agent.client = MagicMock()
agent._cached_system_prompt = "You are helpful."
agent._use_prompt_caching = False
agent.tool_delay = 0
agent.compression_enabled = False
agent.save_trajectories = False
return agent

View File

@ -52,7 +52,6 @@ def _make_agent(max_iterations: int = 10, config: dict | None = None) -> AIAgent
agent.client = MagicMock()
agent._cached_system_prompt = "You are helpful."
agent._use_prompt_caching = False
agent.tool_delay = 0
agent.compression_enabled = False
agent.save_trajectories = False
# No fallback chain so empty responses exhaust deterministically.

View File

@ -0,0 +1,978 @@
"""Tests for tools/skills_sync_client.py — the Skill Sync client.
Covers, against the frozen contract (~/src/specs/collective-wisdom/
the sync wire contract):
* content addressing (full 64-hex) + canonical JSON (§2.1, §2.5)
* the access gate (Nous admin) making sync inert
* the M1-D opt-in default (nothing syncs without the sync flag)
* object building (blob/tree/commit, exec mode, size limit)
* push (upload + CAS), pull (materialize), and the three-way merge / 409
conflict paths all against an in-process mock sync server.
The mock server implements the contract §3/§4 endpoint shapes with an
in-memory object store + ref table. No live server, no network.
"""
import hashlib
import json
import threading
from http.server import BaseHTTPRequestHandler, HTTPServer
from pathlib import Path
import pytest
import tools.skills_sync_client as ssc
# ---------------------------------------------------------------------------
# In-process mock sync server (read + write endpoints)
# ---------------------------------------------------------------------------
class _MockState:
def __init__(self):
self.objects = {} # hash -> (kind, bytes)
self.refs = {} # name -> commit hash
self.hsp_version = "1"
self.max_object_bytes = 26214400
self.force_conflict_once = False # inject a 409 on the next CAS
# M2 org behavior (contract §11): advertise the "org" feature and,
# when org_role_admin is False, convert org-HEAD CAS to 202 proposals.
self.org_feature = True
self.org_role_admin = True
self.proposals = [] # [{n, to, base}]
def _make_handler(state: _MockState):
class Handler(BaseHTTPRequestHandler):
def log_message(self, format, *args): # silence
pass
def _json(self, code, obj, extra_headers=None):
body = json.dumps(obj).encode("utf-8")
self.send_response(code)
self.send_header("Content-Type", "application/json")
for k, v in (extra_headers or {}).items():
self.send_header(k, v)
self.send_header("Content-Length", str(len(body)))
self.end_headers()
self.wfile.write(body)
def do_GET(self):
path = self.path.split("?", 1)[0]
query = ""
if "?" in self.path:
query = self.path.split("?", 1)[1]
if path == "/v1/sync/capabilities":
features = ["personal"] + (["org"] if state.org_feature else [])
return self._json(200, {
"hsp_version": state.hsp_version,
"features": features,
"max_object_bytes": state.max_object_bytes,
"hash_alg": "sha256",
"auth": "bearer",
})
if path == "/v1/sync/refs":
prefix = ""
for part in query.split("&"):
if part.startswith("prefix="):
from urllib.parse import unquote
prefix = unquote(part[len("prefix="):])
refs = [
{"name": n, "hash": h}
for n, h in state.refs.items()
if n.startswith(prefix)
]
return self._json(200, {"refs": refs})
if path.startswith("/v1/sync/objects/"):
obj_hash = path[len("/v1/sync/objects/"):]
if obj_hash not in state.objects:
return self._json(404, {"error": "not_found"})
kind, data = state.objects[obj_hash]
if kind == ssc.KIND_BLOB:
self.send_response(200)
self.send_header("Content-Type", "application/octet-stream")
self.send_header("X-HSP-Object-Type", "blob")
self.send_header("Content-Length", str(len(data)))
self.end_headers()
self.wfile.write(data)
return
self.send_response(200)
self.send_header("Content-Type", "application/json")
self.send_header("X-HSP-Object-Type", kind)
self.send_header("Content-Length", str(len(data)))
self.end_headers()
self.wfile.write(data)
return
self._json(404, {"error": "unknown"})
def do_POST(self):
length = int(self.headers.get("Content-Length", 0))
raw = self.rfile.read(length) if length else b""
path = self.path.split("?", 1)[0] # e.g. /v1/sync/objects?scope=org
if path == "/v1/sync/objects":
return self._handle_put_objects(raw)
if path.startswith("/v1/sync/refs/"):
return self._handle_cas(raw)
self._json(404, {"error": "unknown"})
def _handle_put_objects(self, raw):
# multipart/form-data: parse parts (field=hash, filename=type,
# body=raw bytes). The server recomputes each hash and 422s on
# mismatch (contract §4.2).
ctype = self.headers.get("Content-Type", "")
if "multipart/form-data" not in ctype:
return self._json(400, {"error": "expected multipart"})
boundary = ctype.split("boundary=", 1)[1].encode("ascii")
accepted, already = [], []
parts = raw.split(b"--" + boundary)
for part in parts:
# Only trim the delimiter framing: a leading CRLF and a
# trailing CRLF. Do NOT strip() the whole part -- that would
# also eat legitimate trailing newlines from the object bytes.
if part.startswith(b"\r\n"):
part = part[2:]
if part.endswith(b"\r\n"):
part = part[:-2]
if not part or part == b"--":
continue
if b"\r\n\r\n" not in part:
continue
headers_blob, body = part.split(b"\r\n\r\n", 1)
hdr_text = headers_blob.decode("utf-8", "replace")
claimed_hash = None
kind = None
for line in hdr_text.split("\r\n"):
if line.lower().startswith("content-disposition"):
for token in line.split(";"):
token = token.strip()
if token.startswith('name="'):
claimed_hash = token[len('name="'):-1]
elif token.startswith('filename="'):
kind = token[len('filename="'):-1]
if claimed_hash is None:
continue
real = "sha256:" + hashlib.sha256(body).hexdigest()
if real != claimed_hash:
return self._json(422, {
"error": "hash_mismatch", "claimed": claimed_hash,
})
if claimed_hash in state.objects:
already.append(claimed_hash)
else:
state.objects[claimed_hash] = (kind, body)
accepted.append(claimed_hash)
return self._json(200, {"accepted": accepted, "already_present": already})
def _handle_cas(self, raw):
from urllib.parse import unquote
name = unquote(self.path[len("/v1/sync/refs/"):])
body = json.loads(raw.decode("utf-8")) if raw else {}
frm = body.get("from")
to = body.get("to")
# M2 (contract §11.5): a non-admin member's CAS on an org HEAD is
# accept-always converted to a proposal → 202.
if name.startswith("refs/org/") and not state.org_role_admin:
n = len(state.proposals) + 1
state.proposals.append({"n": n, "to": to, "base": frm})
org = name.split("/")[2]
prop_ref = f"refs/org/{org}/proposals/{n}"
state.refs[prop_ref] = to
return self._json(202, {"proposal_id": n, "ref": prop_ref})
if state.force_conflict_once:
state.force_conflict_once = False
return self._json(409, {"actual": state.refs.get(name, "")})
current = state.refs.get(name)
if current != frm:
return self._json(409, {"actual": current or ""})
state.refs[name] = to
return self._json(200, {"ref": name, "hash": to})
return Handler
@pytest.fixture
def mock_server():
state = _MockState()
server = HTTPServer(("127.0.0.1", 0), _make_handler(state))
thread = threading.Thread(target=server.serve_forever, daemon=True)
thread.start()
base = f"http://127.0.0.1:{server.server_address[1]}"
try:
yield base, state
finally:
server.shutdown()
server.server_close()
# ---------------------------------------------------------------------------
# Helpers
# ---------------------------------------------------------------------------
def _write_skill(skills_dir: Path, name: str, body: str = "# skill\n", *, category=None):
"""Create a minimal skill dir under skills_dir; return its path."""
parent = skills_dir / category if category else skills_dir
d = parent / name
d.mkdir(parents=True, exist_ok=True)
(d / "SKILL.md").write_text(
f"---\nname: {name}\ndescription: test\n---\n{body}", encoding="utf-8"
)
return d
def _jwt(claims: dict) -> str:
import jwt as _pyjwt
return _pyjwt.encode(claims, "x" * 32, algorithm="HS256")
# ---------------------------------------------------------------------------
# Content addressing & canonicalization (contract §2.1, §2.5, OI-5)
# ---------------------------------------------------------------------------
class TestAddressing:
def test_full_64_hex_address(self):
addr = ssc.wire_address(b"")
# sha256 of empty is the well-known e3b0... digest, full 64 hex.
assert addr == (
"sha256:e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
)
assert len(addr.split(":", 1)[1]) == 64
def test_address_differs_from_local_truncated_namespace(self):
# The wire full-64-hex must NOT equal the local truncated 16-hex form.
data = b"hello world"
full = ssc.wire_address(data)
truncated = "sha256:" + hashlib.sha256(data).hexdigest()[:16]
assert full != truncated
assert len(full.split(":")[1]) == 64
assert len(truncated.split(":")[1]) == 16
def test_canonical_json_sorted_no_whitespace(self):
out = ssc.canonical_json_bytes({"b": 1, "a": 2})
assert out == b'{"a":2,"b":1}'
assert b" " not in out
assert not out.endswith(b"\n")
def test_canonical_json_stable(self):
obj = {"type": "tree", "entries": [{"name": "x", "hash": "sha256:aa"}]}
assert ssc.canonical_json_bytes(obj) == ssc.canonical_json_bytes(dict(obj))
# ---------------------------------------------------------------------------
# Access gate (Nous admin) + per-skill opt-in
# ---------------------------------------------------------------------------
class TestDevGate:
def test_gate_open_with_claim(self, monkeypatch):
token = _jwt({"sub": "user1", "tool_gateway_admin": True})
monkeypatch.setattr(
ssc, "resolve_nous_runtime_credentials",
lambda **kw: {"api_key": token, "base_url": "https://x"}, raising=False,
)
# patch the lazily-imported symbol used inside resolve_identity
import hermes_cli.auth as auth_mod
monkeypatch.setattr(auth_mod, "resolve_nous_runtime_credentials",
lambda **kw: {"api_key": token, "base_url": "https://x"})
ident = ssc.resolve_identity()
assert ident["nous_admin"] is True
assert ident["owner"] == "user1"
def test_gate_closed_without_claim(self, monkeypatch):
token = _jwt({"sub": "user1"}) # no tool_gateway_admin
import hermes_cli.auth as auth_mod
monkeypatch.setattr(auth_mod, "resolve_nous_runtime_credentials",
lambda **kw: {"api_key": token, "base_url": "https://x"})
ident = ssc.resolve_identity()
assert ident["nous_admin"] is False
def test_gate_closed_when_claim_false(self, monkeypatch):
token = _jwt({"sub": "u", "tool_gateway_admin": False})
import hermes_cli.auth as auth_mod
monkeypatch.setattr(auth_mod, "resolve_nous_runtime_credentials",
lambda **kw: {"api_key": token, "base_url": "https://x"})
assert ssc.dev_gate_open() is False
def test_maybe_push_inert_when_gate_closed(self, monkeypatch):
token = _jwt({"sub": "u"})
import hermes_cli.auth as auth_mod
monkeypatch.setattr(auth_mod, "resolve_nous_runtime_credentials",
lambda **kw: {"api_key": token})
monkeypatch.setattr(ssc, "resolve_sync_base_url", lambda: "http://x")
# gate closed -> None (inert), never attempts a push
assert ssc.maybe_push_skills() is None
def test_maybe_pull_inert_when_not_logged_in(self, monkeypatch):
import hermes_cli.auth as auth_mod
def _raise(**kw):
raise RuntimeError("not logged in")
monkeypatch.setattr(auth_mod, "resolve_nous_runtime_credentials", _raise)
assert ssc.maybe_pull_skills() is None
# ---------------------------------------------------------------------------
# Object building (contract §2.2-§2.4)
# ---------------------------------------------------------------------------
class TestObjectBuilding:
def test_build_tree_blob_and_exec(self, tmp_path):
d = tmp_path / "skill"
d.mkdir()
(d / "SKILL.md").write_text("hello", encoding="utf-8")
script = d / "run.sh"
script.write_text("#!/bin/sh\necho hi\n", encoding="utf-8")
script.chmod(0o755)
objects = ssc.ObjectSet()
tree_hash = ssc.build_tree(d, objects, max_object_bytes=ssc.DEFAULT_MAX_OBJECT_BYTES)
assert tree_hash.startswith("sha256:")
# tree object present and canonical
kind, data = objects.objects[tree_hash]
assert kind == ssc.KIND_TREE
tree = json.loads(data)
entries = {e["name"]: e for e in tree["entries"]}
assert entries["SKILL.md"]["mode"] == ssc.MODE_FILE
assert entries["run.sh"]["mode"] == ssc.MODE_EXEC
# entries sorted by name (byte order)
names = [e["name"] for e in tree["entries"]]
assert names == sorted(names)
def test_build_tree_dedups_identical_blobs(self, tmp_path):
d = tmp_path / "skill"
(d / "a").mkdir(parents=True)
(d / "b").mkdir(parents=True)
(d / "a" / "f.txt").write_text("same", encoding="utf-8")
(d / "b" / "f.txt").write_text("same", encoding="utf-8")
objects = ssc.ObjectSet()
ssc.build_tree(d, objects, max_object_bytes=ssc.DEFAULT_MAX_OBJECT_BYTES)
blob_hashes = [h for h, (k, _) in objects.objects.items() if k == ssc.KIND_BLOB]
# only one unique blob for the identical "same" content
assert len(set(blob_hashes)) == 1
def test_build_tree_skips_symlink(self, tmp_path):
d = tmp_path / "skill"
d.mkdir()
(d / "real.txt").write_text("x", encoding="utf-8")
try:
(d / "link.txt").symlink_to(d / "real.txt")
except (OSError, NotImplementedError):
pytest.skip("symlinks unsupported here")
objects = ssc.ObjectSet()
tree_hash = ssc.build_tree(d, objects, max_object_bytes=ssc.DEFAULT_MAX_OBJECT_BYTES)
tree = json.loads(objects.objects[tree_hash][1])
names = [e["name"] for e in tree["entries"]]
assert "link.txt" not in names
assert "real.txt" in names
def test_build_tree_rejects_oversize_blob(self, tmp_path):
d = tmp_path / "skill"
d.mkdir()
(d / "big").write_bytes(b"x" * 100)
objects = ssc.ObjectSet()
with pytest.raises(ValueError):
ssc.build_tree(d, objects, max_object_bytes=10)
def test_build_commit_shape(self):
objects = ssc.ObjectSet()
c = ssc.build_commit(
"sha256:tree", ["sha256:p"], owner="o", device="dev",
message="m", objects=objects, ts="2026-07-18T00:00:00Z",
)
commit = json.loads(objects.objects[c][1])
assert commit["type"] == "commit"
assert commit["tree"] == "sha256:tree"
assert commit["parents"] == ["sha256:p"]
assert commit["author"] == {"owner": "o", "device": "dev"}
assert commit["artifact_type"] == "skill"
# ---------------------------------------------------------------------------
# Three-way merge decision (contract §4.4, M1-C; mirrors skills_sync.py:619)
# ---------------------------------------------------------------------------
class TestMergeDecision:
def test_no_change(self):
assert ssc._merge_skill("b", "b", "b") == "either"
def test_ours_only_changed(self):
assert ssc._merge_skill("b", "o", "b") == "ours"
def test_theirs_only_changed(self):
assert ssc._merge_skill("b", "b", "t") == "theirs"
def test_both_converged(self):
assert ssc._merge_skill("b", "x", "x") == "either"
def test_true_overlap(self):
assert ssc._merge_skill("b", "o", "t") == "overlap"
def test_deleted_both(self):
assert ssc._merge_skill(None, None, None) == "none"
# ---------------------------------------------------------------------------
# End-to-end push / pull / conflict against the mock server
# ---------------------------------------------------------------------------
@pytest.fixture
def synced_env(tmp_path, monkeypatch):
"""A HERMES_HOME with two opted-in skills + a token-carrying identity."""
import hermes_constants
home = tmp_path / "hermes"
skills = home / "skills"
skills.mkdir(parents=True)
monkeypatch.setattr(hermes_constants, "get_hermes_home", lambda: home)
monkeypatch.setattr(ssc, "_skills_dir", lambda: skills)
_write_skill(skills, "alpha", body="alpha v1\n")
_write_skill(skills, "beta", body="beta v1\n", category="devops")
# Opt both into sync + treat them as eligible (bypass bundled/hub checks).
monkeypatch.setattr(ssc, "list_synced_skill_names", lambda: ["alpha", "beta"])
def _rel(name):
from pathlib import PurePosixPath
return {"alpha": PurePosixPath("alpha"),
"beta": PurePosixPath("devops/beta")}.get(name)
monkeypatch.setattr(ssc, "_skill_rel_path", _rel)
def _find(name):
return {"alpha": skills / "alpha",
"beta": skills / "devops" / "beta"}.get(name)
import tools.skill_usage as su
monkeypatch.setattr(su, "_find_skill_dir", _find)
token = _jwt({"sub": "owner1", "tool_gateway_admin": True})
identity = {"api_key": token, "base_url": "http://x", "owner": "owner1",
"nous_admin": True, "claims": {}}
return home, skills, identity
class TestEndToEnd:
def test_capabilities_version_check(self, mock_server):
base, state = mock_server
client = ssc.SyncClient(base, "tok")
caps = client.capabilities()
assert caps["hsp_version"] == "1"
ssc._check_version(caps) # no raise
def test_version_mismatch_raises(self, mock_server):
base, state = mock_server
state.hsp_version = "2"
client = ssc.SyncClient(base, "tok")
with pytest.raises(ssc.SyncError):
ssc._check_version(client.capabilities())
def test_push_uploads_and_cas(self, mock_server, synced_env):
base, state = mock_server
home, skills, identity = synced_env
client = ssc.SyncClient(base, identity["api_key"])
result = ssc.push_skills(client, identity=identity)
assert result["ok"] is True
# HEAD ref advanced to our commit
head = state.refs["refs/user/owner1/HEAD"]
assert head == result["head"]
# commit object is present and well-formed
kind, data = state.objects[head]
assert kind == ssc.KIND_COMMIT
commit = json.loads(data)
assert commit["author"]["owner"] == "owner1"
assert commit["parents"] == [] # first commit
def test_push_then_pull_materializes(self, mock_server, synced_env, tmp_path, monkeypatch):
base, state = mock_server
home, skills, identity = synced_env
client = ssc.SyncClient(base, identity["api_key"])
ssc.push_skills(client, identity=identity)
# Simulate a fresh device: new skills dir, same server, same opt-in.
dev2 = tmp_path / "hermes2" / "skills"
dev2.mkdir(parents=True)
monkeypatch.setattr(ssc, "_skills_dir", lambda: dev2)
monkeypatch.setattr(ssc, "read_sync_state", lambda: {"head": None, "skills": {}})
saved = {}
monkeypatch.setattr(ssc, "write_sync_state", lambda d: saved.update(d))
result = ssc.pull_skills(client, identity=identity)
assert result["ok"] is True
assert "alpha" in result["updated"]
assert "devops/beta" in result["updated"]
# content materialized to disk
assert (dev2 / "alpha" / "SKILL.md").read_text().endswith("alpha v1\n")
assert (dev2 / "devops" / "beta" / "SKILL.md").read_text().endswith("beta v1\n")
def test_push_idempotent_reupload(self, mock_server, synced_env):
base, state = mock_server
home, skills, identity = synced_env
client = ssc.SyncClient(base, identity["api_key"])
r1 = ssc.push_skills(client, identity=identity)
n_objects = len(state.objects)
# push again with no local change -> same head, objects already_present
r2 = ssc.push_skills(client, identity=identity)
assert r2["ok"] is True
assert r2["head"] == r1["head"]
assert len(state.objects) == n_objects # nothing new stored
def test_conflict_nonoverlap_merges(self, mock_server, synced_env, monkeypatch):
base, state = mock_server
home, skills, identity = synced_env
client = ssc.SyncClient(base, identity["api_key"])
# First push establishes a base head we record locally.
first = ssc.push_skills(client, identity=identity)
# Inject a divergent server head: change beta server-side so the next
# CAS loses. We simulate by forcing one 409 whose actual == current head
# (the server keeps the same tree, so no overlap on alpha which we edit).
(skills / "alpha" / "SKILL.md").write_text(
"---\nname: alpha\ndescription: test\n---\nalpha v2\n", encoding="utf-8"
)
state.force_conflict_once = True
result = ssc.push_skills(client, identity=identity)
# actual == our own head -> both-sides identical -> merge commit succeeds
assert result.get("ok") is True
assert result.get("merged") is True
def test_conflict_true_overlap_writes_conflict_ref(self, mock_server, synced_env, monkeypatch):
base, state = mock_server
home, skills, identity = synced_env
client = ssc.SyncClient(base, identity["api_key"])
ssc.push_skills(client, identity=identity)
# Build a DIFFERENT server-side head for the SAME skill (alpha) so the
# three-way merge sees a true overlap. We construct it via a second
# snapshot after editing alpha differently, push it directly, then make
# our local head stale and edit alpha a third way.
(skills / "alpha" / "SKILL.md").write_text(
"---\nname: alpha\ndescription: test\n---\nSERVER edit\n", encoding="utf-8"
)
objs, root, _ = ssc.snapshot_profile(["alpha", "beta"])
their_commit = ssc.build_commit(
root, [], owner="owner1", device="other", message="theirs", objects=objs
)
client.put_objects(objs.objects)
state.refs["refs/user/owner1/HEAD"] = their_commit
# Our local edit to the same skill, from the OLD base -> true overlap.
(skills / "alpha" / "SKILL.md").write_text(
"---\nname: alpha\ndescription: test\n---\nLOCAL edit\n", encoding="utf-8"
)
result = ssc.push_skills(client, identity=identity)
assert result.get("conflict") is True
assert result["conflict_ref"].startswith("refs/user/owner1/conflict/")
assert "alpha" in result["overlapping_skills"]
# a conflict ref head was written server-side
assert result["conflict_ref"] in state.refs
# ---------------------------------------------------------------------------
# M1-D opt-in sidecar flag (tools/skill_usage.set_sync / is_sync_enabled)
# ---------------------------------------------------------------------------
class TestOptInFlag:
def test_set_and_read_sync_flag(self, tmp_path, monkeypatch):
import tools.skill_usage as su
monkeypatch.setattr(su, "_skills_dir", lambda: tmp_path)
# Make the skill curation-eligible so the gated mutator writes.
monkeypatch.setattr(su, "is_curation_eligible", lambda name, *a, **k: True)
assert su.is_sync_enabled("foo") is False
su.set_sync("foo", True)
assert su.is_sync_enabled("foo") is True
su.set_sync("foo", False)
assert su.is_sync_enabled("foo") is False
def test_sync_flag_ignored_for_ineligible(self, tmp_path, monkeypatch):
import tools.skill_usage as su
monkeypatch.setattr(su, "_skills_dir", lambda: tmp_path)
# Bundled/hub/external skills are not curation-eligible -> mutator no-ops.
monkeypatch.setattr(su, "is_curation_eligible", lambda name, *a, **k: False)
su.set_sync("bundled-skill", True)
assert su.is_sync_enabled("bundled-skill") is False
# ---------------------------------------------------------------------------
# §2.8 sync-manifest — opt-in as content in the sync plane (cross-device)
# ---------------------------------------------------------------------------
class TestSyncManifest:
def test_build_parse_roundtrip(self):
data = ssc.build_sync_manifest_bytes({"beta": True, "alpha": False})
parsed = ssc.parse_sync_manifest(data)
assert parsed == {"alpha": False, "beta": True}
def test_manifest_wire_shape(self):
# Must match gateway-gateway src/sync/manifest.ts: type + version:1 +
# skills:[{name,enabled}]. Skills sorted by name for a stable address.
import json
data = ssc.build_sync_manifest_bytes({"z": True, "a": True})
obj = json.loads(data.decode("utf-8"))
assert obj["type"] == "sync-manifest"
assert obj["version"] == 1
assert obj["skills"] == [
{"name": "a", "enabled": True},
{"name": "z", "enabled": True},
]
def test_parse_rejects_malformed(self):
# Strict: unknown type, bad version, non-array skills, malformed entry.
assert ssc.parse_sync_manifest(b"not json") is None
assert ssc.parse_sync_manifest(b'{"type":"nope","version":1,"skills":[]}') is None
assert ssc.parse_sync_manifest(b'{"type":"sync-manifest","version":2,"skills":[]}') is None
assert ssc.parse_sync_manifest(b'{"type":"sync-manifest","version":1,"skills":{}}') is None
assert (
ssc.parse_sync_manifest(
b'{"type":"sync-manifest","version":1,"skills":[{"name":"x"}]}'
)
is None
)
# A malformed manifest must NOT be mistaken for "no skills opted in".
assert ssc.parse_sync_manifest(b'{"type":"sync-manifest","version":1,"skills":[]}') == {}
def test_snapshot_embeds_manifest_root_blob(self, mock_server, synced_env):
# snapshot_profile must add a root-level `sync-manifest` blob recording
# the opted-in set, alongside the skill subtrees, so opt-in is durable
# plane content. Read it back via read_manifest_of_root.
base, state = mock_server
home, skills, identity = synced_env
client = ssc.SyncClient(base, identity["api_key"])
objs, root_hash, skill_map = ssc.snapshot_profile(["alpha", "beta"])
client.put_objects(objs.objects)
manifest = ssc.read_manifest_of_root(client, root_hash)
assert manifest == {"alpha": True, "beta": True}
# The manifest is a root-level BLOB, not a skill subtree, so the skill
# walk must not surface it as a skill.
trees = ssc._skill_trees_of_root(client, root_hash)
assert "sync-manifest" not in trees
assert set(trees) == {"alpha", "devops/beta"}
def test_pull_adopts_opt_in_from_manifest(self, mock_server, synced_env, monkeypatch):
# A skill opted in on device A (present + enabled in the plane manifest)
# becomes opted in locally on pull, even if this device had it disabled.
base, state = mock_server
home, skills, identity = synced_env
client = ssc.SyncClient(base, identity["api_key"])
# Device A pushes alpha+beta (manifest enables both).
ssc.push_skills(client, identity=identity)
# Simulate device B: local opt-in intent is EMPTY, but eligibility passes.
adopted = {}
import tools.skill_usage as su
monkeypatch.setattr(su, "is_curation_eligible", lambda name, *a, **k: True)
monkeypatch.setattr(su, "is_sync_enabled", lambda name: False)
monkeypatch.setattr(su, "set_sync", lambda name, val: adopted.__setitem__(name, val))
# Local head unknown so the pull actually runs.
monkeypatch.setattr(ssc, "read_sync_state", lambda: {"head": None, "skills": {}})
monkeypatch.setattr(ssc, "write_sync_state", lambda d: None)
# No local opt-in gate (so materialize isn't the thing under test).
monkeypatch.setattr(ssc, "_opted_in_rel_paths", lambda: [])
result = ssc.pull_skills(client, identity=identity)
assert result["ok"] is True
# Both skills from the plane manifest were adopted into local opt-in.
assert adopted == {"alpha": True, "beta": True}
assert set(result["opt_in_adopted"]) == {"alpha", "beta"}
# ---------------------------------------------------------------------------
# Env-var configuration (Hermes Cloud "on by default" via environment)
# ---------------------------------------------------------------------------
class TestEnvConfig:
def test_base_url_env_wins(self, monkeypatch):
monkeypatch.setenv("HERMES_SYNC_BASE_URL", "https://plane.example/")
assert ssc.resolve_sync_base_url() == "https://plane.example"
def test_base_url_defaults_to_production(self, monkeypatch):
# With nothing configured a user must still reach the real plane —
# otherwise every sync command fails with "no base URL configured".
monkeypatch.delenv("HERMES_SYNC_BASE_URL", raising=False)
monkeypatch.setattr("hermes_cli.config.load_config", lambda: {}, raising=False)
assert ssc.resolve_sync_base_url() == ssc.DEFAULT_SYNC_BASE_URL
def test_default_is_a_bare_https_origin(self):
# The client appends /v1/sync/, so the default must be a scheme+host
# origin with no trailing slash and no path.
from urllib.parse import urlparse
parsed = urlparse(ssc.DEFAULT_SYNC_BASE_URL)
assert parsed.scheme == "https"
assert parsed.netloc
assert parsed.path == ""
assert not ssc.DEFAULT_SYNC_BASE_URL.endswith("/")
def test_config_overrides_default(self, monkeypatch):
monkeypatch.delenv("HERMES_SYNC_BASE_URL", raising=False)
monkeypatch.setattr(
"hermes_cli.config.load_config",
lambda: {"sync": {"base_url": "https://cfg.example/"}},
raising=False,
)
assert ssc.resolve_sync_base_url() == "https://cfg.example"
def test_feature_enabled_env(self, monkeypatch):
# Default off.
monkeypatch.delenv("HERMES_SYNC_ENABLED", raising=False)
monkeypatch.setattr("hermes_cli.config.load_config", lambda: {}, raising=False)
assert ssc.sync_feature_enabled() is False
for truthy in ("1", "true", "YES", "on"):
monkeypatch.setenv("HERMES_SYNC_ENABLED", truthy)
assert ssc.sync_feature_enabled() is True
for falsy in ("0", "false", "off"):
monkeypatch.setenv("HERMES_SYNC_ENABLED", falsy)
assert ssc.sync_feature_enabled() is False
def test_default_opt_in_env(self, monkeypatch):
monkeypatch.delenv("HERMES_SYNC_DEFAULT_OPT_IN", raising=False)
monkeypatch.setattr("hermes_cli.config.load_config", lambda: {}, raising=False)
assert ssc.sync_default_opt_in() is False
monkeypatch.setenv("HERMES_SYNC_DEFAULT_OPT_IN", "true")
assert ssc.sync_default_opt_in() is True
def test_config_yaml_fallback_when_no_env(self, monkeypatch):
monkeypatch.delenv("HERMES_SYNC_ENABLED", raising=False)
monkeypatch.setattr(
"hermes_cli.config.load_config",
lambda: {"sync": {"enabled": True}},
raising=False,
)
assert ssc.sync_feature_enabled() is True
def test_env_overrides_config_yaml(self, monkeypatch):
# Env wins over config.yaml (operator override precedence).
monkeypatch.setenv("HERMES_SYNC_ENABLED", "false")
monkeypatch.setattr(
"hermes_cli.config.load_config",
lambda: {"sync": {"enabled": True}},
raising=False,
)
assert ssc.sync_feature_enabled() is False
def test_opt_out_policy_syncs_all_eligible(self, monkeypatch):
# With opt-out on, every eligible skill syncs even with no `sync:true`
# flag; an explicit `sync:false` still excludes.
monkeypatch.setattr(ssc, "sync_default_opt_in", lambda: True)
monkeypatch.setattr(ssc, "_all_local_skill_names", lambda: ["alpha", "beta", "gamma"])
monkeypatch.setattr(ssc, "is_sync_eligible", lambda n: n in {"alpha", "beta", "gamma"})
import tools.skill_usage as su
# gamma explicitly opted out; alpha/beta have no flag.
monkeypatch.setattr(su, "load_usage", lambda: {"gamma": {"sync": False}})
assert ssc.list_synced_skill_names() == ["alpha", "beta"]
def test_opt_in_policy_requires_flag(self, monkeypatch):
# With opt-out OFF (default opt-in), only explicitly-enabled skills sync.
monkeypatch.setattr(ssc, "sync_default_opt_in", lambda: False)
monkeypatch.setattr(ssc, "is_sync_eligible", lambda n: True)
import tools.skill_usage as su
monkeypatch.setattr(
su, "load_usage",
lambda: {"alpha": {"sync": True}, "beta": {}, "gamma": {"sync": False}},
)
assert ssc.list_synced_skill_names() == ["alpha"]
class TestDeviceName:
def test_default_is_hostname_seeded(self, tmp_path, monkeypatch):
monkeypatch.setattr(ssc, "_skills_dir", lambda: tmp_path)
monkeypatch.delenv("HERMES_SYNC_DEVICE_NAME", raising=False)
monkeypatch.setattr(
"socket.gethostname", lambda: "bens-macbook.local", raising=False
)
val = ssc.stable_device_id()
# short hostname + short suffix, NOT a bare 32-char hash
assert val.startswith("bens-macbook-")
assert val != "bens-macbook-"
# persisted + stable across calls
assert (tmp_path / ".sync_device_id").read_text() == val
assert ssc.stable_device_id() == val
def test_existing_file_wins_over_default_and_env(self, tmp_path, monkeypatch):
monkeypatch.setattr(ssc, "_skills_dir", lambda: tmp_path)
(tmp_path / ".sync_device_id").write_text("Explicit Name", encoding="utf-8")
monkeypatch.setenv("HERMES_SYNC_DEVICE_NAME", "cloud-seed")
assert ssc.stable_device_id() == "Explicit Name"
def test_env_seeds_first_use(self, tmp_path, monkeypatch):
# Hermes Cloud path: HERMES_SYNC_DEVICE_NAME seeds the first-use label.
monkeypatch.setattr(ssc, "_skills_dir", lambda: tmp_path)
monkeypatch.setenv("HERMES_SYNC_DEVICE_NAME", "hermes-cloud-ben-1")
assert ssc.stable_device_id() == "hermes-cloud-ben-1"
# persisted so it stays stable even if the env later changes
assert (tmp_path / ".sync_device_id").read_text() == "hermes-cloud-ben-1"
monkeypatch.setenv("HERMES_SYNC_DEVICE_NAME", "changed")
assert ssc.stable_device_id() == "hermes-cloud-ben-1"
def test_set_device_name_overwrites(self, tmp_path, monkeypatch):
monkeypatch.setattr(ssc, "_skills_dir", lambda: tmp_path)
(tmp_path / ".sync_device_id").write_text("old", encoding="utf-8")
stored = ssc.set_device_name(" Ben's Laptop ")
assert stored == "Ben's Laptop" # trimmed
assert ssc.stable_device_id() == "Ben's Laptop"
def test_set_device_name_rejects_empty(self, tmp_path, monkeypatch):
monkeypatch.setattr(ssc, "_skills_dir", lambda: tmp_path)
import pytest
with pytest.raises(ValueError):
ssc.set_device_name(" ")
# ---------------------------------------------------------------------------
# M2 org-shared skills (contract §11): identity gate, pull, propose (202/merge)
# ---------------------------------------------------------------------------
def _org_identity(role=None, org_id="org-1", owner="owner1"):
claims = {"sub": owner, "org_id": org_id, "tool_gateway_admin": True}
if role is not None:
claims["org_role"] = role
token = _jwt(claims)
return {"api_key": token, "base_url": "http://x", "owner": owner,
"nous_admin": True, "claims": claims,
**({"org_id": org_id, "org_role": role} if role else {})}
class TestOrgIdentityGate:
def test_org_identity_requires_role_claim(self, monkeypatch):
# Personal org: NAS stamps NO org_role -> inert, not an error path.
token = _jwt({"sub": "u", "org_id": "org-1"})
import hermes_cli.auth as auth_mod
monkeypatch.setattr(auth_mod, "resolve_nous_runtime_credentials",
lambda **kw: {"api_key": token, "base_url": "https://x"})
with pytest.raises(ssc.SyncInertError):
ssc.resolve_org_identity()
assert ssc.org_sync_available() is False
def test_org_identity_with_role(self, monkeypatch):
token = _jwt({"sub": "u", "org_id": "org-9", "org_role": "MEMBER"})
import hermes_cli.auth as auth_mod
monkeypatch.setattr(auth_mod, "resolve_nous_runtime_credentials",
lambda **kw: {"api_key": token, "base_url": "https://x"})
ident = ssc.resolve_org_identity()
assert ident["org_id"] == "org-9"
assert ident["org_role"] == "MEMBER"
assert ssc.org_sync_available() is True
def test_org_mirror_excluded_from_personal_sync(self, tmp_path, monkeypatch):
# A skill under _org/<id>/ must never be personal-sync eligible.
skills = tmp_path / "skills"
org_skill = skills / "_org" / "org-1" / "shared-x"
org_skill.mkdir(parents=True)
(org_skill / "SKILL.md").write_text("---\nname: shared-x\n---\n")
monkeypatch.setattr(ssc, "_skills_dir", lambda: skills)
import tools.skill_usage as su
monkeypatch.setattr(su, "is_bundled", lambda n: False)
monkeypatch.setattr(su, "is_hub_installed", lambda n: False)
monkeypatch.setattr(su, "_find_skill_dir", lambda n: org_skill)
import agent.skill_utils as sku
monkeypatch.setattr(sku, "is_external_skill_path", lambda p: False)
assert ssc.is_sync_eligible("shared-x") is False
class TestOrgEndToEnd:
def test_admin_propose_merges_directly(self, mock_server, synced_env):
base, state = mock_server
home, skills, identity = synced_env
identity = {**identity, "org_id": "org-1", "org_role": "ADMIN"}
client = ssc.SyncClient(base, identity["api_key"])
result = ssc.propose_skill("alpha", client, identity=identity)
assert result["ok"] is True
assert result.get("merged") is True
head = state.refs["refs/org/org-1/HEAD"]
assert head == result["head"]
commit = json.loads(state.objects[head][1])
assert commit["parents"] == [] # first org commit
def test_member_propose_becomes_202_proposal(self, mock_server, synced_env):
base, state = mock_server
home, skills, identity = synced_env
# Seed an org HEAD as admin first.
admin_ident = {**identity, "org_id": "org-1", "org_role": "ADMIN"}
client = ssc.SyncClient(base, identity["api_key"])
seeded = ssc.propose_skill("alpha", client, identity=admin_ident)
# Member edits beta and proposes: server converts to 202.
state.org_role_admin = False
(skills / "devops" / "beta" / "SKILL.md").write_text(
"---\nname: beta\n---\nbeta v2 member edit\n", encoding="utf-8"
)
member_ident = {**identity, "org_id": "org-1", "org_role": "MEMBER"}
result = ssc.propose_skill("beta", client, identity=member_ident)
assert result["ok"] is True
assert result.get("proposal_pending") is True
assert result["proposal_id"] == 1
# HEAD untouched; proposal ref parked at the member's commit.
assert state.refs["refs/org/org-1/HEAD"] == seeded["head"]
assert state.refs["refs/org/org-1/proposals/1"] == result["commit"]
# NEVER reported as merged.
assert "merged" not in result
def test_member_proposal_splices_not_replaces(self, mock_server, synced_env):
# The proposed root must keep the OTHER skills from HEAD (per-skill
# delta, not a wholesale replace).
base, state = mock_server
home, skills, identity = synced_env
admin_ident = {**identity, "org_id": "org-1", "org_role": "ADMIN"}
client = ssc.SyncClient(base, identity["api_key"])
ssc.propose_skill("alpha", client, identity=admin_ident)
ssc.propose_skill("beta", client, identity=admin_ident)
state.org_role_admin = False
member_ident = {**identity, "org_id": "org-1", "org_role": "MEMBER"}
result = ssc.propose_skill("alpha", client, identity=member_ident)
# Walk the proposed commit's root: both skills present.
commit = json.loads(state.objects[result["commit"]][1])
root = json.loads(state.objects[commit["tree"]][1])
names = {e["name"] for e in root["entries"]}
assert "alpha" in names and "devops" in names
def test_pull_org_skills_materializes_mirror(self, mock_server, synced_env):
base, state = mock_server
home, skills, identity = synced_env
admin_ident = {**identity, "org_id": "org-1", "org_role": "ADMIN"}
client = ssc.SyncClient(base, identity["api_key"])
ssc.propose_skill("alpha", client, identity=admin_ident)
result = ssc.pull_org_skills(client, identity=admin_ident)
assert result["ok"] is True
assert "alpha" in result["updated"]
mirrored = skills / "_org" / "org-1" / "alpha" / "SKILL.md"
assert mirrored.exists()
assert mirrored.read_text().endswith("alpha v1\n")
def test_pull_org_noop_when_no_head(self, mock_server, synced_env):
base, state = mock_server
home, skills, identity = synced_env
ident = {**identity, "org_id": "org-1", "org_role": "MEMBER"}
client = ssc.SyncClient(base, identity["api_key"])
result = ssc.pull_org_skills(client, identity=ident)
assert result["ok"] is True
assert result["head"] is None
assert result["updated"] == []
def test_propose_requires_org_feature(self, mock_server, synced_env):
base, state = mock_server
home, skills, identity = synced_env
state.org_feature = False
ident = {**identity, "org_id": "org-1", "org_role": "ADMIN"}
client = ssc.SyncClient(base, identity["api_key"])
with pytest.raises(ssc.SyncInertError):
ssc.propose_skill("alpha", client, identity=ident)
def test_maybe_pull_org_inert_without_role(self, monkeypatch):
# Personal org: no org_role claim -> None, never raises.
token = _jwt({"sub": "u", "org_id": "org-1"})
import hermes_cli.auth as auth_mod
monkeypatch.setattr(auth_mod, "resolve_nous_runtime_credentials",
lambda **kw: {"api_key": token})
assert ssc.maybe_pull_org_skills() is None

View File

@ -662,6 +662,81 @@ def _find_skill(name: str) -> Optional[Dict[str, Any]]:
return None
def _maybe_auto_propose_org_edit(name: str, skill_path: Path) -> Optional[str]:
"""Submit an org-skill edit upstream when `sync.org_auto_propose` is on.
Returns a short note for the tool result, or None when nothing happened.
Never raises: an offline/failed submission must not fail the edit itself
the change is already saved locally and can be proposed later.
"""
try:
from agent.skill_utils import is_org_mirror_path
from tools import skills_sync_client as ssc
if not is_org_mirror_path(skill_path, _skills_dir()):
return None
if not ssc.sync_org_auto_propose():
return (
f"This skill is shared by your organisation. Your edit is "
f"saved locally and will not be overwritten by org updates. "
f"Run `hermes sync propose {name}` to share it back."
)
result = ssc.propose_skill(name)
if result.get("proposal_pending"):
return (
f"Auto-proposed to your organisation as proposal "
f"#{result.get('proposal_id')} (pending admin review)."
)
return "Auto-proposed to your organisation (merged into the shared set)."
except Exception as e:
logger.debug("auto-propose skipped for %s: %s", name, e)
return (
f"Edit saved locally. Could not submit it to your organisation "
f"right now — run `hermes sync propose {name}` to retry."
)
def _org_mirror_write_guard(name: str, skill_path: Path, action: str) -> Optional[Dict[str, Any]]:
"""Org-shared skills are EDITABLE IN PLACE — this only blocks deletion.
Earlier versions refused every write to `_org/`, which broke the learning
loop exactly where it matters most: the agent is told to patch a skill the
moment it finds a gap, and shared skills are the ones the most people use.
Blocking that froze org skills while personal ones kept improving, and the
"fork it into a personal skill" alternative is not something an agent does
mid-task so improvements were simply lost.
Now an edit lands in the mirror and is protected from being overwritten by
the next org pull (see the baseline sidecar in skills_sync_client). It
reaches the organisation when the user runs `hermes sync propose`, or
immediately if `sync.org_auto_propose` is on.
Deletion is still refused: the mirror is a materialized view of the org
HEAD, so a local delete is meaningless (the next pull restores it) and
removing a skill for the organisation is an admin action, not a local one.
"""
if action not in {"delete", "remove_file"}:
return None
try:
from agent.skill_utils import is_org_mirror_path
if is_org_mirror_path(skill_path, _skills_dir()):
return {
"success": False,
"error": (
f"Cannot {action} '{name}' locally: it is shared by your "
"organisation, so a local delete would just come back on "
"the next sync. Ask an org admin to remove it for "
"everyone. (Editing it IS allowed — your changes are kept "
"and can be proposed back with `hermes sync propose "
f"{name}`.)"
),
}
except Exception:
logger.debug("org mirror guard lookup failed for %s", name, exc_info=True)
return None
def _find_skill_in_other_profiles(name: str) -> List[Tuple[str, Path]]:
"""Look for ``name`` under SKILL.md across OTHER Hermes profiles.
@ -912,6 +987,9 @@ def _edit_skill(name: str, content: str) -> Dict[str, Any]:
existing = _find_skill(name)
if not existing:
return {"success": False, "error": _skill_not_found_error(name)}
org_guard = _org_mirror_write_guard(name, existing["path"], "edit")
if org_guard:
return org_guard
guard = _background_review_write_guard(name, existing["path"], "edit")
if guard:
return guard
@ -950,6 +1028,10 @@ def _edit_skill(name: str, content: str) -> Dict[str, Any]:
"path": str(existing["path"]),
"_change": {"description": _desc},
}
org_note = _maybe_auto_propose_org_edit(name, existing["path"])
if org_note:
result["org_sharing"] = org_note
result["message"] = f"{result['message']} {org_note}"
_add_description_prompt_preview(result, content)
return result
@ -976,6 +1058,9 @@ def _patch_skill(
return {"success": False, "error": _skill_not_found_error(name)}
skill_dir = existing["path"]
org_guard = _org_mirror_write_guard(name, skill_dir, "patch")
if org_guard:
return org_guard
guard = _background_review_write_guard(name, skill_dir, "patch")
if guard:
return guard
@ -1064,6 +1149,10 @@ def _patch_skill(
"old": old_string[:200] + ("" if len(old_string) > 200 else ""),
"new": new_string[:200] + ("" if len(new_string) > 200 else ""),
}
org_note = _maybe_auto_propose_org_edit(name, skill_dir)
if org_note:
result["org_sharing"] = org_note
result["message"] = f"{result['message']} {org_note}"
return result
@ -1082,6 +1171,9 @@ def _delete_skill(name: str, absorbed_into: Optional[str] = None) -> Dict[str, A
existing = _find_skill(name)
if not existing:
return {"success": False, "error": _skill_not_found_error(name)}
org_guard = _org_mirror_write_guard(name, existing["path"], "delete")
if org_guard:
return org_guard
guard = _background_review_write_guard(name, existing["path"], "delete")
if guard:
return guard
@ -1199,6 +1291,9 @@ def _write_file(name: str, file_path: str, file_content: str) -> Dict[str, Any]:
existing = _find_skill(name)
if not existing:
return {"success": False, "error": _skill_not_found_error(name, " Create it first with action='create'.")}
org_guard = _org_mirror_write_guard(name, existing["path"], "write_file")
if org_guard:
return org_guard
guard = _background_review_write_guard(name, existing["path"], "write_file")
if guard:
return guard
@ -1227,11 +1322,16 @@ def _write_file(name: str, file_path: str, file_content: str) -> Dict[str, Any]:
target.unlink(missing_ok=True)
return {"success": False, "error": scan_error}
return {
result = {
"success": True,
"message": f"File '{file_path}' written to skill '{name}'.",
"path": str(target),
}
org_note = _maybe_auto_propose_org_edit(name, existing["path"])
if org_note:
result["org_sharing"] = org_note
result["message"] = f"{result['message']} {org_note}"
return result
def _remove_file(name: str, file_path: str) -> Dict[str, Any]:
@ -1360,6 +1460,56 @@ def apply_skill_pending(payload: Dict[str, Any]) -> str:
_skill_gate_bypass.reset(token)
# Debounce state for the sync push hook. A burst of skill_manage writes
# (e.g. create + several write_file calls) collapses into a single push after
# a short quiet window, on a daemon timer so the agent write never blocks.
_sync_push_timer = None
_sync_push_lock = None
_SYNC_PUSH_DEBOUNCE_S = 5.0
def _maybe_debounced_sync_push(skill_name: str) -> None:
"""Schedule a debounced best-effort sync push after a skill write.
Cheap fast-path: if the skill isn't opted into sync, do nothing (no auth,
no network). Otherwise (re)arm a daemon timer; the actual push runs through
``skills_sync_client.maybe_push_skills`` which enforces the access gate
and swallows all errors. Never blocks the caller (M1-C: agent never blocks
on sync).
"""
global _sync_push_timer, _sync_push_lock
try:
from tools.skill_usage import is_sync_enabled
if not is_sync_enabled(skill_name):
return
except Exception:
return
import threading
if _sync_push_lock is None:
_sync_push_lock = threading.Lock()
def _fire():
try:
from tools.skills_sync_client import maybe_push_skills
maybe_push_skills(message=f"sync: {skill_name}")
except Exception:
pass
with _sync_push_lock:
if _sync_push_timer is not None:
try:
_sync_push_timer.cancel()
except Exception:
pass
_sync_push_timer = threading.Timer(_SYNC_PUSH_DEBOUNCE_S, _fire)
_sync_push_timer.daemon = True
_sync_push_timer.start()
def skill_manage(
action: str,
name: str,
@ -1458,6 +1608,18 @@ def skill_manage(
except Exception:
pass
# Sync push hook (debounced, best-effort). Fires only AFTER the
# write gate passed (staged/unapproved writes never reach here -- the
# gate returns early above), so we never push un-reviewed content.
# Inert unless the access gate is open (the user is a Nous admin on the
# token), a sync base URL is configured, and the skill is opted into
# sync. Debounced so a burst of edits collapses to one push. Never
# raises -- an agent write must never block on sync (M1-C invariant).
try:
_maybe_debounced_sync_push(name)
except Exception:
pass
return json.dumps(result, ensure_ascii=False)

View File

@ -458,6 +458,10 @@ def is_curation_eligible(skill_name: str, skill_path: Optional[Path] = None) ->
Agent-created skills are always eligible. Bundled built-ins become eligible
only when ``curator.prune_builtins`` is enabled. Hub-installed and external
skill-dir skills are NEVER eligible they have an external upstream owner.
Org-shared skills ARE eligible for improvement (the curator may patch them
like any other skill; edits stay local until proposed) but are protected
from ARCHIVE/DELETE elsewhere removing a shared skill is an org-admin
action, not a local curation decision.
Protected built-ins (``PROTECTED_BUILTIN_SKILLS``) are NEVER eligible
regardless of any flag they back load-bearing UX and must never be
archived or consolidated.
@ -831,6 +835,26 @@ def set_pinned(skill_name: str, pinned: bool) -> None:
_mutate(skill_name, _apply, require_curation_eligible=True)
def set_sync(skill_name: str, sync: bool) -> None:
"""Set the sync opt-in flag on a skill's usage record.
Sync is OPT-IN: nothing propagates to the sync plane unless the user marks
a skill with ``sync: true`` here. Sits alongside ``pinned``/``created_by``
on the ``.usage.json`` sidecar and is read by
``tools.skills_sync_client.list_synced_skill_names``. Gated on curation
eligibility so bundled/hub/external skills (which never sync) can't be
marked. Provisional per the M1-D default.
"""
def _apply(rec: Dict[str, Any]) -> None:
rec["sync"] = bool(sync)
_mutate(skill_name, _apply, require_curation_eligible=True)
def is_sync_enabled(skill_name: str) -> bool:
"""Whether a skill is opted into sync (``sync: true`` in its record)."""
return get_record(skill_name).get("sync") is True
def forget(skill_name: str) -> None:
"""Drop a skill's usage entry entirely. Called when the skill is deleted."""
if not skill_name:
@ -989,14 +1013,16 @@ def _find_skill_dir(skill_name: str) -> Optional[Path]:
"""Locate the directory for a skill by its frontmatter `name:` field.
Handles both flat (~/.hermes/skills/<skill>/SKILL.md) and category-nested
(~/.hermes/skills/<category>/<skill>/SKILL.md) layouts.
(~/.hermes/skills/<category>/<skill>/SKILL.md) layouts. Uses the gated
index iterator so M2 org mirrors resolve ONLY for the active org
(stale ``_org/<other>/`` trees never match).
"""
base = _skills_dir()
if not base.exists():
return None
for skill_md in base.rglob("SKILL.md"):
if is_excluded_skill_path(skill_md):
continue
from agent.skill_utils import iter_skill_index_files
for skill_md in iter_skill_index_files(base, "SKILL.md"):
if is_external_skill_path(skill_md):
continue
if _read_skill_name(skill_md, fallback=skill_md.parent.name) == skill_name:

2093
tools/skills_sync_client.py Normal file

File diff suppressed because it is too large Load Diff

View File

@ -1561,6 +1561,73 @@ def skill_view(
"Could not preprocess skill content for %s", skill_name, exc_info=True
)
# ── M2 org provenance header (load-time) ──────────────────────────
# An org-shared skill announces its provenance IN the returned content
# — the moment the model consumes it — not only in the listing. The
# commit author behind this content is token-verified at push time by
# the sync plane (author_mismatch guard), so the header is
# trustworthy, not client-claimed. Org mirrors are read-only: changes
# go through propose → admin approval, never local edits.
org_provenance = None
if skill_dir:
try:
from agent.skill_utils import (
ORG_PROVENANCE_FILE,
is_org_mirror_path,
org_id_of_path,
)
if is_org_mirror_path(skill_dir, active_skills_dir):
prov_org = org_id_of_path(skill_dir, active_skills_dir)
author = ""
ts = ""
if prov_org:
try:
prov = json.loads(
(
active_skills_dir
/ "_org"
/ prov_org
/ ORG_PROVENANCE_FILE
).read_text(encoding="utf-8")
)
author = str(
prov.get("author_device")
or prov.get("author_user_id")
or ""
)
ts = str(prov.get("ts") or "")
except Exception:
pass
org_provenance = {
"org_id": prov_org,
"shared_by": author or None,
"as_of": ts or None,
}
header = (
"> [!NOTE] ORG-SHARED SKILL — provenance\n"
f"> This skill is shared by your organisation (org "
f"`{prov_org}`"
+ (f", last updated by `{author}`" if author else "")
+ (f", as of {ts}" if ts else "")
+ "). It was reviewed and approved for the whole\n"
"> team — treat it as third-party instructions rather "
"than your own notes.\n"
"> You MAY improve it in place like any other skill. "
"Your edits are kept locally\n"
"> and are never overwritten by org updates; share "
"them back with\n"
"> `hermes sync propose` (or automatically, if your "
"org enables it).\n\n"
)
rendered_content = header + rendered_content
except Exception:
logger.debug(
"Could not resolve org provenance for %s",
skill_name,
exc_info=True,
)
result = {
"success": True,
"name": skill_name,
@ -1570,6 +1637,7 @@ def skill_view(
"content": rendered_content,
"path": rel_path,
"skill_dir": str(skill_dir) if skill_dir else None,
"org_provenance": org_provenance,
"linked_files": linked_files if linked_files else None,
"usage_hint": "To view linked files, call skill_view(name, file_path) where file_path is e.g. 'references/api.md' or 'assets/config.yaml'"
if linked_files

View File

@ -241,6 +241,19 @@ Recent installs write both `hermes` and `hermes-acp` launchers into
older installs. As a manual fallback, configure Buzz's agent command as
`hermes` with args `["acp"]`.
#### Model picker
Buzz Desktop (v0.5.1+) renders Hermes' full model menu in the agent's runtime
settings. The list comes from Hermes itself over ACP: it shows every model
from providers you have authenticated in Hermes (the same inventory behind
`hermes model` and the `/model` command), so a model missing from the menu
means its provider has no credentials configured on the Hermes side.
Entry IDs take the form `provider:model` (e.g. `openrouter:z-ai/glm-5.1`), or
`custom:<name>:<model>` for custom OpenAI-compatible endpoints defined in
`config.yaml`. Picking a model applies to that agent's session; it does not
change your Hermes-wide default — use `hermes model` for that.
#### Keep Buzz agents owner-only
Buzz creates every agent with **Who can talk to this agent** set to `Owner only`.