feat(#102): core — stdlib TOML read/validate/surgical-write + atomic sidecar writer

The pure, fully-tested half of the in-app sidecar editor. No new runtime
dependency: CI's test job installs only `pytest cryptography` (not
requirements.txt) and the catalog-signature job imports bcc_core with only
cryptography, and the workflow is off-limits — so a top-level TOML import is
impossible and any pip TOML dep would leave this code untested/red on CI.

- read_toml_section() — lenient, section-aware scalar reader (string/int/
  float/bool); values it can't confidently decode are omitted (the writer
  preserves them regardless).
- toml_sections() — ordered profile/section groups for the editor's picker.
- set_toml_value() / update_toml() — SURGICAL writer: rewrites only the one
  key it's asked to, so comments, formatting, ordering and unknown keys/tables
  round-trip untouched (strictly safer than parse->dict->reserialize, which
  tomli-w wouldn't comment-preserve either). CRLF-preserving; correct scalar
  quoting/escaping; creates a missing section; deletes on value=None.
- validate_sidecar_values() — enum/port validation against ServerSpec.schema
  for the MANAGED fields only; unknown keys pass through (preserved, never a
  save-blocker) — reconciles "reject out-of-enum" with "preserve unknown keys".
- write_sidecar() — atomic temp-write + os.replace + rotating backup (reuses
  the extracted _atomic_write_text + _make_backup) then chmod 0600/0700 via
  #93's fix_permissions. Never routes through apply_servers (cardinal rule).
- _make_backup() generalised to the file's own suffix (JSON naming unchanged),
  so a .toml sidecar and a .json config keep separate backup pools.

Co-Authored-By: Claude Opus 4.8 (1M context) <noreply@anthropic.com>
This commit is contained in:
Cowork Supervisor
2026-08-13 00:18:41 -04:00
co-authored by Claude Opus 4.8
parent ded1eef2dd
commit c5a6bdd1d1
2 changed files with 544 additions and 12 deletions
+373 -12
View File
@@ -1049,21 +1049,40 @@ def _make_backup(path: Path) -> Path:
bdir = path.parent / BACKUP_DIRNAME
bdir.mkdir(exist_ok=True)
stamp = time.strftime("%Y%m%d-%H%M%S")
dest = bdir / f"{path.stem}.{stamp}.json"
# Back up under the source file's own extension so a JSON config and a TOML
# sidecar in the same directory keep separate backup pools. For a .json
# config this is byte-identical to the historical ".json" naming.
suffix = path.suffix or ".json"
dest = bdir / f"{path.stem}.{stamp}{suffix}"
# Avoid clobbering a same-second backup
n = 1
while dest.exists():
dest = bdir / f"{path.stem}.{stamp}.{n}.json"
dest = bdir / f"{path.stem}.{stamp}.{n}{suffix}"
n += 1
shutil.copy2(path, dest)
# Prune oldest beyond MAX_BACKUPS
backups = sorted(bdir.glob(f"{path.stem}.*.json"))
# Prune oldest beyond MAX_BACKUPS (scoped to this file's extension)
backups = sorted(bdir.glob(f"{path.stem}.*{suffix}"))
for old in backups[:-MAX_BACKUPS]:
with contextlib.suppress(OSError):
old.unlink()
return dest
def _atomic_write_text(p: Path, payload: str, *, prefix: str) -> None:
"""Write `payload` to `p` via a temp file + atomic os.replace on the same
filesystem. The shared mechanism behind both write_config (JSON) and
write_sidecar (TOML) — the temp file is created 0600 by mkstemp, so a secret
is never briefly world-readable mid-write."""
fd, tmp = tempfile.mkstemp(dir=str(p.parent), prefix=prefix, suffix=p.suffix or "")
try:
with os.fdopen(fd, "w", encoding="utf-8") as f:
f.write(payload)
os.replace(tmp, p) # atomic on the same filesystem
finally:
if os.path.exists(tmp):
os.remove(tmp)
def write_config(path: str | os.PathLike, cfg: dict) -> Path | None:
"""
Atomically write `cfg` to `path` (2-space pretty JSON). Backs up any existing
@@ -1075,14 +1094,7 @@ def write_config(path: str | os.PathLike, cfg: dict) -> Path | None:
backup = _make_backup(p) if p.is_file() else None
payload = json.dumps(cfg, indent=2, ensure_ascii=False) + "\n"
fd, tmp = tempfile.mkstemp(dir=str(p.parent), prefix=".tmp_mcp_", suffix=".json")
try:
with os.fdopen(fd, "w", encoding="utf-8") as f:
f.write(payload)
os.replace(tmp, p) # atomic on the same filesystem
finally:
if os.path.exists(tmp):
os.remove(tmp)
_atomic_write_text(p, payload, prefix=".tmp_mcp_")
return backup
@@ -3192,6 +3204,355 @@ def sidecar_state_changed(before: tuple | None, after: tuple | None) -> bool:
return before != after
# --------------------------------------------------------------------------- #
# Sidecar TOML editing (issue #102, epic #94)
#
# #91 shipped read-only sidecar DETECTION; #102 lets a layperson EDIT that file
# from inside BCC. The write target is a new file (ssh-mcp's config.toml) with a
# different shape, so it gets its own writer (write_sidecar) that REUSES the
# atomic-rename + rotating-backup machinery — it never routes through
# apply_servers (the cardinal rule: that only ever writes mcpServers /
# _disabledMcpServers).
#
# Dependency decision (see the PR): NO new runtime dep. CI's test job installs
# only `pytest cryptography` (not requirements.txt) and the catalog-signature
# job imports bcc_core with only cryptography present, and the workflow is
# off-limits — so a top-level TOML import is impossible and any pip TOML dep
# would leave this core untested/red on CI. Instead a stdlib-only, deliberately
# minimal reader + a SURGICAL line-editing writer: the writer only ever rewrites
# the specific key it is asked to change, so comments, formatting and every key
# it does not understand round-trip untouched — strictly safer than a
# parse->dict->reserialize pass (which tomli-w would also not comment-preserve).
# The GUI keeps a raw-text fallback for anything the form doesn't model.
# --------------------------------------------------------------------------- #
_TOML_UNPARSED = object() # sentinel: a value this lenient reader won't decode
# A top-level `[table]` or `[[array-of-tables]]` header line.
_TOML_HEADER_RE = re.compile(r"^\s*\[\[?\s*([^\[\]]+?)\s*\]\]?\s*(?:#.*)?$")
# A `key = value` line: bare, "quoted" or 'quoted' key, capturing indent, the
# key token verbatim, the '=' run (with its spacing) and the remaining RHS.
_TOML_KV_RE = re.compile(r"^(\s*)((?:\"[^\"]*\")|(?:'[^']*')|[A-Za-z0-9_.\-]+)(\s*=\s*)(.*)$")
def _toml_dump_scalar(value) -> str:
"""Serialize a Python scalar to a TOML value token.
Supports the scalar types the editor writes: bool, int, float, str. A string
is emitted as a TOML basic string with the standard escapes. Raises TypeError
for anything else — the form only ever hands us managed scalars, and the
surgical writer never re-serializes values it didn't originate.
"""
if isinstance(value, bool): # bool before int — bool is an int subclass
return "true" if value else "false"
if isinstance(value, int):
return str(value)
if isinstance(value, float):
return repr(value)
if isinstance(value, str):
esc = (
value.replace("\\", "\\\\")
.replace('"', '\\"')
.replace("\n", "\\n")
.replace("\r", "\\r")
.replace("\t", "\\t")
)
return f'"{esc}"'
raise TypeError(f"unsupported TOML scalar type: {type(value).__name__}")
def _toml_unescape_basic(s: str) -> str:
"""Decode the escape sequences a TOML basic string can carry."""
out: list[str] = []
i = 0
table = {"n": "\n", "t": "\t", "r": "\r", "\\": "\\", '"': '"', "b": "\b", "f": "\f", "0": "\0"}
while i < len(s):
ch = s[i]
if ch == "\\" and i + 1 < len(s):
out.append(table.get(s[i + 1], s[i + 1]))
i += 2
else:
out.append(ch)
i += 1
return "".join(out)
def _toml_parse_scalar(token: str):
"""Parse a TOML scalar token into a Python value, or ``_TOML_UNPARSED``.
Handles basic/literal strings, integers, floats and booleans — the values
the editor's managed fields use. Arrays, inline tables, dates and multiline
strings return the sentinel so callers leave them untouched (the surgical
writer preserves the original text regardless).
"""
t = token.strip()
if not t:
return _TOML_UNPARSED
if t in ("true", "false"):
return t == "true"
if len(t) >= 2 and t[0] == t[-1] and t[0] in "\"'":
inner = t[1:-1]
return inner if t[0] == "'" else _toml_unescape_basic(inner)
cleaned = t.replace("_", "")
if re.fullmatch(r"[+-]?\d+", cleaned):
try:
return int(cleaned)
except ValueError:
return _TOML_UNPARSED
if ("." in cleaned or "e" in cleaned.lower()) and re.fullmatch(
r"[+-]?(?:\d+\.\d*|\.\d+|\d+)(?:[eE][+-]?\d+)?", cleaned
):
try:
return float(cleaned)
except ValueError:
return _TOML_UNPARSED
return _TOML_UNPARSED
def _split_value_comment(rhs: str) -> tuple[str, str]:
"""Split a RHS into (value_token, trailing) at the first unquoted ``#``.
``trailing`` includes the run of whitespace immediately before the ``#`` so a
rewrite can splice a new value in front of it and preserve the exact spacing
and the comment. Quotes are respected so a ``#`` inside a string is not
mistaken for a comment.
"""
quote = None
for i, ch in enumerate(rhs):
if quote:
if ch == quote:
quote = None
elif ch in "\"'":
quote = ch
elif ch == "#":
j = i
while j > 0 and rhs[j - 1] in " \t":
j -= 1
return rhs[:j].rstrip(), rhs[j:]
return rhs.rstrip(), ""
def _toml_section_span(lines: list[str], section: str | None):
"""(header_idx, body_start, body_end) for a section's line range.
``section=None`` targets the top-level block (before the first header):
header_idx is None, body is [0, first-header). A named section returns the
header line index and its body [after-header, next-header). When a named
section is absent, returns (None, None, None).
"""
if section is None:
end = len(lines)
for i, raw in enumerate(lines):
if _TOML_HEADER_RE.match(raw):
end = i
break
return (None, 0, end)
for i, raw in enumerate(lines):
h = _TOML_HEADER_RE.match(raw)
if h and h.group(1).strip().strip("'\"") == section:
body_end = len(lines)
for j in range(i + 1, len(lines)):
if _TOML_HEADER_RE.match(lines[j]):
body_end = j
break
return (i, i + 1, body_end)
return (None, None, None)
def _find_key_line(lines: list[str], start: int, end: int, key: str) -> int | None:
for i in range(start, end):
m = _TOML_KV_RE.match(lines[i].rstrip("\r\n"))
if m and m.group(2).strip().strip("'\"") == key:
return i
return None
def read_toml_section(text: str, section: str | None = None) -> dict:
"""Scalar ``key -> value`` map for one section (top-level when ``section`` is None).
Lenient and deliberately minimal: only simple scalars (string/int/float/bool)
are returned. Keys whose values this reader can't confidently decode (arrays,
inline tables, dates, multiline strings) are omitted from the result — the
surgical writer preserves them regardless, and the GUI's raw-text fallback
covers editing them by hand.
"""
out: dict = {}
lines = (text or "").splitlines(keepends=True)
_, body_start, body_end = _toml_section_span(lines, section)
if body_start is None:
return out
for i in range(body_start, body_end):
raw = lines[i].rstrip("\r\n")
stripped = raw.strip()
if not stripped or stripped.startswith("#"):
continue
m = _TOML_KV_RE.match(raw)
if not m:
continue
key = m.group(2).strip().strip("'\"")
value_token, _comment = _split_value_comment(m.group(4))
value = _toml_parse_scalar(value_token)
if value is not _TOML_UNPARSED:
out[key] = value
return out
def toml_sections(text: str) -> list[str]:
"""Ordered, de-duplicated list of the top-level section names in `text`.
A ``[server]`` / ``[[hosts]]`` / ``[prod.auth]`` header contributes its first
dotted component (so ``[prod]`` and ``[prod.auth]`` are one profile group),
mirroring ``count_toml_profiles``. Lets the editor offer the sections a user
can target. Top-level (pre-header) keys are represented by ``""`` when present.
"""
names: list[str] = []
seen: set[str] = set()
saw_top = False
for raw in (text or "").splitlines():
line = raw.strip()
if not line or line.startswith("#"):
continue
h = _TOML_HEADER_RE.match(raw)
if h:
top = h.group(1).split(".", 1)[0].strip().strip("'\"")
if top and top not in seen:
seen.add(top)
names.append(top)
elif not names and _TOML_KV_RE.match(raw):
saw_top = True
if saw_top:
names.insert(0, "")
return names
def _rewrite_key_line(line: str, value, nl: str) -> str:
"""Replace only the value on an existing ``key = value`` line.
Preserves indentation, the key's original spelling, the ``=`` spacing, any
trailing comment and the line's own ending — everything but the value.
"""
stripped = line.rstrip("\r\n")
ending = line[len(stripped) :] or nl
m = _TOML_KV_RE.match(stripped)
indent, keytok, eq, rest = m.group(1), m.group(2), m.group(3), m.group(4)
_oldval, comment = _split_value_comment(rest)
return f"{indent}{keytok}{eq}{_toml_dump_scalar(value)}{comment}{ending}"
def set_toml_value(text: str, key: str, value, *, section: str | None = None) -> str:
"""Return `text` with ``[section] key`` set to `value`, surgically.
* An existing key line has only its value replaced (comment/formatting kept).
* A missing key is appended at the end of the section body.
* A missing section is created (with the key) at the end of the file.
* ``value=None`` deletes the key line (no-op if absent).
Everything else in the document — comments, blank lines, unknown keys and
tables, ordering — round-trips untouched. This is the writer the cardinal
rule calls for: it reuses nothing from apply_servers and only ever changes
the one key it is asked to.
"""
text = text or ""
nl = "\r\n" if "\r\n" in text else "\n"
lines = text.splitlines(keepends=True)
_, body_start, body_end = _toml_section_span(lines, section)
if body_start is None: # named section absent
if value is None:
return text
chunk = ""
if lines and not lines[-1].endswith(("\n", "\r")):
chunk += nl # terminate a final line that lacked a newline
if lines and lines[-1].strip():
chunk += nl # readability blank line before the new section
chunk += f"[{section}]{nl}{key} = {_toml_dump_scalar(value)}{nl}"
return text + chunk
ki = _find_key_line(lines, body_start, body_end, key)
if ki is not None:
if value is None:
del lines[ki]
else:
lines[ki] = _rewrite_key_line(lines[ki], value, nl)
return "".join(lines)
if value is None:
return text # nothing to delete
# Insert a new key at the end of the section body, before any trailing blank
# lines (which usually separate it from the next section).
insert_at = body_end
while insert_at > body_start and lines[insert_at - 1].strip() == "":
insert_at -= 1
if insert_at > 0 and not lines[insert_at - 1].endswith(("\n", "\r")):
lines[insert_at - 1] = lines[insert_at - 1] + nl
lines.insert(insert_at, f"{key} = {_toml_dump_scalar(value)}{nl}")
return "".join(lines)
def update_toml(text: str, updates: dict, *, section: str | None = None) -> str:
"""Apply several key updates to one section, surgically (see set_toml_value).
A ``None`` value deletes that key. Applied in order; each edit preserves the
rest of the document, so the result round-trips every untouched line.
"""
for key, value in updates.items():
text = set_toml_value(text, key, value, section=section)
return text
def validate_sidecar_values(values: dict, schema: dict) -> list[str]:
"""Validate proposed managed values against a ServerSpec schema.
Only keys the schema *manages* are policed (the enum/range fields the editor
renders as pick-lists / a numeric field). Any other key is passed through
untouched — BCC preserves unknown keys rather than rejecting them, so an
out-of-schema key a user already has never blocks a save. Returns a list of
plain-language problems; empty == all good.
"""
problems: list[str] = []
for key, val in values.items():
rule = schema.get(key)
if rule is None:
continue # unmanaged key — preserved, not policed
if isinstance(rule, list): # enum
if val not in rule:
allowed = ", ".join(str(x) for x in rule)
problems.append(f"{key}: {val!r} is not allowed — choose one of: {allowed}")
elif isinstance(rule, dict) and "min" in rule and "max" in rule: # numeric range
lo, hi = rule["min"], rule["max"]
if isinstance(val, bool) or not isinstance(val, int) or not (lo <= val <= hi):
problems.append(f"{key}: must be a whole number between {lo} and {hi}")
return problems
def write_sidecar(
path: str | os.PathLike,
text: str,
*,
platform: str | None = None,
chmod=None,
) -> Path | None:
"""Atomically write TOML `text` to a sidecar `path`, safely.
Reuses the same temp-write + atomic ``os.replace`` + rotating timestamped
backup as ``write_config``, then tightens the file to 0600 and its directory
to 0700 (POSIX; a clean no-op on Windows) via #93's ``fix_permissions``. It
does **not** go through ``apply_servers`` — the cardinal rule reserves that
for the JSON server keys; the sidecar is a different file with a different
shape. Returns the backup path, or None if there was no prior file to back
up. ``chmod`` is injectable for tests.
"""
p = Path(path)
p.parent.mkdir(parents=True, exist_ok=True)
backup = _make_backup(p) if p.is_file() else None
payload = text if text.endswith("\n") else text + "\n"
_atomic_write_text(p, payload, prefix=".tmp_sidecar_")
# Tighten permissions after the file is in place (the temp file was already
# 0600 from mkstemp, so the secret was never briefly world-readable).
fix_permissions(p, platform=platform, chmod=chmod)
return backup
# --------------------------------------------------------------------------- #
# Validation
# --------------------------------------------------------------------------- #