From f69e8be76b04088d0f4926b5d12a1ec193cf0026 Mon Sep 17 00:00:00 2001 From: Mother Seara Date: Thu, 10 Sep 2026 18:58:38 +0900 Subject: [PATCH 1/2] feat: independent managed workspace CLI and durable MCP tasks --- .github/workflows/ci.yml | 17 + CHANGELOG.md | 13 + MANIFEST.in | 2 + README.md | 29 +- RUNTIME_CONTRACT.md | 155 +++++++++ WORKSPACE_GUIDE.md | 108 +++++++ docs/STACK_CANONICAL.md | 2 +- mirror_stack_mcp/__init__.py | 2 +- mirror_stack_mcp/ots_anchor.py | 9 +- mirror_stack_mcp/product.py | 21 ++ mirror_stack_mcp/runtime.py | 370 ++++++++++++++++++++++ mirror_stack_mcp/server.py | 68 +++- mirror_stack_mcp/workspace.py | 562 +++++++++++++++++++++++++++++++++ pyproject.toml | 3 +- tests/smoke_installed.py | 4 +- tests/test_product.py | 217 +++++++++++++ tests/test_runtime_contract.py | 271 ++++++++++++++++ tests/test_server.py | 8 +- tests/test_stdio_runtime.py | 43 +++ 19 files changed, 1882 insertions(+), 22 deletions(-) create mode 100644 MANIFEST.in create mode 100644 RUNTIME_CONTRACT.md create mode 100644 WORKSPACE_GUIDE.md create mode 100644 mirror_stack_mcp/product.py create mode 100644 mirror_stack_mcp/runtime.py create mode 100644 mirror_stack_mcp/workspace.py create mode 100644 tests/test_product.py create mode 100644 tests/test_runtime_contract.py create mode 100644 tests/test_stdio_runtime.py diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index d83c4cf..a81ce0d 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -21,6 +21,21 @@ jobs: - name: Run tests run: pytest -v + portability: + runs-on: ${{ matrix.os }} + strategy: + fail-fast: false + matrix: + os: [macos-latest, windows-latest] + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-python@v5 + with: + python-version: "3.12" + - run: pip install -e ".[test]" + - name: Native workspace permissions, locks, CLI and stdio contracts + run: pytest -v tests/test_runtime_contract.py tests/test_stdio_runtime.py tests/test_product.py + package: # Build the wheel and exercise it in a CLEAN env — catches "works in the repo, # broken once installed" (missing modules / entry point / a stale-or-renamed @@ -41,3 +56,5 @@ jobs: run: pip install dist/*.whl - name: Smoke-test the installed package (run from /tmp, not the repo) run: cd /tmp && python "$GITHUB_WORKSPACE/tests/smoke_installed.py" + - name: Installed product CLI and MCP outside the checkout + run: cd /tmp && PRODUCT_TEST_INSTALLED=1 python "$GITHUB_WORKSPACE/tests/test_product.py" diff --git a/CHANGELOG.md b/CHANGELOG.md index 702f2b7..479ee0f 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -1,5 +1,18 @@ # Changelog +## [0.3.0] — 2026-09-10 + +- Independent `mirror-stack` setup, record, verify, doctor, connect and serve commands. +- Shared managed CLI/MCP permissions, OS-held workspace locks, durable delivery + receipts, explicit task preparation and restart-safe replay. +- Exact-file optional read-ledger grants; no required Yeoul or LaneStack workspace. +- Operator-only audited recovery retires interrupted IDs, including missing receipts; + no rollback, automatic success marking, or worker-facing unlock. +- Fail-closed ledger/path checks and collision-resistant OTS manifests. Existing + trusted-local entrypoints remain available, explicitly outside managed guarantees. +- See [workspace guide](WORKSPACE_GUIDE.md) and [runtime contract](RUNTIME_CONTRACT.md) + for migration, supported entrypoints and OS/process limitations. + ## [0.2.14] — 2026-09-09 - Pin measure-mirror v0.41.2 and provenance-mirror v0.3.1 to their exact tagged diff --git a/MANIFEST.in b/MANIFEST.in new file mode 100644 index 0000000..f7e02ef --- /dev/null +++ b/MANIFEST.in @@ -0,0 +1,2 @@ +include RUNTIME_CONTRACT.md +include WORKSPACE_GUIDE.md diff --git a/README.md b/README.md index 5131011..7730afa 100644 --- a/README.md +++ b/README.md @@ -12,6 +12,23 @@ the verdict (pass / kill / retracted / inconclusive), and every number auto-reco tamper with a sealed value in the browser and watch the hash break. Shows what the discipline looks like in practice, including the failures and retractions — not just the wins. +## Standalone workspace (v0.3.0) + +Mirror is independent of Yeoul and LaneStack. Start with the +[workspace guide / 간편 사용·복구 안내](WORKSPACE_GUIDE.md). + +```bash +pip install git+https://github.com/mirror-stack/mirror-stack-mcp@v0.3.0 +mirror-stack setup ./my-records --mode record +mirror-stack record "First note" --workspace ./my-records +mirror-stack doctor --workspace ./my-records +mirror-stack connect --workspace ./my-records +``` + +The product CLI and managed MCP share operation receipts and locks. Setup asks for +confirmation; scripts can add `--yes` after review. Existing client settings are +not changed. Raw core CLIs and arbitrary shell commands are outside this boundary. + ## Get all mirrors at once ```bash @@ -21,7 +38,11 @@ pip install git+https://github.com/mirror-stack/mirror-stack-mcp That single install pulls **measure-mirror + action-mirror + provenance-mirror** as dependencies — you don't clone four repos. (Apache-2.0, zero-dep cores.) -## Add one MCP server +## Legacy trusted-local MCP configuration + +For workspace-scoped permissions, serialized writes and restart-safe delivery +receipts, see the [managed stdio runtime contract](RUNTIME_CONTRACT.md). +The minimal configuration below is trusted-local compatibility mode, not managed mode. ```json { @@ -34,7 +55,11 @@ dependencies — you don't clone four repos. (Apache-2.0, zero-dep cores.) No `cwd`, no `PYTHONPATH` — it's a proper installed entry point. Works in Claude Code, Cursor, Windsurf, any MCP client. -## Tools (19) +## Tools (22: 19 business tools + 3 workspace tools) + +`workspace_prepare`, `workspace_execute` and `workspace_tasks` manage durable task +handles without caller-created operation IDs. They never approve permissions or +clear interrupted operations. See the [workflow contract](WORKSPACE_GUIDE.md). ### Verification depth and gate migration (0.2.14) diff --git a/RUNTIME_CONTRACT.md b/RUNTIME_CONTRACT.md new file mode 100644 index 0000000..3e61eed --- /dev/null +++ b/RUNTIME_CONTRACT.md @@ -0,0 +1,155 @@ +# Managed stdio runtime contract + +Introduced in v0.3.0; does not automatically change existing client configurations. +The transport remains **stdio**. Start with the [workspace guide](WORKSPACE_GUIDE.md): +`mirror-stack setup` saves a product-owned profile and `mirror-stack serve` applies it. + +## Modes and authority + +With `MIRROR_MCP_ROOT` unset, the server retains its **trusted-local compatibility +mode**: the caller has the server user's filesystem authority, paths keep their +previous meaning, and managed locking/receipts are not active. A startup warning +goes to stderr, never the MCP stdout stream. Setting an empty/invalid root fails; +it does not silently fall back to trusted mode. + +To opt in, set an existing, canonical, absolute workspace in the MCP client's +server environment. These are operator settings, not tool arguments: + +```json +{ + "mcpServers": { + "mirror-stack": { + "command": "mirror-stack-mcp", + "env": { + "MIRROR_MCP_ROOT": "/absolute/project", + "MIRROR_MCP_ALLOW_WRITE": "1", + "MIRROR_MCP_WRITE_TOOLS": "mm_preregister,mm_retract,am_record,am_witness,pm_verify" + } + } + } +} +``` + +- Without `MIRROR_MCP_ALLOW_WRITE=1`, managed mode is business-data read-only. + Runtime lock metadata may still be created by read calls. +- Optional `MIRROR_MCP_WRITE_TOOLS` is an exact comma-separated allowlist. An empty + value denies all writers; omission allows writers if ALLOW_WRITE is granted. +- `pm_verify` **writes a provenance ledger** and is therefore a writer, despite its name. +- OTS stamping, upgrading and verification invoke an external program and may use + the network. They additionally require BOTH `MIRROR_MCP_ALLOW_NETWORK=1` and + `MIRROR_MCP_ALLOW_EXEC=1`. Neither is enabled by the example. +- The explorer argument must exactly equal operator setting `MIRROR_MCP_EXPLORER` + (default `https://blockstream.info/api`). This is not a complete network sandbox: + redirects, OTS calendars and executable behavior still require trusted deployment + and, where necessary, OS/network egress restrictions. `OTS_BIN` is operator-owned. +- Tool paths resolve against the managed root, not the client's current directory. + Outside paths, parent traversal, symlinks, hard-linked files, special files and + targets inside `.mirror-mcp-runtime` or the workspace profile are refused. + Embedded anchor ledger paths are checked too. `MIRROR_MCP_READ_LEDGERS` is an + operator-owned JSON array of exact absolute ledger filenames, granting read + access only in designated ledger inputs. `mirror-stack link/unlink` manages it. + No external writes or automatic fallback to another project's ledger are allowed. + +This boundary is **not authentication of the `agent` string**, per-request human +approval, or an OS sandbox. An operator grants the connection these capabilities. +Use separate restricted processes for different trust levels. An actor able to +rewrite the server, change its environment, race filesystem paths, or invoke the +underlying CLIs directly can bypass this cooperative boundary. Protect the workspace +and runtime control directory with OS permissions. Windows deployments must provision +an owner-only ACL; POSIX mode checks cannot establish a Windows ACL. + +## Serialization and state + +Managed requests use one exclusive OS-held lock per workspace, including readers, +so cooperating product CLI/MCP processes using the same root do not inspect partially appended ledger writes. +The lock covers validation, ledger verification, execution and receipt persistence. +Waiting is bounded to five seconds; contention returns `BUSY`, never success. +The OS releases the lock when the process dies. **Do not delete the lock file**: +unlinking it could allow a second live lock inode and concurrent writers. + +This deliberately favors correctness over parallel throughput. All cooperating +processes must use the same canonical root. Legacy MCPs, direct mm/am/pm calls, +other writers, overlapping but differently configured roots and network filesystems +are not coordinated by this lock. Do not claim distributed or cross-tool locking. + +Before managed ledger writes, the complete existing ledger's hashes and links are +checked; empty, malformed, duplicate-key, tampered and unterminated-tail ledgers +are refused. A missing destination may be created. Post-write verification and +fsync precede the completed receipt. No old ledger is rewritten or rehashed. +The original formats and legacy hash-strength limits remain unchanged. + +The server's reminder `_shown` set is only presentation state. It resets on restart +and is not an approval, ledger, job queue, or evidence source. + +## Delivery retries and interruption + +The seven writing tools accept an additional optional `operation_id` parameter: +`mm_preregister`, `mm_retract`, `am_record`, `am_witness`, `pm_verify`, +`mm_anchor_bitcoin`, `mm_anchor_upgrade`. It is **required in managed mode**. +Supplying an ID without managed mode is refused, not silently ignored. + +```python +am_record(ledger_path="actions.jsonl", agent="worker", action="result", target="claim-1", + payload={"status": "fail", "summary": "Observed failure", "prereg_seal": "..."}, + operation_id="lane-12-session-7-result-1") +``` + +An ID is 1–128 ASCII letters/digits/`_.:-`, starting with a letter or digit. It is +workspace-wide across tools. The receipt stores a digest of normalized tool name +and arguments; it does not store the raw request separately. + +1. Persist the workspace's active marker, then a `prepared` receipt, before business + side effects. A marker pointing to a missing receipt also requires reconciliation. +2. Run under the workspace lock, then persist `done` and the exact returned result. +3. Same ID + same arguments replays the stored result without repeating the work. +4. Same ID + different arguments/tool returns `CONFLICT`. +5. An exception, timeout or crash after preparation leaves `prepared`. Subsequent + attempts return `RECONCILE_REQUIRED`, even after the process is restarted. + The active-operation marker also blocks new writes with different IDs until + reconciliation; read-only tools remain available for diagnosis. + +`done` means the tool returned and its response was persisted. It is NOT scientific +success: a persisted response can report failure or a pending Bitcoin confirmation. +Replaying returns the historical response, not a fresh verification of current files. +To deliberately refresh/poll/make a new attempt, use a new ID; do not change IDs to +work around an uncertain outcome. These are guarded retries, **not an exactly-once +transaction spanning external programs, networks, filesystem and receipts**. + +Read an operation's status without starting/retrying it: + +```bash +python -m mirror_stack_mcp.runtime lane-12-session-7-result-1 +``` + +Set the same `MIRROR_MCP_ROOT` for this diagnostic. `unknown` means no receipt was +found, not proof that nothing ran. For `prepared`, stop writers, preserve the receipt +and inspect the actual ledger/artifacts/child process state. Record an operator's +reconciliation before any new request. There is no tool that deletes pending receipts, +rolls back ledger entries, or automatically marks ambiguous operations successful. + +Receipts live in `.mirror-mcp-runtime/.json`, with owner-only POSIX +permissions and atomic replacement; file and parent directory are fsynced on POSIX. +Windows does not receive a POSIX directory-fsync guarantee. Power-loss behavior still +depends on the storage system. Receipts contain responses and may contain sensitive +data: back them up privately with their workspace, not in public exports. Do not prune +receipts while their IDs can be retried. Removing one removes its deduplication memory. + +## Integration and verification + +The product exposes `workspace_prepare`, `workspace_execute`, and `workspace_tasks`. +Any consumer, including LaneStack, can retain the returned task ID and use this +public contract. Advanced clients can still persist explicit operation IDs before +sending business-tool calls. Show `BUSY`, `CONFLICT`, `RECONCILE_REQUIRED`, and +business results separately. No consumer owns the product's permissions or state. + +Operator-only `mirror-stack recover` defaults to inspection. Explicit acknowledgement +requires a note and a stopped-children attestation. It writes a durable recovery +tombstone before removing only the active pointer; the old ID stays non-retriable +even if its receipt was missing. It does not stop processes or prove consistency. +See the [recovery procedure](WORKSPACE_GUIDE.md). + +Tests: `python -m pytest tests/test_runtime_contract.py` exercises real writers, +parallel processes, duplicate delivery, crashes, path/capability denial, invalid +ledgers and actual FastMCP tool schemas. Existing source tests remain separate. +Windows/macOS and installed-wheel verification must be reported for the environments +actually exercised; local Linux success does not certify those platforms. diff --git a/WORKSPACE_GUIDE.md b/WORKSPACE_GUIDE.md new file mode 100644 index 0000000..cf935e4 --- /dev/null +++ b/WORKSPACE_GUIDE.md @@ -0,0 +1,108 @@ +# 독립 제품 사용·운영 안내 + +거울은 독립적으로 설치·설정·사용합니다. LaneStack은 선택적인 소비자일 뿐이며, +다른 제품이나 공통 폴더, LaneStack 설정 파일이 필요하지 않습니다. + +## 처음 사용 + +```bash +pip install git+https://github.com/mirror-stack/mirror-stack-mcp@v0.3.0 +mirror-stack setup ./my-records --mode record +mirror-stack record "첫 기록" --workspace ./my-records +mirror-stack verify --workspace ./my-records +mirror-stack doctor --workspace ./my-records +mirror-stack connect --workspace ./my-records +``` + +대화형 `setup`은 폴더와 사용 방식을 묻습니다. 자동화할 때만 `--yes`를 붙이세요. +기존 파일을 덮어쓰거나 다른 제품의 설정을 변경하지 않습니다. 기본 방식은 `observe`입니다. +`doctor`는 설정·실행 전제·미처리 작업을 검사할 뿐, 업무 결과의 진실성을 인증하지 않습니다. + +- `observe`: 조회만 허용. +- `record`: 사전등록·철회·행동 기록·증언·출처 검사 기록 허용. +- 외부 OTS 실행·네트워크는 간편 프로필에서 허용하지 않습니다. 필요하면 별도 운영 설정을 검토하세요. +- `verify` 간편 명령은 기본 `actions.jsonl`의 행동 원장을 검사하며 전체 스택 검증은 아닙니다. + +`connect`가 출력한 설정에서 필요한 서버 항목만 사용 중인 MCP 클라이언트에 추가하세요. +기존 클라이언트 설정을 자동 수정하지 않습니다. 직접 서버를 띄울 때는 +`mirror-stack serve --workspace FOLDER`를 사용합니다. 전송은 stdio이며 업무 상태는 파일에 남습니다. +프로필 변경·권한 취소·업그레이드 후에는 실행 중인 서버를 종료하고 다시 연결해야 합니다. + +## 정상 사용과 재시도 + +일반 명령은 작업을 준비하여 ID를 디스크에 저장하고, ID를 표준 오류에 표시한 뒤 실행합니다. +ID를 사용자가 만들 필요는 없습니다. 고급 명령은 +`mirror-stack run TOOL --arguments '{"key":"value"}'`입니다. + +MCP 클라이언트는 다음 공개 도구를 사용합니다. + +1. `workspace_prepare(tool, arguments)`: 업무 실행 없이 작업을 저장하고 `task_id` 반환. +2. ID를 보관한 뒤 `workspace_execute(task_id)`: 기존 안전 처리로 실행. +3. 응답이 끊기면 `workspace_tasks()`로 상태를 보고 **동일 ID**로 다시 요청. + +준비 응답 유실 시 실행하지 않은 준비 작업이 중복될 수 있습니다. 같은 내용을 고의로 두 번 +요청하는 경우도 있으므로 내용만으로 사용자 의도를 합치지 않습니다. 완료 기록이 있는 +재시도는 과거 응답을 반환합니다. 읽기 전용 작업은 처리 영수증 없이 다시 조회합니다. +업무 실패나 게이트 거부도 완료된 응답일 수 있습니다. 명시적으로 내용을 수정한 다음 +새 시도를 할 때는 새 작업을 준비하세요. 불확실한 중단을 새 ID로 우회하지 마세요. + +CLI에서는 `mirror-stack tasks --workspace FOLDER`, +`mirror-stack retry TASK_ID --workspace FOLDER`를 사용합니다. +`state=returned`는 응답 전달 완료일 뿐 검증 통과가 아닙니다. +원래 결과의 `exit_code`, `ok`, `decision`, 검증 finding을 별도로 확인해야 합니다. + +## 선택적 원장 연결 + +두 제품의 작업 폴더는 **분리해도 됩니다**. 먼저 생산자에서 원장을 만들고, +소비자에서 `mirror-stack link /absolute/path/to/claims.jsonl --workspace FOLDER`를 실행합니다. +원장의 해시·체인을 확인한 뒤 그 **파일 하나의 읽기 권한**만 설정합니다. +폴더 권한이나 외부 쓰기 권한은 주지 않습니다. +`mirror-stack unlink /absolute/path/to/claims.jsonl --workspace FOLDER`로 취소할 수 있습니다. + +이 권한은 원장 입력에만 적용됩니다. 임의 작업 폴더·출력 파일·검증 명령·네트워크 권한으로 +확대되지 않습니다. 연결했다고 업무 아크에 봉인이 자동 연결되거나 증거가 자동 출판되지도 +않습니다. 원장 생산자가 쓰는 도중 소비자가 읽는 것까지 원자적으로 묶는 교차 제품 잠금은 +없습니다. 쓰기 완료 후 검증된 원장을 읽거나 별도의 안정된 스냅샷을 사용하세요. +해시 검증은 내용의 진실·작성자 신원·외부 시각을 인증하지 않습니다. + +## 중단과 복구 + +`mirror-stack recover --workspace FOLDER`는 **조회만** 합니다. +작업자용 MCP에는 설정 변경·명령 승인·중단 해제 도구를 제공하지 않습니다. + +운영자가 다음을 확인해야 합니다. + +1. 영향을 받는 서버와 남아 있는 자식 프로세스를 정지. +2. 작업 폴더 전체와 처리 기록을 함께 백업. +3. 실제 변경 파일·원장·영수증을 비교해 일관된 상태인지 판단. +4. 근거를 남긴 뒤에만 아래 명령을 실행. + +```bash +mirror-stack recover --workspace FOLDER --acknowledge --children-stopped --note "확인한 파일과 판단 근거" +``` + +이 명령은 운영자의 확인 진술을 기록하는 것이며 프로세스 종료·업무 복구를 자동 증명하지 +않습니다. 감사 기록과 재실행 금지 표식을 먼저 저장하고 활성 중단 포인터만 제거합니다. +이전 중단 작업은 영수증이 없던 경우에도 다시 실행할 수 없습니다. +완료로 위조하지 않으며 원장 삭제·롤백·재봉인도 하지 않습니다. +메타데이터가 손상됐다면 임의 삭제하지 말고 신뢰할 수 있는 백업과 대조해야 합니다. +잠금 파일은 지우지 마세요. + +## 지원 경계와 업그레이드 + +제품 CLI의 업무 명령과 MCP 도구는 같은 guarded 함수·잠금·영수증 경로를 사용합니다. +**기존 원시 CLI/스크립트, 임의 셸 명령, 편집기, 옛 서버는 이 잠금에 참여하지 않습니다.** +제품 CLI를 지원 경로로 사용하고 다른 작성자가 동시에 파일을 바꾸지 않게 운영하세요. +프로필·처리 기록·설치 코드·승인된 테스트를 작업자가 임의 수정할 수 있다면 보호가 무력화됩니다. +실행 계정 권한, Windows ACL, 필요시 컨테이너/별도 계정·네트워크 제한은 운영자가 설정해야 합니다. + +프로필 스키마는 1입니다. 설정 변경 전 이전 프로필을 비공개 이력으로 보존하고, +동시 변경 시 오래된 설정으로 덮어쓰지 않습니다. +업그레이드는 서버 정지 → 폴더 전체 백업 → 패키지 재설치 → `doctor` → 재연결 순서입니다. +업무 원장·아크와 숨김 처리 기록을 함께 유지하세요. 설정의 절대 경로를 바꾸는 폴더 이동이나 +알 수 없는 스키마는 자동 마이그레이션하지 않습니다. 먼저 별도로 검토해야 합니다. +설치 코드에는 사용자 업무 자료를 보관하지 마세요. + +네트워크 파일시스템·서로 겹치는 루트·비협조적 작성자에 대한 동시성 보장이나, +외부 명령까지 포함한 exactly-once 트랜잭션을 주장하지 않습니다. +상세 제한은 [runtime contract](RUNTIME_CONTRACT.md)를 확인하세요. diff --git a/docs/STACK_CANONICAL.md b/docs/STACK_CANONICAL.md index 0784b4a..c17d358 100644 --- a/docs/STACK_CANONICAL.md +++ b/docs/STACK_CANONICAL.md @@ -23,7 +23,7 @@ so "which one is authoritative?" has a written answer. | Surface | Lives in | Owns | |---|---|---| | **measure-mirror `stack/`** | [`mirror-stack/measure-mirror`](https://github.com/mirror-stack/measure-mirror) `stack/` | the stack **conventions + honesty docs** (`MIRROR_STACK.md`, `DISCIPLINE.md`, `PILLARS.md`, the case study); **`verify_self.py`** (one claims ledger: L1 self-chain + L3 anchors, zero external tool); **`verify_all.py`** (the orchestrator — adds L2 cross-witness *between* ledgers via `am`); `tombstone.py` | -| **mirror-stack-mcp** | this repo | the **agent MCP server** (`server.py`, 19 tools); the **`mirror-stack-gate`** enforcer CLI (`gate.py`); the **zero-config outsider** `mirror-stack-verify` CLI (`verify.py`); Bitcoin/OTS anchoring (`ots_anchor.py`) | +| **mirror-stack-mcp** | this repo | the **agent MCP server** (`server.py`, 22 tools); independent managed **`mirror-stack`** product CLI (`product.py`); the **`mirror-stack-gate`** enforcer CLI (`gate.py`); the **zero-config outsider** `mirror-stack-verify` CLI (`verify.py`); Bitcoin/OTS anchoring (`ots_anchor.py`) | Rule of thumb: **measure-mirror `stack/` = the conventions and the self/orchestrated verification that ships with the library**; **mirror-stack-mcp = how an *agent* (MCP) diff --git a/mirror_stack_mcp/__init__.py b/mirror_stack_mcp/__init__.py index 2aa8599..1d5a708 100644 --- a/mirror_stack_mcp/__init__.py +++ b/mirror_stack_mcp/__init__.py @@ -1,2 +1,2 @@ """🪞🔎🪪 Mirror Stack unified MCP server.""" -__version__ = "0.2.14" +__version__ = "0.3.0" diff --git a/mirror_stack_mcp/ots_anchor.py b/mirror_stack_mcp/ots_anchor.py index a8a2fd3..f413feb 100644 --- a/mirror_stack_mcp/ots_anchor.py +++ b/mirror_stack_mcp/ots_anchor.py @@ -21,6 +21,7 @@ import re import subprocess import urllib.request +import uuid from pathlib import Path OTS = os.environ.get("OTS_BIN", "ots") @@ -56,13 +57,17 @@ def build_manifest(ledger_paths: list[str], out_dir: str) -> tuple[str, str]: "bytes": Path(f).stat().st_size, "sha256": _sha256(f), "head_seal": _head_seal(f)}) stamp = datetime.datetime.utcnow().strftime("%Y%m%d_%H%M%S") - man = os.path.join(out_dir, f"manifest_{stamp}.json") + # Different legitimate operations in the same second must not replace evidence. + man = os.path.join(out_dir, f"manifest_{stamp}_{uuid.uuid4().hex}.json") manifest = {"_type": "ots_anchor_manifest", "ts": datetime.datetime.utcnow().isoformat() + "Z", "purpose": "Bitcoin timestamp of mirror-stack ledger heads — " "proves no-backdating, NOT content truth.", "ledger_count": len(rows), "ledgers": rows} - Path(man).write_text(json.dumps(manifest, ensure_ascii=False, indent=2)) + with open(man, "x", encoding="utf-8") as stream: + stream.write(json.dumps(manifest, ensure_ascii=False, indent=2)) + stream.flush() + os.fsync(stream.fileno()) return man, _sha256(man) diff --git a/mirror_stack_mcp/product.py b/mirror_stack_mcp/product.py new file mode 100644 index 0000000..2fcb52b --- /dev/null +++ b/mirror_stack_mcp/product.py @@ -0,0 +1,21 @@ +"""Independent product entrypoint; no consumer application dependency.""" +import os +from .workspace import Workspace + +workspace = Workspace( + product="mirror-stack", prefix="MIRROR", + control=".mirror-mcp-runtime", package="mirror_stack_mcp", + modes={"observe":[],"record":["mm_preregister","mm_retract","am_record","am_witness","pm_verify"]}, defaults={"record":["am_record","text",""],"verify":["am_verify","ledger_path","actions.jsonl"]}, +) + +def root(): + value = os.environ.get("MIRROR_MCP_ROOT") + if not value: + raise ValueError("Use mirror-stack setup, then mirror-stack serve --workspace FOLDER.") + return workspace.root(value) + +def main(): + raise SystemExit(workspace.cli()) + +if __name__ == "__main__": + main() diff --git a/mirror_stack_mcp/runtime.py b/mirror_stack_mcp/runtime.py new file mode 100644 index 0000000..f958ca5 --- /dev/null +++ b/mirror_stack_mcp/runtime.py @@ -0,0 +1,370 @@ +"""Opt-in managed stdio boundary. Files are state; MCP connections are not. + +This is cooperative serialization and capability checking, NOT an OS sandbox. +Every writer sharing a workspace must use this boundary and the same root. +""" +import errno +import functools +import hashlib +import inspect +import json +import os +from pathlib import Path +import re +import stat +import sys +import tempfile +import time +from contextlib import contextmanager + +from .integrity import read_verified + +CONTROL = ".mirror-mcp-runtime" +_ID = re.compile(r"[A-Za-z0-9][A-Za-z0-9_.:-]{0,127}\Z") + + +class RuntimeRefusal(ValueError): + """No authorization, conflicting request, or unresolved prior operation.""" + + +def _json(value): + return json.dumps(value, ensure_ascii=False, sort_keys=True, allow_nan=False) + + +def _object(pairs): + result = {} + for key, value in pairs: + if key in result: + raise RuntimeRefusal("duplicate receipt key") + result[key] = value + return result + + +def managed_root(): + value = os.environ.get("MIRROR_MCP_ROOT") + if value is None: + return None + root = Path(value) + if not value or not root.is_absolute() or not root.is_dir(): + raise RuntimeRefusal("MIRROR_MCP_ROOT must be an existing absolute workspace") + if root.resolve() == Path(root.anchor): + raise RuntimeRefusal("filesystem root is not an allowed workspace") + if root.resolve() != root: + raise RuntimeRefusal("MIRROR_MCP_ROOT must be canonical and not a symlink") + return root + + +def scoped_path(value, root=None, external_read=False): + root = root or managed_root() + if root is None: + return str(value) + path = Path(value) + path = path if path.is_absolute() else root / path + if ".." in path.parts: + raise RuntimeRefusal("parent traversal is not permitted") + try: + relative = path.relative_to(root) + except ValueError: + if external_read: + from .workspace import checked_path + grants = json.loads(os.environ.get("MIRROR_MCP_READ_LEDGERS", "[]")) + if (not isinstance(grants, list) or len(grants) > 100 or + any(not isinstance(p, str) or not Path(p).is_absolute() for p in grants)): + raise RuntimeRefusal("invalid external ledger grants") + if str(path) in grants and checked_path(path, existing=True).is_file(): + return str(path) + raise RuntimeRefusal("path is outside MIRROR_MCP_ROOT") from None + if CONTROL in relative.parts or ".mirror-stack-workspace.json" in relative.parts: + raise RuntimeRefusal("runtime control files are not tool targets") + cursor = root + for part in relative.parts: + if ":" in part or part.endswith((" ", ".")) or any(ord(c) < 32 for c in part): + raise RuntimeRefusal("ambiguous path component is not permitted") + cursor = cursor / part + if cursor.is_symlink(): + raise RuntimeRefusal("symlink targets are not permitted") + if cursor.exists(): + info = cursor.lstat() + if getattr(info, "st_file_attributes", 0) & 0x400: + raise RuntimeRefusal("reparse point targets are not permitted") + if not (stat.S_ISDIR(info.st_mode) or stat.S_ISREG(info.st_mode)): + raise RuntimeRefusal("special file targets are not permitted") + if stat.S_ISREG(info.st_mode) and info.st_nlink != 1: + raise RuntimeRefusal("hard-linked file targets are not permitted") + return str(path) + + +def _control(root): + directory = root / CONTROL + if directory.is_symlink(): + raise RuntimeRefusal("runtime directory cannot be a symlink") + directory.mkdir(mode=0o700, exist_ok=True) + info = directory.lstat() + if getattr(info, "st_file_attributes", 0) & 0x400: + raise RuntimeRefusal("runtime directory cannot be a reparse point") + if not stat.S_ISDIR(info.st_mode): + raise RuntimeRefusal("invalid runtime directory") + if os.name == "posix" and (info.st_uid != os.geteuid() or info.st_mode & 0o077): + raise RuntimeRefusal("runtime directory must be owner-only (0700)") + _sync_directory(root) + return directory + + +def _private_open(path, flags): + if path.is_symlink() or (path.exists() and getattr(path.lstat(), "st_file_attributes", 0) & 0x400): + raise RuntimeRefusal("runtime file cannot be a symlink") + fd = os.open(path, flags | getattr(os, "O_NOFOLLOW", 0), 0o600) + info = os.fstat(fd) + if (not stat.S_ISREG(info.st_mode) or info.st_nlink != 1 or + (os.name == "posix" and (info.st_uid != os.geteuid() or info.st_mode & 0o077))): + os.close(fd) + raise RuntimeRefusal("runtime file must be a private regular file") + return fd + + +@contextmanager +def workspace_lock(root, timeout=5.0): + """OS-held lock: process death releases it; the lock file is NEVER deleted.""" + directory = _control(root) + fd = _private_open(directory / "workspace.lock", os.O_CREAT | os.O_RDWR) + acquired = False + try: + if os.fstat(fd).st_size == 0: + os.write(fd, b"\0") + end = time.monotonic() + timeout + while True: + try: + if os.name == "nt": + import msvcrt + os.lseek(fd, 0, os.SEEK_SET) + msvcrt.locking(fd, msvcrt.LK_NBLCK, 1) + else: + import fcntl + fcntl.flock(fd, fcntl.LOCK_EX | fcntl.LOCK_NB) + acquired = True + break + except OSError as exc: + if exc.errno not in (errno.EACCES, errno.EAGAIN, errno.EDEADLK): + raise + if time.monotonic() >= end: + raise RuntimeRefusal("BUSY: workspace locked; retry same operation_id") from None + time.sleep(0.025) + yield directory + finally: + if acquired: + if os.name == "nt": + import msvcrt + os.lseek(fd, 0, os.SEEK_SET) + msvcrt.locking(fd, msvcrt.LK_UNLCK, 1) + else: + import fcntl + fcntl.flock(fd, fcntl.LOCK_UN) + os.close(fd) + + +def _sync_directory(directory): + if os.name == "posix": + fd = os.open(directory, os.O_RDONLY | getattr(os, "O_DIRECTORY", 0)) + try: + os.fsync(fd) + finally: + os.close(fd) + + +def _store(path, record): + raw = _json(record) + if len(raw.encode("utf-8")) > 16 * 1024 * 1024: + raise RuntimeRefusal("RECONCILE_REQUIRED: response exceeds receipt 16 MiB limit") + fd, name = tempfile.mkstemp(prefix=".receipt-", dir=path.parent) + try: + with os.fdopen(fd, "w", encoding="utf-8") as stream: + stream.write(raw + "\n") + stream.flush() + os.fsync(stream.fileno()) + os.replace(name, path) + _sync_directory(path.parent) + finally: + if os.path.exists(name): + os.unlink(name) + + +def _read_receipt(path): + fd = _private_open(path, os.O_RDONLY) + with os.fdopen(fd, encoding="utf-8") as stream: + if os.fstat(stream.fileno()).st_size > 16 * 1024 * 1024: + raise RuntimeRefusal("RECONCILE_REQUIRED: oversized receipt") + try: + row = json.load(stream, object_pairs_hook=_object, + parse_constant=lambda _: (_ for _ in ()).throw(ValueError("nonfinite JSON"))) + except (ValueError, UnicodeError): + raise RuntimeRefusal("RECONCILE_REQUIRED: invalid receipt; preserve it") from None + if not isinstance(row, dict) or row.get("schema") != 1: + raise RuntimeRefusal("RECONCILE_REQUIRED: unsupported receipt") + return row + + +def receipt_path(root, operation_id): + if not isinstance(operation_id, str) or not _ID.fullmatch(operation_id): + raise RuntimeRefusal("operation_id must be 1..128 ASCII letters/digits/_.:-") + return root / CONTROL / (hashlib.sha256(operation_id.encode()).hexdigest() + ".json") + + +def operation_status(operation_id): + """Read-only receipt summary; absence is not evidence that an operation never ran.""" + root = managed_root() + if root is None: + raise RuntimeRefusal("operation status requires MIRROR_MCP_ROOT") + path = receipt_path(root, operation_id) + if (path.parent / "recovery" / path.name).exists(): + return {"operation_id": operation_id, "status": "retired"} + if not path.exists() and not path.is_symlink(): + return {"operation_id": operation_id, "status": "unknown"} + row = _read_receipt(path) + if row.get("operation_id") != operation_id: + raise RuntimeRefusal("RECONCILE_REQUIRED: receipt identity mismatch") + return {key: row.get(key) for key in ("operation_id", "tool", "status", "started_at", "finished_at")} + + +def _check_ledgers(arguments, names, allow_missing): + for name in names: + value = arguments.get(name) + if not value: + continue + for item in value if isinstance(value, list) else [value]: + path = Path(item) + if not path.exists() and allow_missing: + continue + _, error = read_verified(path) + if error: + raise RuntimeRefusal("INVALID_LEDGER: " + error) + # A complete JSON record without a terminator is readable but not append-safe. + with path.open("rb") as stream: + stream.seek(-1, os.SEEK_END) + if stream.read(1) not in (b"\n", b"\r"): + raise RuntimeRefusal("INVALID_LEDGER: unterminated tail; explicit recovery required") + + +def guarded(*, paths=(), write=False, ledgers=(), read_ledgers=(), read_paths=(), network=False): + """Preserve public signatures for FastMCP; operation_id is explicit on writers.""" + def decorate(function): + signature = inspect.signature(function) + + @functools.wraps(function) + def call(*args, **kwargs): + bound = signature.bind(*args, **kwargs) + bound.apply_defaults() + arguments = bound.arguments + operation_id = arguments.get("operation_id") + root = managed_root() + if root is None: + if operation_id is not None: + raise RuntimeRefusal("operation_id requires managed mode: set MIRROR_MCP_ROOT") + return function(*args, **kwargs) + if write and os.environ.get("MIRROR_MCP_ALLOW_WRITE") != "1": + raise RuntimeRefusal("DENIED: managed mode is read-only") + allowed = os.environ.get("MIRROR_MCP_WRITE_TOOLS") + if write and allowed is not None and function.__name__ not in allowed.split(","): + raise RuntimeRefusal("DENIED: tool not in MIRROR_MCP_WRITE_TOOLS") + if network and (os.environ.get("MIRROR_MCP_ALLOW_NETWORK") != "1" or + os.environ.get("MIRROR_MCP_ALLOW_EXEC") != "1"): + raise RuntimeRefusal("DENIED: this tool requires network and subprocess capabilities") + if network and "explorer" in arguments: + approved = os.environ.get("MIRROR_MCP_EXPLORER", "https://blockstream.info/api") + if arguments["explorer"] != approved: + raise RuntimeRefusal("DENIED: explorer is not the operator-approved endpoint") + receipt = receipt_path(root, operation_id) if write else None + for name in paths: + value = arguments.get(name) + if value is not None and value != "": + external = name in read_ledgers or name in read_paths + arguments[name] = ([scoped_path(p, root, external) for p in value] if isinstance(value, list) + else scoped_path(value, root, external)) + encoded = _json({"tool": function.__name__, "arguments": arguments}).encode() + if len(encoded) > 8 * 1024 * 1024: + raise RuntimeRefusal("request exceeds managed 8 MiB limit") + digest = hashlib.sha256(encoded).hexdigest() + with workspace_lock(root): + # Recheck after waiting, before inspecting receipts or touching inputs. + for name in paths: + value = arguments.get(name) + if value: + for item in value if isinstance(value, list) else [value]: + scoped_path(item, root, name in read_ledgers or name in read_paths) + if receipt is not None and (root / CONTROL / "recovery" / receipt.name).exists(): + raise RuntimeRefusal("RECONCILE_REQUIRED: retired operation cannot be retried") + if receipt is not None and (receipt.exists() or receipt.is_symlink()): + previous = _read_receipt(receipt) + if previous.get("digest") != digest or previous.get("operation_id") != operation_id: + raise RuntimeRefusal("CONFLICT: operation_id already bound to different inputs") + if previous.get("status") != "done" or "result" not in previous: + raise RuntimeRefusal("RECONCILE_REQUIRED: interrupted operation; do not retry with a new ID") + return previous["result"] + if write: + active = root / CONTROL / "active.json" + if active.exists() or active.is_symlink(): + marker = _read_receipt(active) + name = marker.get("receipt", "") + if not isinstance(name, str) or not re.fullmatch(r"[0-9a-f]{64}\.json", name): + raise RuntimeRefusal("RECONCILE_REQUIRED: invalid active operation marker") + try: + previous = _read_receipt(active.parent / name) + except OSError: + raise RuntimeRefusal("RECONCILE_REQUIRED: active operation receipt unavailable") from None + if previous.get("status") != "done": + raise RuntimeRefusal("RECONCILE_REQUIRED: workspace has an interrupted operation") + for name in read_ledgers: + value = arguments.get(name) + for item in value if isinstance(value, list) else [value]: + _, error = read_verified(item) + if error: + raise RuntimeRefusal("INVALID_LEDGER: " + error) + _check_ledgers(arguments, ledgers, allow_missing=True) + pending = {"schema": 1, "operation_id": operation_id, "tool": function.__name__, + "digest": digest, "status": "prepared", "started_at": time.time()} + # Arm the workspace first. A crash between these two writes must + # block, even if the receipt is still missing and no effect ran. + _store(active, {"schema": 1, "receipt": receipt.name}) + _store(receipt, pending) + # Exceptions leave the prepared receipt intact, including uncertain timeouts. + result = function(*bound.args, **bound.kwargs) + if write: + _check_ledgers(arguments, ledgers, allow_missing=False) + for name in ledgers: + value = arguments.get(name) + for item in (value if isinstance(value, list) else [value]) if value else []: + with open(item, "r+b") as stream: + os.fsync(stream.fileno()) + _sync_directory(Path(item).parent) + # Match first delivery's JSON order to replay, including MCP text blocks. + result = json.loads(_json(result)) + _store(receipt, {**pending, "status": "done", "result": result, + "finished_at": time.time()}) + return result + if write: + call.__doc__ = (function.__doc__ or "") + ( + "\nManaged mode requires operation_id and operator-granted write capability. " + "Reuse the same ID only for delivery retries of the same request; completed " + "responses replay, changed arguments conflict, interrupted requests require reconciliation.") + return call + return decorate + + +def announce(): + root = managed_root() + if root is None: + print("mirror-stack: trusted-local mode; no managed path/receipt/serialization boundary. " + "Set MIRROR_MCP_ROOT for managed mode.", file=sys.stderr) + else: + print("mirror-stack: managed stdio; read-only unless MIRROR_MCP_ALLOW_WRITE=1; " + "cooperative locking, not an OS sandbox.", file=sys.stderr) + + +if __name__ == "__main__": + import argparse + parser = argparse.ArgumentParser(description="Read a managed operation's receipt status (no recovery writes)") + parser.add_argument("operation_id") + args = parser.parse_args() + try: + print(_json(operation_status(args.operation_id))) + except (RuntimeRefusal, OSError) as exc: + parser.exit(2, str(exc) + "\n") diff --git a/mirror_stack_mcp/server.py b/mirror_stack_mcp/server.py index fd061fb..3b2da95 100644 --- a/mirror_stack_mcp/server.py +++ b/mirror_stack_mcp/server.py @@ -24,6 +24,7 @@ from . import ots_anchor from . import __version__ +from .runtime import guarded, scoped_path, announce DISCIPLINE = """\ 🪞🔎🪪 MIRROR STACK — discipline for honest measurement (read on connect). @@ -212,12 +213,14 @@ def _birth(path: str, existed: bool) -> dict: # ───────────────────────── 🪞 measure-mirror (claims) ───────────────────────── @mcp.tool() +@guarded(paths=("ledger_path",), write=True, ledgers=("ledger_path",)) def mm_preregister(ledger_path: str, claim_id: str, metric: str, min_n: int = 200, baseline: float = 0.5, pass_threshold: float = 0.6, kill_condition: str | None = None, kill_threshold: dict | None = None, depends_on: list[str] | None = None, metric_range: list | str | None = None, chance: float | None = None, - pre_seal_checks: list[str | dict] | None = None) -> dict: + pre_seal_checks: list[str | dict] | None = None, + operation_id: str | None = None) -> dict: """Seal a claim BEFORE measuring (preregistration). kill_condition/threshold = what falsifies it. For a non-[0,1] metric, declare metric_range (e.g. [0,100] for a %, or "unbounded" for a @@ -249,12 +252,14 @@ def mm_preregister(ledger_path: str, claim_id: str, metric: str, min_n: int = 20 @mcp.tool() +@guarded(paths=("ledger_path",), read_paths=("ledger_path",)) def mm_verify(ledger_path: str, data: dict, groups: list[str] | None = None) -> list[str]: """Umbrella verify: runs every probe whose input key is present in `data` (acc/n/seed_results/scores/...).""" return _remind("mm_verify", _compact(_findings(mm.verify(ledger_path, data, groups=groups)))) @mcp.tool() +@guarded(paths=("ledger_path",), read_paths=("ledger_path",)) def mm_audit(ledger_path: str, claim_id: str, reported_metric: str, reported_acc: float, n: int, baseline: float | None = None, metric_range: list | str | None = None, chance: float | None = None) -> list[str]: @@ -269,6 +274,7 @@ def mm_audit(ledger_path: str, claim_id: str, reported_metric: str, reported_acc @mcp.tool() +@guarded() def mm_power_check(n: int, baseline: float, min_detectable_effect: float = 0.05, target_power: float = 0.8) -> str: """False-negative guard: is n big enough to detect the minimum effect? (design-time).""" @@ -277,6 +283,7 @@ def mm_power_check(n: int, baseline: float, min_detectable_effect: float = 0.05, @mcp.tool() +@guarded(paths=("ledger_path", "am_ledger"), read_paths=("ledger_path", "am_ledger")) def mm_falsifiability_check(ledger_path: str, claim_id: str, reported_acc: float | None = None, am_ledger: str | None = None) -> str: @@ -291,6 +298,7 @@ def mm_falsifiability_check(ledger_path: str, claim_id: str, @mcp.tool() +@guarded(paths=("ledger_path",), read_paths=("ledger_path",)) def mm_prereg_lint(ledger_path: str, claim_id: str | None = None) -> list[str]: """🔍 Lint a sealed preregistration for QUALITY defects — the cheap machine-check to run right before spending compute. @@ -313,19 +321,23 @@ def mm_prereg_lint(ledger_path: str, claim_id: str | None = None) -> list[str]: @mcp.tool() +@guarded() def mm_leakage_check(train_items: list, test_items: list) -> str: """Detect train∩test contamination via hash intersection.""" return str(mm.leakage_check(train_items, test_items)) @mcp.tool() +@guarded() def mm_multiseed_check(seed_results: list[float], baseline: float = 0.5) -> str: """Flag unstable signal / lucky seed across multiple runs.""" return str(mm.multiseed_check(seed_results, baseline=baseline)) @mcp.tool() -def mm_retract(ledger_path: str, claim_id: str, reason: str) -> dict: +@guarded(paths=("ledger_path",), write=True, ledgers=("ledger_path",)) +def mm_retract(ledger_path: str, claim_id: str, reason: str, + operation_id: str | None = None) -> dict: """Append a chain-linked retraction (cannot be silently deleted; dependents go STALE).""" existed = os.path.exists(ledger_path) return _remind("mm_retract", {**mm.retract(ledger_path, claim_id, reason), @@ -333,13 +345,16 @@ def mm_retract(ledger_path: str, claim_id: str, reason: str) -> dict: @mcp.tool() +@guarded(paths=("ledger_path",), read_paths=("ledger_path",)) def mm_anchor(ledger_path: str) -> dict: """Tamper-evident snapshot (entry_count, head_seal, file hash) to store OUTSIDE the ledger.""" return mm.anchor(ledger_path) @mcp.tool() -def mm_anchor_bitcoin(ledger_paths: list[str], out_dir: str) -> dict: +@guarded(paths=("ledger_paths", "out_dir"), write=True, network=True, read_ledgers=("ledger_paths",)) +def mm_anchor_bitcoin(ledger_paths: list[str], out_dir: str, + operation_id: str | None = None) -> dict: """Real external anchor (L3): timestamp ledger HEADS into Bitcoin via OpenTimestamps. Upgrades mm_anchor from a LOCAL snapshot to an EXTERNAL clock — proves the heads existed @@ -351,14 +366,17 @@ def mm_anchor_bitcoin(ledger_paths: list[str], out_dir: str) -> dict: @mcp.tool() -def mm_anchor_upgrade(ots_path: str) -> dict: +@guarded(paths=("ots_path",), write=True, network=True) +def mm_anchor_upgrade(ots_path: str, operation_id: str | None = None) -> dict: """Retrieve the Bitcoin block attestation for a pending .ots proof (run ~1-3h after stamping). Returns state='bitcoin_confirmed' with block_height once a calendar has committed to a block, else 'pending' (retry later).""" + scoped_path(ots_path + ".bak") return ots_anchor.upgrade(ots_path) @mcp.tool() +@guarded(paths=("ots_path",), network=True) def mm_anchor_verify(ots_path: str, explorer: str = "https://blockstream.info/api") -> dict: """Verify a Bitcoin-anchored proof WITHOUT a local Bitcoin node, by cross-checking the block merkle root against a public explorer. Returns block height, block time, and whether the @@ -367,6 +385,7 @@ def mm_anchor_verify(ots_path: str, explorer: str = "https://blockstream.info/ap @mcp.tool() +@guarded(paths=("ledger_path", "am_ledger"), read_paths=("ledger_path", "am_ledger")) def mm_preflight(ledger_path: str, claim_id: str, gate: str = "compute", am_ledger: str | None = None, reported_acc: float | None = None) -> dict: """GO/BLOCK gate primitive — wire it into a compute launcher or a pre-publish/commit hook. @@ -391,8 +410,9 @@ def mm_preflight(ledger_path: str, claim_id: str, gate: str = "compute", # ───────────────────────── 🪪 action-mirror (actions) ───────────────────────── @mcp.tool() +@guarded(paths=("ledger_path",), write=True, ledgers=("ledger_path",)) def am_record(ledger_path: str, agent: str, action: str, target: str | None = None, - payload: dict | None = None) -> dict: + payload: dict | None = None, operation_id: str | None = None) -> dict: """Seal one agent action. Set target= to tie the action to a claim (J1).""" existed = os.path.exists(ledger_path) return _remind("am_record", @@ -402,7 +422,9 @@ def am_record(ledger_path: str, agent: str, action: str, target: str | None = No @mcp.tool() -def am_witness(my_ledger: str, peer_ledger: str, peer_name: str) -> dict: +@guarded(paths=("my_ledger", "peer_ledger"), write=True, ledgers=("my_ledger",), read_ledgers=("peer_ledger",)) +def am_witness(my_ledger: str, peer_ledger: str, peer_name: str, + operation_id: str | None = None) -> dict: """Pin a peer ledger's head into mine (J3). Catches whole-ledger replacement that chains miss.""" existed = os.path.exists(my_ledger) return {**am.witness_peer(my_ledger, peer_ledger, peer_name=peer_name), @@ -410,6 +432,7 @@ def am_witness(my_ledger: str, peer_ledger: str, peer_name: str) -> dict: @mcp.tool() +@guarded(paths=("ledger_path",), read_paths=("ledger_path",)) def am_verify(ledger_path: str) -> list[str]: """Verify an action ledger's hash chain (edits/deletions/insertions detected).""" return _compact(_findings(am.verify_chain(ledger_path))) @@ -417,8 +440,9 @@ def am_verify(ledger_path: str) -> list[str]: # ───────────────────────── 🔎 provenance-mirror (artifacts) ──────────────────── @mcp.tool() +@guarded(paths=("file_path", "ledger_path"), write=True, ledgers=("ledger_path",)) def pm_verify(file_path: str, ledger_path: str = "pm_ledger.jsonl", - origin: str | None = None) -> dict: + origin: str | None = None, operation_id: str | None = None) -> dict: """Verify a content file's provenance/integrity across 5 signals (a verifier, not a detector).""" existed = os.path.exists(ledger_path) return {**pm.verify(file_path, ledger_path=ledger_path, origin=origin), @@ -427,6 +451,7 @@ def pm_verify(file_path: str, ledger_path: str = "pm_ledger.jsonl", # ───────────────────────── 🪞🔎🪪 stack-level ───────────────────────────────── @mcp.tool() +@guarded(paths=("mm_ledger", "anchor_dir", "am_ledger"), read_paths=("mm_ledger", "am_ledger")) def stack_verify_all(mm_ledger: str, anchor_dir: str | None = None, am_ledger: str | None = None, am_peer_name: str | None = None) -> dict: """Verify the layers you point it at: mm chain (L1) + anchors (L3) + cross-witness (L2). @@ -459,10 +484,11 @@ def add(level, layer, name, msg): if anchor_dir: for af in sorted(Path(anchor_dir).glob("anchor_*.json")): + scoped_path(af) a = json.loads(af.read_text()) - lp = Path(a["ledger_path"]) + lp = Path(scoped_path(a["ledger_path"], external_read=True)) if not lp.exists(): - lp = af.parent / lp.name + lp = Path(scoped_path(af.parent / lp.name)) cur = hashlib.sha256(lp.read_bytes()).hexdigest() if lp.exists() else "" if cur == a["anchor_hash"]: add(True, "L3 anchor", af.name, "intact") @@ -505,7 +531,31 @@ def add(level, layer, name, msg): return _remind("stack_verify_all", result) + +@mcp.tool() +def workspace_prepare(tool: str, arguments: dict) -> dict: + """Prepare a durable task WITHOUT executing it. Retain task_id before execute. + Reuse the same task_id after a lost execute response; never create a new task + to bypass an interrupted operation. Product modes still restrict permissions.""" + from .product import workspace, root + return workspace.prepare(root(), tool, arguments) + + +@mcp.tool() +def workspace_execute(task_id: str) -> dict: + """Execute one prepared task, or replay its recorded response. No recovery or elevation.""" + from .product import workspace, root + return workspace.execute(root(), task_id) + + +@mcp.tool() +def workspace_tasks() -> dict: + """Inspect task receipts, without executing business work or clearing pending state.""" + from .product import workspace, root + return workspace.tasks(root()) + def main(): + announce() mcp.run() diff --git a/mirror_stack_mcp/workspace.py b/mirror_stack_mcp/workspace.py new file mode 100644 index 0000000..d438583 --- /dev/null +++ b/mirror_stack_mcp/workspace.py @@ -0,0 +1,562 @@ +"""Standalone workspace UX, vendored per product; stdlib only. + +No dependency on LaneStack or the other product. Operator configuration is not +a sandbox. All business operations dispatch to the same guarded functions as MCP. +""" +import argparse +from contextlib import contextmanager +import hashlib +import importlib +import importlib.util +import inspect +import json +import os +from pathlib import Path +import re +import shutil +import stat +import sys +import tempfile +import time +import uuid + +JOB_ID = re.compile(r"[a-f0-9]{32}\Z") + + +def canonical(value): + return json.dumps(value, ensure_ascii=False, sort_keys=True, allow_nan=False) + + +def checked_path(value, existing=False): + path = Path(value).absolute() + if ".." in path.parts or path.parent == path: + raise ValueError("Choose one project folder, not a filesystem root.") + for part in [*reversed(path.parents), path]: + try: + info = part.lstat() + except FileNotFoundError: + continue + if stat.S_ISLNK(info.st_mode) or getattr(info, "st_file_attributes", 0) & 0x400: + raise ValueError("Linked paths are not allowed: " + str(part)) + if not (stat.S_ISDIR(info.st_mode) or stat.S_ISREG(info.st_mode)): + raise ValueError("Special files are not allowed.") + if stat.S_ISREG(info.st_mode) and info.st_nlink != 1: + raise ValueError("Hardlinked files are not allowed.") + if existing and not path.exists(): + raise ValueError("Path does not exist: " + str(path)) + return path + + +def read_json(path): + path = checked_path(path, existing=True) + if path.stat().st_size > 8 * 1024 * 1024: + raise ValueError("JSON exceeds 8 MiB.") + def pairs(items): + result = {} + for key, value in items: + if key in result: + raise ValueError("Duplicate JSON key: " + key) + result[key] = value + return result + def constant(value): + raise ValueError("Nonfinite JSON: " + value) + result = json.loads(path.read_text(encoding="utf-8"), object_pairs_hook=pairs, + parse_constant=constant) + canonical(result) # Also rejects overflowed floats. + return result + + +def sync_dir(path): + if os.name == "posix": + fd = os.open(path, os.O_RDONLY | os.O_DIRECTORY) + try: + os.fsync(fd) + finally: + os.close(fd) + + +def write_json(path, value): + checked_path(path) + raw = canonical(value) + if len(raw.encode()) > 8 * 1024 * 1024: + raise ValueError("JSON exceeds 8 MiB.") + fd, temporary = tempfile.mkstemp(prefix=".write-", dir=path.parent) + try: + with os.fdopen(fd, "w", encoding="utf-8") as stream: + stream.write(raw + "\n") + stream.flush() + os.fsync(stream.fileno()) + os.replace(temporary, path) + sync_dir(path.parent) + finally: + if os.path.exists(temporary): + os.unlink(temporary) + + +class Workspace: + def __init__(self, product, prefix, control, package, modes, defaults): + self.product, self.prefix, self.control = product, prefix, control + self.package, self.modes, self.defaults = package, modes, defaults + self.config_name = "." + product + "-workspace.json" + + @property + def runtime(self): + return importlib.import_module(self.package + ".runtime") + + def tools(self): + server = importlib.import_module(self.package + ".server") + return {tool.name: getattr(server, tool.name) for tool in server.mcp._tool_manager.list_tools() + if not tool.name.startswith("workspace_")} + + def root(self, value): + root = checked_path(value, existing=True) + if not root.is_dir(): + raise ValueError("Workspace must be a directory.") + return root + + def load(self, root): + root = self.root(root) + config = read_json(root / self.config_name) + expected = {"schema", "product", "root", "revision", "mode", "external_ledgers", "verification"} + if (not isinstance(config, dict) or set(config) != expected or type(config["schema"]) is not int + or config["schema"] != 1 or config["product"] != self.product or config["root"] != str(root) + or config["mode"] not in self.modes or not isinstance(config["revision"], str) + or not isinstance(config["external_ledgers"], list) or len(config["external_ledgers"]) > 100 + or any(not isinstance(p, str) or not Path(p).is_absolute() for p in config["external_ledgers"]) + or (config["verification"] is not None and + (not isinstance(config["verification"], dict) or + set(config["verification"]) != {"path", "sha256"}))): + raise ValueError("Invalid or relocated workspace config; do not silently repair it.") + verification = config["verification"] + if verification is not None and (not isinstance(verification["path"], str) or + not isinstance(verification["sha256"], str)): + raise ValueError("Invalid verification approval.") + return config + + def save_config(self, root, config, initial=False, expected_revision=None): + with self.runtime.workspace_lock(root): + path = root / self.config_name + if initial and path.exists(): + raise ValueError("Already configured; use configure, not setup.") + history = root / self.control / "config-history" + checked_path(history) + history.mkdir(mode=0o700, exist_ok=True) + if path.exists(): + previous = self.load(root) + if expected_revision is not None and previous["revision"] != expected_revision: + raise ValueError("Configuration changed concurrently; reload before updating.") + write_json(history / (uuid.uuid4().hex + ".json"), previous) + write_json(path, config) + + def setup(self, folder, mode): + if mode not in self.modes: + raise ValueError("Unknown mode.") + root = checked_path(folder) + if (root / self.config_name).exists(): + raise ValueError("Already configured; no files changed.") + root.mkdir(mode=0o700, parents=True, exist_ok=True) + config = dict(schema=1, product=self.product, root=str(root), revision=uuid.uuid4().hex, + mode=mode, external_ledgers=[], verification=None) + self.save_config(root, config, initial=True) + return {"workspace": str(root), "mode": mode, "next": self.product + " doctor --workspace " + str(root)} + + def configure(self, root, mode): + config = self.load(root) + revision = config["revision"] + if mode not in self.modes: + raise ValueError("Unknown mode.") + config.update(mode=mode, revision=uuid.uuid4().hex) + # Mode changes revoke shell approval, never silently keep an old execution grant. + config["verification"] = None + self.save_config(Path(root), config, expected_revision=revision) + return config + + def environment(self, config): + root = config["root"] + allowed = self.modes[config["mode"]] + env = {self.prefix + "_MCP_ROOT": root, + self.prefix + "_MCP_ALLOW_WRITE": "1" if allowed else "0", + self.prefix + "_MCP_WRITE_TOOLS": ",".join(allowed), + self.prefix + "_MCP_ALLOW_EXEC": "0", + self.prefix + "_MCP_ALLOW_NETWORK": "0", + self.prefix + "_MCP_READ_LEDGERS": canonical(config["external_ledgers"])} + if self.product == "yeoul": + env.update(YEOUL_PROJECTS=str(Path(root) / "projects"), + YEOUL_INDEX=str(Path(root) / "KNOWLEDGE_INDEX.md"), + YEOUL_CLOSED_REGISTRY=str(Path(root) / "registry/closed_questions.jsonl")) + approval = config["verification"] + if config["mode"] == "develop" and approval: + path = checked_path(approval["path"], existing=True) + if not path.is_relative_to(Path(root) / ".yeoul-approved"): + raise ValueError("Approved baseline must be inside .yeoul-approved.") + if hashlib.sha256(path.read_bytes()).hexdigest() != approval["sha256"]: + raise ValueError("Approved baseline changed; re-approval required.") + env.update(YEOUL_MCP_ALLOW_EXEC="1", YEOUL_MCP_VERIFY_BASELINE=str(path), + YEOUL_MCP_VERIFY_BASELINE_SHA256=approval["sha256"]) + return env + + @contextmanager + def activated(self, root): + config = self.load(root) + # Only CLI/startup uses this. MCP task tools never elevate by loading a profile. + previous = dict(os.environ) + clean = [key for key in os.environ if key.startswith(self.prefix + "_")] + try: + for key in clean: + os.environ.pop(key, None) + os.environ.update(self.environment(config)) + yield config + finally: + os.environ.clear() + os.environ.update(previous) + + def connection(self, root): + self.load(root) + return {"mcpServers": {self.product: { + "command": sys.executable, + "args": ["-m", self.package + ".product", "serve", "--workspace", str(self.root(root))] + }}} + + def require_managed(self, root): + value = os.environ.get(self.prefix + "_MCP_ROOT") + if not value or self.root(value) != self.root(root): + raise ValueError("Start this workspace's managed server before preparing tasks.") + + def job_path(self, root, job_id): + if not isinstance(job_id, str) or not JOB_ID.fullmatch(job_id): + raise ValueError("Invalid task ID.") + return Path(root) / self.control / "tasks" / (job_id + ".json") + + def load_job(self, root, job_id): + job = read_json(self.job_path(root, job_id)) + if (not isinstance(job, dict) or set(job) != {"schema", "id", "tool", "arguments", "created", "digest"} + or type(job["schema"]) is not int or job["schema"] != 1 or job["id"] != job_id + or not isinstance(job["tool"], str) or not isinstance(job["arguments"], dict) + or "operation_id" in job["arguments"] or type(job["created"]) not in (int, float)): + raise ValueError("Invalid task record; preserve it for inspection.") + unsigned = {k: v for k, v in job.items() if k != "digest"} + if hashlib.sha256(canonical(unsigned).encode()).hexdigest() != job["digest"]: + raise ValueError("Task record changed; refusing execution.") + return job + + def prepare(self, root, tool, arguments): + self.require_managed(root) + if not isinstance(arguments, dict) or "operation_id" in arguments: + raise ValueError("Arguments must be an object; task IDs are managed internally.") + tools = self.tools() + if tool not in tools: + raise ValueError("Unknown business tool.") + fn = tools[tool] + inspect.signature(fn).bind(**arguments) + if "operation_id" in inspect.signature(fn).parameters: + if os.environ.get(self.prefix + "_MCP_ALLOW_WRITE") != "1": + raise ValueError("This workspace is read-only.") + allowed = os.environ.get(self.prefix + "_MCP_WRITE_TOOLS") + if allowed is not None and tool not in allowed.split(","): + raise ValueError("This tool is not permitted in the selected mode.") + job = dict(schema=1, id=uuid.uuid4().hex, tool=tool, arguments=arguments, created=time.time()) + job["digest"] = hashlib.sha256(canonical(job).encode()).hexdigest() + with self.runtime.workspace_lock(Path(root)): + directory = self.job_path(root, job["id"]).parent + checked_path(directory) + directory.mkdir(mode=0o700, exist_ok=True) + write_json(self.job_path(root, job["id"]), job) + return {"task_id": job["id"], "state": "prepared", "tool": tool, + "message": "Task prepared. Retain this ID before execution and reuse it for delivery retries."} + + def execute(self, root, job_id): + self.require_managed(root) + with self.runtime.workspace_lock(Path(root)): + job = self.load_job(root, job_id) + tools = self.tools() + if job["tool"] not in tools: + raise ValueError("Tool no longer exists; task cannot be migrated silently.") + fn = tools[job["tool"]] + arguments = dict(job["arguments"]) + if "operation_id" in inspect.signature(fn).parameters: + arguments["operation_id"] = job_id + try: + result = fn(**arguments) + except (ValueError, OSError) as exc: + text = str(exc) + state = "needs_attention" if "RECONCILE" in text or "CONFLICT" in text else "blocked" + return {"task_id": job_id, "state": state, "error": text, + "message": "Stopped. Inspect the workspace with doctor or recover."} + if isinstance(result, dict) and result.get("runtime_status"): + return {"task_id": job_id, "state": "needs_attention", "result": result, + "message": "Inspection required. Do not bypass an uncertain operation with a new task ID."} + return {"task_id": job_id, "state": "returned", "result": result, + "message": "Tool response delivered. Execution completion is not verification success."} + + def receipt(self, root, job_id): + path = Path(root) / self.control / (hashlib.sha256(job_id.encode()).hexdigest() + ".json") + if (path.parent / "recovery" / path.name).exists(): + return {"state": "retired", "note": "Operator reconciled; this ID cannot execute again."} + if not path.exists(): + return {"state": "not_recorded", "note": "No receipt is not proof of no effects."} + row = self.runtime._read_receipt(path) if self.product == "mirror-stack" else self.runtime.read_json(path) + state = row.get("status", row.get("state")) + return {"state": state, "tool": row.get("tool", (row.get("request") or {}).get("tool"))} + + def tasks(self, root): + directory = Path(root) / self.control / "tasks" + checked_path(directory) + out = [] + if directory.exists(): + for path in sorted(directory.glob("*.json"), key=lambda p: p.name): + if len(out) >= 1000: + break + job = self.load_job(root, path.stem) + out.append({"task_id": job["id"], "tool": job["tool"], "created": job["created"], + **self.receipt(root, job["id"])}) + return {"tasks": out, "limit": 1000, "message": "Status inspection does not execute tasks or clear pending state."} + + def active(self, root): + path = Path(root) / self.control / "active.json" + if not path.exists(): + return None + reader = self.runtime._read_receipt if self.product == "mirror-stack" else self.runtime.read_json + marker = reader(path) + name = marker.get("receipt") + if not isinstance(name, str) or not re.fullmatch(r"[0-9a-f]{64}\.json", name): + raise ValueError("Invalid active marker; restore metadata from evidence.") + receipt = path.parent / name + if not receipt.exists(): + return {"receipt": name, "state": "missing", "needs_attention": True} + row = reader(receipt) + state = row.get("status", row.get("state")) + return {"receipt": name, "state": state, "needs_attention": state not in ("done", "complete")} + + def doctor(self, root): + checks = [] + try: + config = self.load(root) + self.environment(config) + checks.append({"name": "workspace/config", "ok": True}) + except (ValueError, OSError) as exc: + return {"ok": False, "checks": [{"name": "workspace/config", "ok": False, "error": str(exc)}]} + try: + active = self.active(root) + checks.append({"name": "pending operation", "ok": not active or not active["needs_attention"], + "detail": active}) + except (ValueError, OSError) as exc: + checks.append({"name": "pending operation", "ok": False, "error": str(exc)}) + if self.product == "yeoul": + try: + from .server import bash_command + bash = bash_command() + found = Path(bash).is_file() if Path(bash).is_absolute() else shutil.which(bash) + checks.append({"name": "Bash", "ok": bool(found)}) + except OSError as exc: + checks.append({"name": "Bash", "ok": False, "error": str(exc)}) + for value in config["external_ledgers"]: + try: + checked_path(value, existing=True) + checks.append({"name": "linked ledger", "path": value, "ok": True}) + except (ValueError, OSError) as exc: + checks.append({"name": "linked ledger", "path": value, "ok": False, "error": str(exc)}) + return {"ok": all(row["ok"] for row in checks), "checks": checks, + "mode": config["mode"], "scope": "configuration and local prerequisites; not business verification", + "message": "Readiness check only; not certification of ledger truth or business success."} + + def link(self, root, ledger, remove=False): + config = self.load(root) + revision = config["revision"] + path = checked_path(ledger, existing=not remove) + if not remove and not path.is_file(): + raise ValueError("Select one ledger file, not a directory.") + if remove: + config["external_ledgers"] = [p for p in config["external_ledgers"] if p != str(path)] + elif str(path) not in config["external_ledgers"]: + if self.product == "mirror-stack": + from .integrity import read_verified + _, error = read_verified(path) + if error: + raise ValueError(error) + else: + from .server import BIN + spec = importlib.util.spec_from_file_location("_yeoul_prereg_check", BIN / "prereg_check.py") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + module.read_verified(path) + config["external_ledgers"].append(str(path)) + config["revision"] = uuid.uuid4().hex + self.save_config(Path(root), config, expected_revision=revision) + return {"ledger": str(path), "access": "revoked" if remove else "read-only", + "message": "Reconnect to apply the change. The source ledger was not modified."} + + def recover(self, root, acknowledge=False, note="", children_stopped=False): + self.load(root) + if not acknowledge: + return {"active": self.active(root), "tasks": self.tasks(root), + "message": "Preserve business data and receipts. Inspect changed files and surviving children. No automatic retry."} + if not note.strip() or not children_stopped: + raise ValueError("A reconciliation note and --children-stopped are required.") + with self.runtime.workspace_lock(Path(root)): + active = self.active(root) + if not active or not active["needs_attention"]: + return {"changed": False, "message": "No interrupted active operation to reconcile."} + directory = Path(root) / self.control / "recovery" + checked_path(directory) + directory.mkdir(mode=0o700, exist_ok=True) + record = {"schema": 1, "active": active, "note": note.strip(), + "children_stopped_attested": True, "time": time.time(), + "scope": "operator attestation, not automatic verification"} + # A deterministic tombstone also blocks replay when the original receipt + # was never created. Both low-level and product entrypoints check it. + write_json(directory / active["receipt"], record) + # Remove ONLY the pointer, under its own workspace lock, after durable audit. + # The old pending receipt remains non-retriable. + (Path(root) / self.control / "active.json").unlink() + sync_dir(Path(root) / self.control) + return {"changed": True, "message": "Reconciliation recorded. The interrupted task is retired; only new work may proceed."} + + def approve(self, root, todo): + if self.product != "yeoul": + raise ValueError("Verification approval is a Yeoul operator function.") + config = self.load(root) + revision = config["revision"] + if config["mode"] != "develop": + raise ValueError("Select develop mode first.") + path = checked_path(todo, existing=True) + if not path.is_relative_to(Path(root)): + raise ValueError("TODO must be inside this workspace.") + from .server import BIN + # Import the same verifier without executing any verification command. + spec = importlib.util.spec_from_file_location("_yeoul_verify_core", BIN / "verify_core.py") + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + content = path.read_text(encoding="utf-8") + if not module.eligible(content): + raise ValueError("TODO has missing/invalid verification criteria.") + directory = Path(root) / ".yeoul-approved" + checked_path(directory) + directory.mkdir(mode=0o700, exist_ok=True) + baseline = directory / (uuid.uuid4().hex + ".json") + write_json(baseline, {"version": 1, "todo": str(path), "text": module.canonical(content)}) + config.update(verification={"path": str(baseline), + "sha256": hashlib.sha256(baseline.read_bytes()).hexdigest()}, + revision=uuid.uuid4().hex) + self.save_config(Path(root), config, expected_revision=revision) + return {"approved": str(baseline), "message": "Commands approved. OS isolation is still required where needed. Reconnect to apply."} + + def cli(self, argv=None): + parser = argparse.ArgumentParser(prog=self.product, description="Standalone workspace setup, execution and diagnosis") + parser.add_argument("--workspace", default=os.getcwd()) + sub = parser.add_subparsers(dest="command") + def command(name): + p = sub.add_parser(name) + p.add_argument("--workspace", default=argparse.SUPPRESS) + return p + for name in ("setup", "configure"): + p = command(name) + p.add_argument("folder", nargs="?") + p.add_argument("--mode", choices=list(self.modes), default=None) + p.add_argument("--yes", action="store_true") + for name in ("doctor", "connect", "tasks", "serve"): + command(name) + p = command("run") + p.add_argument("tool") + p.add_argument("--arguments", default="{}") + p.add_argument("--yes", action="store_true") + p = command("retry") + p.add_argument("task_id") + p.add_argument("--yes", action="store_true") + for name in ("link", "unlink"): + p = command(name) + p.add_argument("ledger") + p.add_argument("--yes", action="store_true") + p = command("recover") + p.add_argument("--acknowledge", action="store_true") + p.add_argument("--children-stopped", action="store_true") + p.add_argument("--note", default="") + p.add_argument("--yes", action="store_true") + p = command("approve") + p.add_argument("todo") + p.add_argument("--yes", action="store_true") + for name, (tool, parameter, default) in self.defaults.items(): + p = command(name) + p.add_argument(parameter, nargs="?", default=default) + p.add_argument("--yes", action="store_true") + args = parser.parse_args(argv) + root = args.workspace + cmd = args.command + def confirm(message): + if getattr(args, "yes", False): + return + if not sys.stdin.isatty(): + raise ValueError("Interactive approval required; review first, then use --yes.") + if input(message + " [y/N] ").strip().lower() not in ("y", "yes"): + raise ValueError("Cancelled. No changes made.") + try: + if cmd is None: + parser.print_help() + print("\nStart here: " + self.product + " setup") + return 0 + if cmd in ("setup", "configure"): + folder = args.folder or root + if not args.folder and sys.stdin.isatty(): + folder = input("Workspace folder [%s]: " % folder).strip() or folder + mode = args.mode or "observe" + if not args.mode and sys.stdin.isatty(): + mode = input("Mode %s [%s]: " % ("/".join(self.modes), mode)).strip() or mode + confirm("Configure folder %s / mode %s for %s." % (folder, mode, self.product)) + result = self.setup(folder, mode) if cmd == "setup" else self.configure(self.root(folder), mode) + elif cmd == "doctor": + result = self.doctor(root) + elif cmd == "connect": + result = self.connection(root) + elif cmd == "serve": + with self.activated(root): + importlib.import_module(self.package + ".server").main() + return 0 + elif cmd == "tasks": + result = self.tasks(self.root(root)) + elif cmd == "recover": + if args.acknowledge: + confirm("Clear only the pending marker after operator inspection of files and surviving processes.") + result = self.recover(self.root(root), args.acknowledge, args.note, args.children_stopped) + elif cmd in ("link", "unlink"): + confirm("Change external ledger read permission: " + cmd + " " + args.ledger) + result = self.link(self.root(root), args.ledger, remove=cmd == "unlink") + elif cmd == "approve": + path = checked_path(args.todo, existing=True) + print(path.read_text(encoding="utf-8"), file=sys.stderr) + confirm("These commands run with the server account's authority. Approve these criteria and commands?") + result = self.approve(self.root(root), str(path)) + elif cmd in ("run", "retry") or cmd in self.defaults: + with self.activated(root): + if cmd == "retry": + confirm("Resume this task. Completed responses replay; interrupted tasks do not execute again.") + result = self.execute(self.root(root), args.task_id) + else: + if cmd in self.defaults: + tool, parameter, _ = self.defaults[cmd] + values = {parameter: getattr(args, parameter)} + if self.product == "mirror-stack" and cmd == "record": + values = dict(ledger_path="actions.jsonl", agent="user", action="note", + payload={"text": values["text"]}) + else: + tool, values = args.tool, json.loads(args.arguments) + if tool in self.tools() and "operation_id" in inspect.signature(self.tools()[tool]).parameters: + confirm("Execute task: %s %s" % (tool, canonical(values))) + task = self.prepare(self.root(root), tool, values) + print("task_id=" + task["task_id"], file=sys.stderr, flush=True) + result = self.execute(self.root(root), task["task_id"]) + else: + raise ValueError("Unknown command.") + print(json.dumps(result, ensure_ascii=False, indent=2)) + if isinstance(result, dict): + if result.get("ok") is False or result.get("state") in ("blocked", "needs_attention"): + return 2 + business = result.get("result") + if isinstance(business, dict) and business.get("exit_code", 0) != 0: + return 1 + if isinstance(business, dict) and (business.get("ok") is False or business.get("decision") == "BLOCK"): + return 1 + if isinstance(business, list) and any("FAIL" in str(item) for item in business): + return 1 + return 0 + except (ValueError, OSError, KeyError, TypeError) as exc: + print("Stopped: " + str(exc), file=sys.stderr) + return 2 diff --git a/pyproject.toml b/pyproject.toml index 8e32d5b..9b26887 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta" [project] name = "mirror-stack-mcp" -version = "0.2.14" +version = "0.3.0" description = "Unified MCP server for the Mirror Stack — claims, actions, provenance + verify-all in one server" readme = "README.md" requires-python = ">=3.10" @@ -35,6 +35,7 @@ bitcoin = ["opentimestamps-client>=0.7"] test = ["pytest>=7"] [project.scripts] +mirror-stack = "mirror_stack_mcp.product:main" mirror-stack-mcp = "mirror_stack_mcp.server:main" mirror-stack-verify = "mirror_stack_mcp.verify:main" mirror-stack-gate = "mirror_stack_mcp.gate:main" diff --git a/tests/smoke_installed.py b/tests/smoke_installed.py index fc20e33..5d72033 100644 --- a/tests/smoke_installed.py +++ b/tests/smoke_installed.py @@ -2,13 +2,13 @@ This is the exact check that would have caught our local install going silently stale: 'works in the repo, broken/old once installed'. It asserts the unified server imports -in a clean environment, exposes all 19 tools, and that the three re-exposed mirror +in a clean environment, exposes all 22 tools, and that the three re-exposed mirror packages are importable (a missing/renamed mirror dep fails here, loudly). """ import mirror_stack_mcp.server as s tools = {t.name for t in s.mcp._tool_manager.list_tools()} -assert len(tools) == 19, f"expected 19 tools, got {len(tools)}: {sorted(tools)}" +assert len(tools) == 22, f"expected 22 tools, got {len(tools)}: {sorted(tools)}" assert callable(s.main), "entry point mirror_stack_mcp.server:main missing" import measure_mirror # noqa: F401 re-exposed claim probes diff --git a/tests/test_product.py b/tests/test_product.py new file mode 100644 index 0000000..a9ccdba --- /dev/null +++ b/tests/test_product.py @@ -0,0 +1,217 @@ +"""Standalone product contract. All data and approvals stay in temporary folders.""" +import hashlib +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile +import unittest +from unittest.mock import patch + +SOURCE = str(Path(__file__).resolve().parents[1]) +if not os.environ.get("PRODUCT_TEST_INSTALLED"): + sys.path.insert(0, SOURCE) +from mirror_stack_mcp.product import workspace +from mirror_stack_mcp.workspace import write_json +from mirror_stack_mcp import server +if os.environ.get("PRODUCT_TEST_INSTALLED"): + SOURCE = str(Path(server.__file__).resolve().parents[1]) +MODE, TOOL, EFFECT, COMPLETE, PACKAGE = "record", "am_record", "actions.jsonl", "done", "mirror_stack_mcp" +ARGUMENTS = {"ledger_path":"actions.jsonl", "agent":"user", "action":"note", "payload":{"text":"hello"}} + +class ProductContract(unittest.TestCase): + def setUp(self): + self.tmp = tempfile.TemporaryDirectory(prefix="standalone product ") + self.addCleanup(self.tmp.cleanup) + self.base = Path(self.tmp.name).resolve() + self.root = self.base / "workspace" + self.env = patch.dict(os.environ, {k:v for k,v in os.environ.items() + if not k.startswith(workspace.prefix + "_")}, clear=True) + self.env.start() + self.addCleanup(self.env.stop) + workspace.setup(self.root, MODE) + + def test_setup_preserves_data_and_refuses_reinitialize(self): + data = self.root / "business.txt" + data.write_text("preserve") + with self.assertRaises(ValueError): + workspace.setup(self.root, MODE) + self.assertEqual(data.read_text(), "preserve") + self.assertTrue(workspace.doctor(self.root)["ok"]) + connection = workspace.connection(self.root) + config = connection["mcpServers"][workspace.product] + self.assertEqual(config["args"][-1], str(self.root)) + self.assertNotIn("lane", json.dumps(config)) + + def test_prepare_then_execute_shared_with_mcp(self): + with workspace.activated(self.root): + task = server.workspace_prepare(TOOL, ARGUMENTS) + self.assertEqual(task["state"], "prepared") + self.assertFalse((self.root / EFFECT).exists()) + first = workspace.execute(self.root, task["task_id"]) + self.assertEqual(first["state"], "returned", first) + self.assertTrue((self.root / EFFECT).exists(), first) + snapshot = self.snapshot() + second = server.workspace_execute(task["task_id"]) + self.assertEqual(first, second) + self.assertEqual(snapshot, self.snapshot()) + direct = getattr(server, TOOL)(**ARGUMENTS, operation_id=task["task_id"]) + self.assertEqual(direct, first["result"]) + self.assertEqual(snapshot, self.snapshot()) + self.assertEqual(server.workspace_tasks()["tasks"][0]["state"], COMPLETE) + + def snapshot(self): + return {str(p.relative_to(self.root)): p.read_bytes() + for p in self.root.rglob("*") if p.is_file() + and workspace.control not in p.parts} + + def test_readonly_does_not_elevate_from_writable_profile(self): + with workspace.activated(self.root): + os.environ[workspace.prefix + "_MCP_ALLOW_WRITE"] = "0" + with self.assertRaises(ValueError): + server.workspace_prepare(TOOL, ARGUMENTS) + self.assertFalse((self.root / EFFECT).exists()) + + def test_changed_task_refused(self): + with workspace.activated(self.root): + task = workspace.prepare(self.root, TOOL, ARGUMENTS) + path = workspace.job_path(self.root, task["task_id"]) + job = json.loads(path.read_text()) + job["arguments"] = {} + write_json(path, job) + with self.assertRaises(ValueError): + workspace.execute(self.root, task["task_id"]) + self.assertFalse((self.root / EFFECT).exists()) + + def test_operator_metadata_cannot_be_business_target(self): + with workspace.activated(self.root): + if workspace.product == "mirror-stack": + with self.assertRaises(ValueError): + server.am_record(workspace.config_name, "u", "note", operation_id="bad") + else: + result = server.verify_gate(workspace.config_name, operation_id="bad") + self.assertNotEqual(result["exit_code"], 0) + self.assertEqual(workspace.load(self.root)["mode"], MODE) + + def test_missing_receipt_recovery_retires_old_id(self): + with workspace.activated(self.root): + task = workspace.prepare(self.root, TOOL, ARGUMENTS) + receipt_name = hashlib.sha256(task["task_id"].encode()).hexdigest() + ".json" + marker = {"receipt": receipt_name} + if workspace.product == "mirror-stack": + marker["schema"] = 1 + write_json(self.root / workspace.control / "active.json", marker) + self.assertFalse(workspace.doctor(self.root)["ok"]) + self.assertTrue(workspace.recover(self.root)["active"]["needs_attention"]) + with self.assertRaises(ValueError): + workspace.recover(self.root, True, "", True) + self.assertFalse((self.root / EFFECT).exists()) + got = workspace.execute(self.root, task["task_id"]) + self.assertEqual(got["state"], "needs_attention", got) + result = workspace.recover(self.root, True, "Inspected files; no surviving child.", True) + self.assertTrue(result["changed"]) + got = workspace.execute(self.root, task["task_id"]) + self.assertEqual(got["state"], "needs_attention", got) + self.assertFalse((self.root / EFFECT).exists()) + self.assertEqual(workspace.tasks(self.root)["tasks"][0]["state"], "retired") + new_task = workspace.prepare(self.root, TOOL, ARGUMENTS) + self.assertEqual(workspace.execute(self.root, new_task["task_id"])["state"], "returned") + + def test_mode_change_backs_up_and_revokes(self): + old = workspace.load(self.root) + workspace.configure(self.root, "observe") + self.assertEqual(workspace.load(self.root)["mode"], "observe") + history = list((self.root / workspace.control / "config-history").glob("*.json")) + self.assertTrue(any(json.loads(p.read_text()) == old for p in history)) + with workspace.activated(self.root), self.assertRaises(ValueError): + workspace.prepare(self.root, TOOL, ARGUMENTS) + with self.assertRaises(ValueError): + workspace.save_config(self.root, old, expected_revision=old["revision"]) + + def test_external_grant_exact_file_readonly_and_revocable(self): + ledger = self.base / "outside.jsonl" + body = {"prev_seal": "genesis", "claim_id": "c", "metric": "m", "kill_condition": "n < 20"} + body["seal"] = hashlib.sha256(json.dumps(body, sort_keys=True, ensure_ascii=False).encode()).hexdigest() + ledger.write_text(json.dumps(body) + "\n") + original = ledger.read_bytes() + workspace.link(self.root, ledger) + with workspace.activated(self.root): + if workspace.product == "mirror-stack": + from mirror_stack_mcp.runtime import scoped_path + self.assertEqual(scoped_path(ledger, external_read=True), str(ledger)) + server.mm_anchor(str(ledger)) + with self.assertRaises(ValueError): + server.am_record(str(ledger), "u", "note", operation_id="external-write") + else: + from yeoul_mcp.runtime import Policy + policy = Policy({}) + self.assertEqual(policy.path(str(ledger), external_read=True), ledger) + with self.assertRaises(ValueError): + policy.path(str(ledger)) + arc = self.root / "arc" + arc.mkdir() + (arc / ".prereg").write_text("c\n" + str(ledger) + "\nseal\n") + policy.validate("status", {}) + self.assertEqual(ledger.read_bytes(), original) + workspace.link(self.root, ledger, remove=True) + with workspace.activated(self.root): + if workspace.product == "mirror-stack": + with self.assertRaises(ValueError): + scoped_path(ledger, external_read=True) + else: + with self.assertRaises(ValueError): + Policy({}).path(str(ledger), external_read=True) + self.assertEqual(ledger.read_bytes(), original) + + def test_cli_outside_launch_directory(self): + env = dict(os.environ, PYTHONPATH=SOURCE, PYTHONDONTWRITEBYTECODE="1") + def cli(*args): + return subprocess.run([sys.executable, "-B", "-m", PACKAGE + ".product", *args, + "--workspace", str(self.root)], cwd=self.base, env=env, + text=True, capture_output=True, timeout=30) + self.assertEqual(cli("doctor").returncode, 0) + result = cli("run", TOOL, "--arguments", json.dumps(ARGUMENTS), "--yes") + self.assertEqual(result.returncode, 0, result.stderr + result.stdout) + task_id = json.loads(result.stdout)["task_id"] + self.assertIn("task_id=" + task_id, result.stderr) + repeated = cli("retry", task_id, "--yes") + self.assertEqual(repeated.returncode, 0, repeated.stderr) + self.assertEqual(json.loads(result.stdout), json.loads(repeated.stdout)) + + + def test_product_stdio_reconnect_replays_task(self): + import asyncio + from mcp import ClientSession, StdioServerParameters + from mcp.client.stdio import stdio_client + from mirror_stack_mcp import __version__ + async def run(): + settings = StdioServerParameters( + command=sys.executable, + args=["-B", "-m", PACKAGE + ".product", "serve", "--workspace", str(self.root)], + env=dict(os.environ, PYTHONPATH=SOURCE, PYTHONDONTWRITEBYTECODE="1")) + async def connect(task_id=None): + async with stdio_client(settings) as (reader, writer): + async with ClientSession(reader, writer) as client: + hello = await client.initialize() + self.assertEqual(hello.serverInfo.version, __version__) + names = {tool.name for tool in (await client.list_tools()).tools} + self.assertTrue({"workspace_prepare", "workspace_execute", "workspace_tasks"} <= names) + self.assertNotIn("workspace_recover", names) + if task_id is None: + reply = await client.call_tool("workspace_prepare", {"tool": TOOL, "arguments": ARGUMENTS}) + self.assertFalse(reply.isError, reply) + task_id = json.loads(reply.content[0].text)["task_id"] + result = await client.call_tool("workspace_execute", {"task_id": task_id}) + self.assertFalse(result.isError, result) + return task_id, json.loads(result.content[0].text) + task_id, first = await connect() + snapshot = self.snapshot() + _, second = await connect(task_id) + self.assertEqual(first, second) + self.assertEqual(first["state"], "returned") + self.assertEqual(snapshot, self.snapshot()) + asyncio.run(asyncio.wait_for(run(), timeout=60)) + +if __name__ == "__main__": + unittest.main() diff --git a/tests/test_runtime_contract.py b/tests/test_runtime_contract.py new file mode 100644 index 0000000..e41f007 --- /dev/null +++ b/tests/test_runtime_contract.py @@ -0,0 +1,271 @@ +"""Managed stdio boundary: actual writers, multi-process retries and negative controls.""" +import json +import os +from pathlib import Path +import subprocess +import sys + +import pytest + +from mirror_stack_mcp import server as s +from mirror_stack_mcp.integrity import read_verified +from mirror_stack_mcp.runtime import (RuntimeRefusal, guarded, managed_root, + operation_status, receipt_path, workspace_lock) + + +@pytest.fixture +def managed(tmp_path, monkeypatch): + monkeypatch.setenv("MIRROR_MCP_ROOT", str(tmp_path.resolve())) + monkeypatch.setenv("MIRROR_MCP_ALLOW_WRITE", "1") + for key in ("MIRROR_MCP_WRITE_TOOLS", "MIRROR_MCP_ALLOW_NETWORK", "MIRROR_MCP_ALLOW_EXEC"): + monkeypatch.delenv(key, raising=False) + return tmp_path.resolve() + + +def record(operation_id="op-1", **kw): + return s.am_record("actions.jsonl", "worker", "test", operation_id=operation_id, **kw) + + +def test_restart_replay_and_changed_arguments_conflict(managed): + first = record(payload={"n": 1}) + original = (managed / "actions.jsonl").read_bytes() + code = "from mirror_stack_mcp.server import am_record; import json; print(json.dumps(am_record('actions.jsonl','worker','test',payload={'n':1},operation_id='op-1')))" + result = subprocess.run([sys.executable, "-c", code], capture_output=True, text=True, timeout=20) + assert result.returncode == 0, result.stderr + assert json.loads(result.stdout) == first + assert (managed / "actions.jsonl").read_bytes() == original + with pytest.raises(RuntimeRefusal, match="CONFLICT"): + record(payload={"n": 2}) + assert (managed / "actions.jsonl").read_bytes() == original + + +def test_managed_requires_id_and_write_permission(managed, monkeypatch): + with pytest.raises(RuntimeRefusal, match="operation_id"): + record(None) + monkeypatch.setenv("MIRROR_MCP_ALLOW_WRITE", "0") + with pytest.raises(RuntimeRefusal, match="read-only"): + record() + assert not (managed / "actions.jsonl").exists() + + +def test_policy_rechecked_before_replay(managed, monkeypatch): + record() + monkeypatch.setenv("MIRROR_MCP_WRITE_TOOLS", "mm_preregister") + with pytest.raises(RuntimeRefusal, match="not in"): + record() + + +@pytest.mark.parametrize("path", ["../escape.jsonl", ".mirror-mcp-runtime/workspace.lock"]) +def test_scope_escape_and_control_targets_refused(managed, path): + with pytest.raises(RuntimeRefusal): + s.am_record(path, "a", "x", operation_id="escape") + + +def test_absolute_outside_and_read_escape_refused(managed): + with pytest.raises(RuntimeRefusal, match="outside"): + s.mm_anchor(str(managed.parent / "outside.jsonl")) + + +def test_symlink_and_hardlink_refused(managed, tmp_path): + target = managed / "real" + target.write_text("original") + link = managed / "link" + try: + link.symlink_to(target) + except OSError: + pytest.skip("symlink creation unavailable") + with pytest.raises(RuntimeRefusal, match="symlink"): + s.pm_verify("link", "pm.jsonl", operation_id="link") + hard = managed / "hard" + os.link(target, hard) + with pytest.raises(RuntimeRefusal, match="hard-linked"): + s.pm_verify("hard", "pm.jsonl", operation_id="hard") + + +@pytest.mark.parametrize("contents", ["", "broken\n", '{"seal":"a","seal":"b"}\n']) +def test_invalid_existing_ledger_not_extended(managed, contents): + path = managed / "actions.jsonl" + path.write_text(contents) + with pytest.raises(RuntimeRefusal, match="INVALID_LEDGER"): + record() + assert path.read_text() == contents + assert not receipt_path(managed, "op-1").exists() + + +def test_valid_unterminated_tail_refused(managed): + record() + path = managed / "actions.jsonl" + raw = path.read_bytes().rstrip(b"\n") + path.write_bytes(raw) + with pytest.raises(RuntimeRefusal, match="unterminated"): + record("op-2") + assert path.read_bytes() == raw + + +def test_nonfinite_input_refused_before_write(managed): + with pytest.raises(ValueError): + record(payload={"n": float("nan")}) + assert not (managed / "actions.jsonl").exists() + + +def test_interrupted_request_never_reexecutes(managed): + code = ''' +import os +from pathlib import Path +from mirror_stack_mcp.runtime import guarded +@guarded(write=True) +def crash(operation_id=None): + (Path(os.environ['MIRROR_MCP_ROOT']) / 'effect').write_text('once') + os._exit(17) +crash(operation_id='crash-1') +''' + result = subprocess.run([sys.executable, "-c", code], timeout=20) + assert result.returncode == 17 + assert (managed / "effect").read_text() == "once" + + @guarded(write=True) + def crash(operation_id=None): + pytest.fail("interrupted operation was re-executed") + + with pytest.raises(RuntimeRefusal, match="RECONCILE_REQUIRED"): + crash(operation_id="crash-1") + with pytest.raises(RuntimeRefusal, match="RECONCILE_REQUIRED"): + record("different-id") + assert not (managed / "actions.jsonl").exists() + # Kernel lock released on process death, receipt retained. + with workspace_lock(managed, timeout=0.1): + assert json.loads(receipt_path(managed, "crash-1").read_text())["status"] == "prepared" + + +def test_parallel_processes_same_and_distinct_ids(managed): + code = "from mirror_stack_mcp.server import am_record; import sys; am_record('actions.jsonl','a','x',operation_id=sys.argv[1])" + processes = [subprocess.Popen([sys.executable, "-c", code, op], stdout=subprocess.PIPE, + stderr=subprocess.PIPE, text=True) + for op in ("same", "same", "one", "two")] + try: + for process in processes: + _, stderr = process.communicate(timeout=30) + assert process.returncode == 0, stderr + finally: + for process in processes: + if process.poll() is None: + process.kill() + process.wait() + entries, error = read_verified(managed / "actions.jsonl") + assert error is None + assert len(entries) == 3 + + +def test_lock_wait_is_bounded(managed): + with workspace_lock(managed): + with pytest.raises(RuntimeRefusal, match="BUSY"): + with workspace_lock(managed, timeout=0.05): + pytest.fail("second lock acquired") + + +def test_receipt_corruption_refuses_replay(managed): + record() + receipt_path(managed, "op-1").write_text('{"schema":1,"schema":1}') + with pytest.raises(RuntimeRefusal, match="RECONCILE_REQUIRED"): + record() + + +def test_network_tools_require_separate_capabilities(managed): + with pytest.raises(RuntimeRefusal, match="network"): + s.mm_anchor_verify("proof.ots") + with pytest.raises(RuntimeRefusal, match="network"): + s.mm_anchor_bitcoin(["actions.jsonl"], "anchors", operation_id="net") + + +def test_all_tool_schemas_preserved_and_writers_expose_id(managed): + tools = {tool.name: tool for tool in s.mcp._tool_manager.list_tools()} + assert len(tools) == 22 + for name in ("am_record", "am_witness", "pm_verify", "mm_preregister", + "mm_retract", "mm_anchor_bitcoin", "mm_anchor_upgrade"): + assert "operation_id" in tools[name].parameters["properties"] + assert "Managed mode requires operation_id" in tools[name].description + + +def test_provenance_and_prereg_writes_remain_verified(managed): + (managed / "artifact").write_text("ordinary c2pa marker") + got = s.pm_verify("artifact", operation_id="pm") + assert got["verdict"] == "PROVENANCE-UNVERIFIED" + assert read_verified(managed / "pm_ledger.jsonl")[1] is None + s.mm_preregister("claims.jsonl", "c", "acc", kill_condition="below chance", operation_id="pre") + assert read_verified(managed / "claims.jsonl")[1] is None + + +def test_embedded_anchor_path_cannot_escape(managed): + record() + anchors = managed / "anchors" + anchors.mkdir() + (anchors / "anchor_bad.json").write_text(json.dumps({"ledger_path": str(managed.parent / "private")})) + with pytest.raises(RuntimeRefusal, match="outside"): + s.stack_verify_all("actions.jsonl", "anchors") + + +def test_bad_root_never_falls_back_to_trusted(monkeypatch): + monkeypatch.setenv("MIRROR_MCP_ROOT", "") + with pytest.raises(RuntimeRefusal): + managed_root() + + +def test_id_without_managed_mode_is_not_silently_ignored(monkeypatch): + monkeypatch.delenv("MIRROR_MCP_ROOT", raising=False) + with pytest.raises(RuntimeRefusal, match="requires managed"): + record() + + +def test_status_does_not_create_runtime_state_and_reports_done(managed): + assert operation_status("missing") == {"operation_id": "missing", "status": "unknown"} + assert not (managed / ".mirror-mcp-runtime").exists() + record() + before = receipt_path(managed, "op-1").read_bytes() + assert operation_status("op-1")["status"] == "done" + assert receipt_path(managed, "op-1").read_bytes() == before + + +def test_lock_control_symlink_refused(managed): + control = managed / ".mirror-mcp-runtime" + control.mkdir(mode=0o700) + outside = managed / "untouched" + outside.write_text("unchanged") + try: + (control / "workspace.lock").symlink_to(outside) + except OSError: + pytest.skip("symlink creation unavailable") + with pytest.raises(RuntimeRefusal, match="symlink"): + record() + assert outside.read_text() == "unchanged" + + +def test_anchor_manifests_never_overwrite_on_timestamp_collision(managed): + from mirror_stack_mcp.ots_anchor import build_manifest + record() + args = ([str(managed / "actions.jsonl")], str(managed / "anchors")) + first, digest = build_manifest(*args) + content = Path(first).read_bytes() + second, _ = build_manifest(*args) + assert first != second + assert Path(first).read_bytes() == content + assert Path(second).exists() + + +def test_crash_between_marker_and_receipt_fails_closed(managed, monkeypatch): + from mirror_stack_mcp import runtime + original = runtime._store + + def fault(path, row): + if row.get("status") == "prepared": + raise OSError("injected persistence failure") + return original(path, row) + + with monkeypatch.context() as scoped: + scoped.setattr(runtime, "_store", fault) + with pytest.raises(OSError, match="injected"): + record("interrupted") + assert not (managed / "actions.jsonl").exists() + with pytest.raises(RuntimeRefusal, match="RECONCILE_REQUIRED"): + record("new-id") + with pytest.raises(RuntimeRefusal, match="RECONCILE_REQUIRED"): + record("interrupted") diff --git a/tests/test_server.py b/tests/test_server.py index 46503b0..10b0d09 100644 --- a/tests/test_server.py +++ b/tests/test_server.py @@ -1,7 +1,7 @@ """Unit + smoke tests for the unified Mirror Stack MCP server. Covers what a fresh install / reconnect must get right: - • all 19 tools register (drift detector — adding/removing a tool fails this); + • all 22 tools register (19 business + 3 workspace tools); • the loop-safe defaults (compact output, once-per-session reminders); • signal-preserving compaction — OK/INFO collapse, but a WARN/FAIL is NEVER dropped (a hidden negative is the cardinal sin this server exists to prevent); @@ -28,7 +28,7 @@ # 🔎 provenance-mirror "pm_verify", # 🪞🔎🪪 stack-level - "stack_verify_all", + "stack_verify_all", "workspace_prepare", "workspace_execute", "workspace_tasks", } @@ -49,10 +49,10 @@ def _write(path, entries): # ── registration / defaults ─────────────────────────────────────────────────── -def test_all_19_tools_registered(): +def test_all_22_tools_registered(): names = _registered() assert names == EXPECTED_TOOLS, f"tool drift: {names ^ EXPECTED_TOOLS}" - assert len(names) == 19 + assert len(names) == 22 def test_defaults_are_loop_safe(): diff --git a/tests/test_stdio_runtime.py b/tests/test_stdio_runtime.py new file mode 100644 index 0000000..524208f --- /dev/null +++ b/tests/test_stdio_runtime.py @@ -0,0 +1,43 @@ +"""Actual stdio tools/call, not direct Python invocation. All state is temporary.""" +import asyncio +import json +import os +import sys + +from mcp import ClientSession, StdioServerParameters +from mcp.client.stdio import stdio_client + + +def test_stdio_managed_write_replay_and_denial(tmp_path): + async def run(): + env = {**os.environ, "MIRROR_MCP_ROOT": str(tmp_path.resolve()), + "MIRROR_MCP_ALLOW_WRITE": "1", "MIRROR_MCP_WRITE_TOOLS": "am_record"} + server = StdioServerParameters(command=sys.executable, + args=["-m", "mirror_stack_mcp.server"], env=env) + async with stdio_client(server) as (reader, writer): + async with ClientSession(reader, writer) as client: + init = await client.initialize() + assert init.serverInfo.name == "mirror-stack" + tools = await client.list_tools() + assert len(tools.tools) == 22 + arguments = {"ledger_path": "actions.jsonl", "agent": "worker", "action": "test", + "operation_id": "wire-1"} + first = await client.call_tool("am_record", arguments) + assert not first.isError + second = await client.call_tool("am_record", arguments) + assert not second.isError + assert second.content == first.content + outside = await client.call_tool("am_record", {**arguments, "operation_id": "wire-2", + "ledger_path": "../escape.jsonl"}) + assert outside.isError + missing = await client.call_tool("am_record", {k: v for k, v in arguments.items() + if k != "operation_id"}) + assert missing.isError + denied = await client.call_tool("mm_retract", {"ledger_path": "actions.jsonl", + "claim_id": "c", "reason": "x", + "operation_id": "wire-3"}) + assert denied.isError + rows = [json.loads(line) for line in (tmp_path / "actions.jsonl").read_text().splitlines()] + assert len(rows) == 1 + + asyncio.run(asyncio.wait_for(run(), timeout=25)) From 1cb47b86beb39b393fa15514f6a0a531a2f808f5 Mon Sep 17 00:00:00 2001 From: Mother Seara Date: Thu, 10 Sep 2026 19:02:03 +0900 Subject: [PATCH 2/2] fix: preserve UTF-8 CLI output on Windows and verify cross-root witness --- mirror_stack_mcp/workspace.py | 6 ++++++ tests/test_product.py | 7 +++++-- tests/test_runtime_contract.py | 3 ++- 3 files changed, 13 insertions(+), 3 deletions(-) diff --git a/mirror_stack_mcp/workspace.py b/mirror_stack_mcp/workspace.py index d438583..96d0355 100644 --- a/mirror_stack_mcp/workspace.py +++ b/mirror_stack_mcp/workspace.py @@ -441,6 +441,12 @@ def approve(self, root, todo): return {"approved": str(baseline), "message": "Commands approved. OS isolation is still required where needed. Reconnect to apply."} def cli(self, argv=None): + # The CLI emits JSON containing Unicode tool output. Windows redirected + # streams otherwise default to a legacy code page and can fail AFTER a + # committed operation. Fix the transport, never replace verdict characters. + for stream in (sys.stdin, sys.stdout, sys.stderr): + if hasattr(stream, "reconfigure"): + stream.reconfigure(encoding="utf-8", errors="strict") parser = argparse.ArgumentParser(prog=self.product, description="Standalone workspace setup, execution and diagnosis") parser.add_argument("--workspace", default=os.getcwd()) sub = parser.add_subparsers(dest="command") diff --git a/tests/test_product.py b/tests/test_product.py index a9ccdba..3665dbe 100644 --- a/tests/test_product.py +++ b/tests/test_product.py @@ -141,6 +141,8 @@ def test_external_grant_exact_file_readonly_and_revocable(self): from mirror_stack_mcp.runtime import scoped_path self.assertEqual(scoped_path(ledger, external_read=True), str(ledger)) server.mm_anchor(str(ledger)) + witness = server.am_witness("witnesses.jsonl", str(ledger), "peer", operation_id="witness") + self.assertTrue((self.root / "witnesses.jsonl").is_file(), witness) with self.assertRaises(ValueError): server.am_record(str(ledger), "u", "note", operation_id="external-write") else: @@ -165,11 +167,12 @@ def test_external_grant_exact_file_readonly_and_revocable(self): self.assertEqual(ledger.read_bytes(), original) def test_cli_outside_launch_directory(self): - env = dict(os.environ, PYTHONPATH=SOURCE, PYTHONDONTWRITEBYTECODE="1") + env = dict(os.environ, PYTHONPATH=SOURCE, PYTHONDONTWRITEBYTECODE="1", + PYTHONIOENCODING="ascii") def cli(*args): return subprocess.run([sys.executable, "-B", "-m", PACKAGE + ".product", *args, "--workspace", str(self.root)], cwd=self.base, env=env, - text=True, capture_output=True, timeout=30) + text=True, encoding="utf-8", capture_output=True, timeout=30) self.assertEqual(cli("doctor").returncode, 0) result = cli("run", TOOL, "--arguments", json.dumps(ARGUMENTS), "--yes") self.assertEqual(result.returncode, 0, result.stderr + result.stdout) diff --git a/tests/test_runtime_contract.py b/tests/test_runtime_contract.py index e41f007..def276c 100644 --- a/tests/test_runtime_contract.py +++ b/tests/test_runtime_contract.py @@ -95,7 +95,8 @@ def test_invalid_existing_ledger_not_extended(managed, contents): def test_valid_unterminated_tail_refused(managed): record() path = managed / "actions.jsonl" - raw = path.read_bytes().rstrip(b"\n") + # Remove the entire native terminator, including Windows CRLF. + raw = path.read_bytes().rstrip(b"\r\n") path.write_bytes(raw) with pytest.raises(RuntimeRefusal, match="unterminated"): record("op-2")