diff --git a/.gitignore b/.gitignore index 3bdaa40..5a1c6d7 100644 --- a/.gitignore +++ b/.gitignore @@ -18,3 +18,5 @@ docs/* !docs/INTERFACES.md !docs/ZCODE_RUNTIME.md !docs/PHASE7_LIVE_PROGRESS.md +!docs/research/ +!docs/research/** diff --git a/NATIVE_FEEDBACK_RENDERER_RECOMMENDATION.md b/NATIVE_FEEDBACK_RENDERER_RECOMMENDATION.md new file mode 100644 index 0000000..a36b233 --- /dev/null +++ b/NATIVE_FEEDBACK_RENDERER_RECOMMENDATION.md @@ -0,0 +1,110 @@ +# Native Feedback Renderer Recommendation + +This recommendation uses the evidence in [ZCODE_APPSERVER_EVENT_CAPABILITY_PROBE.md](ZCODE_APPSERVER_EVENT_CAPABILITY_PROBE.md). It separates Bridge/runtime observations from Agent-reported result fields. + +## Safe to render now + +### Queued + +```text +▣ ZCode · TASK_ID +○ Queued +``` + +**Source:** Bridge task status. The runtime probe did not observe a queued app-server state. + +### Running + +```text +▣ ZCode · TASK_ID +→ Running +Model: GLM-5.3 · Reasoning: max +``` + +**Source:** Running is the Bridge task status. Model and reasoning values are included only when the runtime snapshot reports them. This probe observed the values above for one task. + +If a safe tool event is available, use a last-observed label: + +```text +Last observed: Tool request · Write +``` + +Do not render `Tool running` from the current probe: `tool.updated` occurred, but its state values and currentness semantics were not validated in the sanitized capture. + +### Interaction observed + +The probe saw `permission.requested` and `permission.resolved` events, but no human approval wait. A renderer may report the event only if its normalized event has a safe summary and correlation: + +```text +Permission event observed +``` + +Do not render “Waiting for permission” unless a validated pending interaction is present. User-input interaction was not observed. + +### Completed + +```text +▣ ZCode · TASK_ID +✓ Runtime turn completed + +Agent report: + +Changed: files +Tests: · +``` + +`Runtime turn completed` is based on `turn.completed` and its `resultType`; it does not prove that the requested task objective was satisfied. In this probe, the runtime returned success but the requested file was absent at host-side inspection. Show a Bridge task `COMPLETED` status only when Bridge has normalized a final result, and continue to label summary, changed files, and test command statuses as Agent-reported claims. Do not display a test-case pass count from command-level status. + +### Failed + +```text +▣ ZCode · TASK_ID · FAILED +Reason: +``` + +Use only after Bridge records a failed terminal status. This probe did not exercise failure handling, so it does not validate a runtime `turn.failed` mapping. + +### Cancelled + +```text +▣ ZCode · TASK_ID · CANCELLED +Result: Task cancellation confirmed by Bridge +``` + +Use only after Bridge confirms cancellation. The runtime cancellation path was not probed. + +### Waiting for master + +```text +▣ ZCode · TASK_ID · WAITING_FOR_MASTER +Agent report requires a master decision. +``` + +This is the Bridge terminal task status derived from the Agent report. It is not a runtime permission/user-input wait. + +## Keep unsupported + +Do not render these as facts until structured runtime evidence and state semantics are validated: + +- `✓ Analysis / → Implementation / ○ Tests / ○ Review` +- `Tests · 38/42`, `73%`, or other progress inferred from tool calls, tokens, elapsed time, or text +- `Tool running · shell` based solely on a tool call/update event +- `Waiting: ZCode requires user input` based only on an earlier request event or absence of a reply in a partial event page +- A generic `Retrying` label based solely on `streamRecovery.updated` +- Any raw model text delta, reasoning delta, tool input/arguments, request headers, response headers, provider telemetry, or protocol dump + +## Event aggregation + +The probe saw 261 events for one task, including 177 `model.streaming` events. Do not forward every event to Codex Native Text. The Bridge’s current `summary` event view only merges adjacent visible model-output chunks; it is not a semantic phase/progress aggregator. A renderer should emit lifecycle changes and verified interaction transitions, retain raw diagnostics outside the default transcript, and throttle repetitive updates. + +## Provenance labels + +| Displayed value | Source label / rule | +|---|---| +| Queued/running/completed/failed/cancelled/waiting-for-master | Bridge task status | +| Turn started/completed and event-derived tool/permission activity | Runtime observed; use only normalized, allowlisted fields | +| Selected model and reasoning level | Runtime reported; omit if absent | +| Summary, changed files, test commands/status, issues | Agent report; not independent verification | +| Phase and generic progress | Unsupported; keep `null` | + +Do not interpret `TaskResult.status = completed` as host acceptance or review PASS. diff --git a/TASK_FEEDBACK_SCHEMA_RECOMMENDATION.md b/TASK_FEEDBACK_SCHEMA_RECOMMENDATION.md new file mode 100644 index 0000000..458f1b7 --- /dev/null +++ b/TASK_FEEDBACK_SCHEMA_RECOMMENDATION.md @@ -0,0 +1,103 @@ +# Task Feedback Schema Recommendation + +**Decision:** Do **not** freeze the full `TaskFeedbackSnapshotV01` candidate yet. Freeze the core envelope and source semantics now; keep uncertain fields nullable/deferred until their state transitions are directly observed. + +Evidence: real ZCode `0.16.9` app-server probe documented in [ZCODE_APPSERVER_EVENT_CAPABILITY_PROBE.md](ZCODE_APPSERVER_EVENT_CAPABILITY_PROBE.md) and [run-001-summary.json](probe/evidence/run-001-summary.json). + +## KEEP + +- `schema_version`, `task_id`, `attempt`, and the existing Bridge task `status` vocabulary. These identify the Bridge task and its attempt; they are not app-server turn states. +- `model`, but source it only from a runtime snapshot or runtime-reported selection. The probe observed `settings.model.current.providerId`, `modelId`, and `options.reasoningLevel`. Preserve provider/model IDs and mark reasoning level nullable. Do not substitute request or catalog defaults and call them actual selection. +- Final `result` fields after a terminal result is available: `summary`, `issues`, `files_changed`, `tests`, `started_at`, and `finished_at`. Keep the Agent-report origin explicit; these are not Bridge-verified file diffs or independently verified test outcomes. +- `phase` and `progress` nullable slots if forward compatibility is valuable. Their values remain `null` until the runtime supplies explicit, structured phase/progress events. + +## CHANGE + +- Define `activity` as **last observed activity**, not “the operation currently running.” The probe observed `model.streaming` tool events and `tool.updated`, but did not establish the exact state mapping or whether the last update remains active. Include `observed_at`; do not render “Tool running” unless a confirmed runtime state says so. +- Narrow activity kinds to values directly supported by normalized, allowlisted events. Do not use model output text or tool arguments as a phase/action classifier. The observed runtime kinds include `tool_call` and tool-related `tool.updated`; `visible_output` is content, not an activity label. +- Give result subfields an explicit source, for example `source: "agent_report"`; give lifecycle/model fields `source: "bridge"` or `source: "runtime_snapshot"`. This prevents reported test/file claims from appearing runtime-verified. +- Define `interaction.state = "not_observed"` as “no interaction was observed in the event/query range,” never as “none exists.” A pending state must be backed by a current pending-request snapshot or an unmatched request event with a validated correlation rule. +- Define `duration_ms` as derived from Bridge `started_at` and `finished_at`, and populate it only when both are available. It is wall-clock task duration, not model execution time. + +## ADD + +- Add an optional runtime snapshot provenance block with `session_status`, `event_seq`, and `state_revision` when read from `session/read`. The probe observed these fields. Do not expose raw `pendingRequestIds`; at most expose a validated count/state after testing a real pending permission and input request. +- Add event `observed_at` and source sequence metadata to normalized activity records so a consumer can distinguish a recent event from an ongoing operation. +- Keep Bridge lifecycle state and app-server session state separate. They are separate state machines; one cannot stand in for the other. + +## REMOVE + +- Remove any promise that `phase` is inferred from tool kind or model text. +- Remove percentage progress derived from token counts, tool counts, elapsed time, message count, or iterations. +- Remove a generic `retry` label based only on `streamRecovery.updated`; its semantics were not established. +- Remove “Tool running · shell” as a default renderer claim from the current evidence. The task observed tool signals, but the redacted capture did not establish a reliable active-state mapping or shell name. + +## DEFER + +- Non-null `phase` and `progress`: no explicit runtime phase/progress event was observed. +- `current_action` / current operation: currentness has not been proven. If retained in the v0.1 type, only populate a `last_activity` interpretation. +- Permission-pending and user-input-pending state: permission request/resolution events appeared, but no human wait round trip or ID correlation was tested. User input was not observed. +- Cancellation, task/tool/infrastructure failure, and retry state: not tested in a real task. +- Full snapshot recovery after app-server restart: same-process reads worked; recovery of the unique empty session failed, and active/completed-task recovery is unverified. +- Cross-version/model/provider guarantees: only ZCode `0.16.9`, one provider, and GLM-5.3 were probed. + +## Freeze recommendation + +Freeze a **core v0.1** contract containing: + +```ts +type TaskFeedbackSnapshotV01 = { + schema_version: "0.1"; + task_id: string; + attempt: number; + status: + | "queued" + | "running" + | "completed" + | "failed" + | "cancelled" + | "waiting_for_master"; + model: { + provider_id: string | null; + model_id: string | null; + reasoning_level: string | null; + source: "runtime"; + } | null; + phase: null; + progress: null; + activity: { + kind: "tool_call" | "tool_update"; + summary: string; + observed_at: string; + currentness: "last_observed"; + } | null; + interaction: { + state: "not_observed" | "pending" | "answered"; + kind: "permission" | "user_input" | null; + } | null; + result: { + source: "agent_report"; + summary: string; + issues: string[]; + files_changed: string[]; + tests: { + command: string; + status: "passed" | "failed" | "not_run"; + details?: string; + }[]; + started_at: string | null; + finished_at: string | null; + duration_ms: number | null; + } | null; +}; +``` + +This proposed type freezes nullability and provenance. It does **not** authorize non-null `phase` or `progress` values, nor a claim that an activity remains active. Keep `interaction.state` weak until a real pending interaction is tested. If the contract cannot define evidence rules for those nullable fields, defer freezing the full schema rather than freezing ambiguous meanings. + +## Renderer implications + +- Safe now: Bridge task status, runtime-reported model metadata when present, successful `turn.completed` as a runtime turn outcome, and final result labeled “Agent report.” +- Conditionally safe: “Last observed tool event · Write,” only from an allowlisted event and only if the renderer does not imply that the tool is still running. +- Unsupported by this probe: phase checklists, test-case counts, percentages, confirmed permission/user-input waiting, generic retry, and recovery guarantees. + +No production schema or code was changed as part of this recommendation. diff --git a/ZCODE_APPSERVER_EVENT_CAPABILITY_PROBE.md b/ZCODE_APPSERVER_EVENT_CAPABILITY_PROBE.md new file mode 100644 index 0000000..52c2887 --- /dev/null +++ b/ZCODE_APPSERVER_EVENT_CAPABILITY_PROBE.md @@ -0,0 +1,180 @@ +# ZCode App-Server Event Capability Probe + +**Date:** 2026-10-05 (Asia/Shanghai) +**Scope:** Probe-only runtime research. No production Bridge source, schema, worker, or renderer was modified. + +## Environment + +| Item | Observed value | +|---|---| +| ZCode CLI / bundled app-server | `0.16.9` (`zcode.cjs --version`; no separate app-server version string was reported) | +| Bridge checkout | `2596759198fa826c2b7ac0478c5682da996e9727` | +| OS | Windows (PowerShell host) | +| Node | `v24.16.0` | +| Provider / model | `account:bigmodel-individual-coding-plan` / `GLM-5.3` | +| Runtime-reported reasoning level | `max` in the session snapshot | +| Second runtime/model path | Not run | + +The probe spawned the installed `zcode.cjs app-server --stdio` directly, created temporary workspaces under the OS temp directory, and used the account-provider reply in memory. Probe records retained event names, sequence numbers, key names, and a small allowlist of metadata. They did **not** retain model text, reasoning text, tool input/arguments, headers, credentials, or raw RPC frames. + +Sanitized evidence summary: [probe/evidence/run-001-summary.json](probe/evidence/run-001-summary.json). + +## Method and evidence levels + +- **Observed:** returned by the real runtime or received on its live `session/event` stream during this probe. +- **Documented:** method/shape is present in the installed runtime bundle or local Bridge code, but this probe did not observe the behavior. +- **Inferred:** plausible interpretation that is not a runtime guarantee. +- **Not observed:** this probe provides no evidence that the behavior exists or is absent. + +The probe performed read RPCs for `runtime/capabilities`, `session/read`, `session/events`, `session/subagents`, and `session/list`; created and closed temporary sessions; and ran one harmless coding task that was asked to create and verify one file in an isolated temporary workspace. A later host-side inspection found that the requested file was absent. A reconnect attempt used a uniquely named empty workspace. The probe did not run destructive commands or approve an external side effect. + +## Probe scenarios + +| Scenario | Result | +|---|---| +| A. Normal coding task | **Partially observed.** The runtime returned `turn.completed` with `resultType: success`, but a later host-side inspection found that the requested `probe.txt` was absent. Runtime turn success did not establish that the task objective was satisfied. | +| B. Tool-heavy lifecycle | **Partially observed.** The same task produced 9 tool calls in the terminal event metadata, 26 `tool.updated` events, and model-stream tool events. The task was not designed as a multi-file stress test. | +| C. Permission request | **Partially observed.** Five `permission.requested` and five `permission.resolved` events appeared during the normal task. No inbound `interaction/requestPermission` RPC requiring a human decision was observed, so a pending human approval round trip was not tested. | +| D. User input request | **Not run.** No `interaction/requestUserInput` RPC was triggered. | +| E. Cancellation | **Not run.** No turn was cancelled. | +| F. Failure | **Not run.** No task-level, tool-level, or infrastructure failure was intentionally induced. | +| G. Retry / recovery | **Partially observed.** Nine `streamRecovery.updated` events appeared. This is evidence of stream-recovery notifications only; it does not establish a general tool/model/provider retry contract. | +| H. Generic progress | **Not observed.** No structured current/total/percentage progress event appeared in the captured taxonomy. | + +## Runtime snapshot and RPC observations + +`runtime/capabilities` succeeded and returned `{ "independentPlanState": true }`. + +`session/create` and same-process `session/read` succeeded. The returned snapshot contained these top-level groups: + +```text +messages, projection, protocol, runtime, session, settings, +slashCommands, todoGroups, todos +``` + +Relevant observed nested fields included: + +- `session`: `sessionId`, `status`, `mode`, `sessionKind`, `model`, `workspace`, timestamps, and title metadata. +- `runtime`: `eventSeq`, `stateRevision`, `pendingRequestIds`, and goal-verification fields. +- `settings`: `mode`, `model`, `permission`, and `thoughtLevel`. +- `settings.model.current`: `providerId`, `modelId`, and `options.reasoningLevel`. + +The session snapshot reported `GLM-5.3`, provider `account:bigmodel-individual-coding-plan`, and reasoning level `max`. This is runtime-reported selection, not the request or catalog default. The completed task snapshot was `idle`, had `eventSeq: 261`, and had zero pending request IDs. The probe did not sample `session/read` while a turn was actively blocked, so these fields do not yet prove a running/waiting state transition. + +`session/events` succeeded in the same app-server process and returned event records after a sequence cursor. `session/subagents` on an empty session returned error code `-32004`; this is **not** evidence that subagents are generally unsupported. + +### Reconnect attempt + +For a newly created, idle session in a unique temporary workspace: + +1. The initial snapshot reported `status: idle` and `eventSeq: 0`. +2. After stopping that app-server process, `session/read` and `session/events` in a new process returned `-32004`. +3. `session/list` in the new process returned 50 session entries but did not contain that unique workspace/session ID. +4. `session/resume` for that exact session ID returned `-32004`. + +An earlier exploratory list contained a probe-workspace session that could be resumed, but the returned record was not identity-matched to the just-created session. It is excluded as recovery evidence. **Recovery of a completed or active task session after app-server restart remains unverified.** The observed behavior does show that direct `session/read` requires the session to be present in the current app-server session registry; a successful runtime query is not available merely from the old session ID. + +## Observed event taxonomy + +The normal task produced 261 live events, with the following event-name counts. The observed stream began at sequence 1 and ended with `turn.completed` at sequence 261. This is a single-run observation, not a cross-run ordering guarantee. + +| Event name | Count | Safe conclusions from this capture | +|---|---:|---| +| `session.titleUpdated` | 1 | Session title metadata changed. Title content was not retained. | +| `turn.started` | 1 | A structured turn-start event exists. `turnId` and `seq` were present in the event envelope. | +| `session.updated` | 36 | Session/model/telemetry update events exist. Payload key names included model/provider metadata and, in some updates, request/response header fields and usage telemetry. Values were not retained. | +| `model.streaming` | 177 | Stream events included `reasoning_delta`, `text_delta`, `tool_input_start`, `tool_input_delta`, `tool_input_end`, and `tool_call`. `delta`, `input`, and tool metadata are not safe to log wholesale. | +| `tool.updated` | 26 | Structured tool update events exist. This capture did not retain enough state values to define a stable started/running/completed/failed mapping. | +| `permission.requested` | 5 | Permission lifecycle request notifications occurred. Request IDs/details were not retained, so request/resolve correlation is not established. | +| `permission.resolved` | 5 | Permission lifecycle resolution notifications occurred. Equal counts do not prove one-to-one correlation. | +| `streamRecovery.updated` | 9 | Structured stream-recovery updates exist; their precise meaning and relation to retries were not established. | +| `turn.completed` | 1 | Terminal turn event reported `resultType: success`; `tokenCount`, `toolCallCount`, duration, and usage metadata were present. Token/usage values are excluded from feedback and evidence. | + +### Sensitive fields and privacy boundary + +The event envelope provided `sessionId`, `turnId`, and numeric `seq` on observed session events. The captured normal-task `model.streaming` payload shape included `delta`; tool-related variants included `input`, `toolCallId`, and `toolName`. `session.updated` payload key names included `requestHeaders`, `responseHeaders`, `baseURL`, request/trace identifiers, and usage metadata. The probe recorded key names only for these fields. + +Do not persist or render raw event payloads. Hidden `reasoning_delta` content, visible text deltas, tool input/arguments, request/response headers, and provider telemetry must stay outside default feedback. A strict field allowlist is required even for diagnostic evidence. + +## Capability findings + +### Lifecycle + +- **Observed:** `turn.started` and successful `turn.completed` in a real task. +- **Not observed:** `turn.failed`, runtime cancellation, or separate task lifecycle states such as queued or waiting-for-master. Those are Bridge task states, not established app-server events by this probe. +- `session.status` was `idle` before and after the task in the snapshots captured. An in-turn `session/read` sample was not taken. + +### Model selection + +- **Observed:** snapshot `settings.model.current` reported provider ID, model ID, and `options.reasoningLevel`; the selected model was `GLM-5.3`, reasoning `max`. +- **Observed:** stream-side `session.updated` payload keys included `modelId` and `providerId`. +- **Limit:** only one provider/model path and one task were sampled. Stability across model switches, provider changes, and runtime versions is not established. + +### Tool behavior and activity + +- **Observed:** `model.streaming` tool-related kinds and `tool.updated` events. A `Write` tool name was observed in a tool event; the task’s terminal metadata reported 9 tool calls. +- **Not established:** the exact `tool.updated` state values, a reliable `tool started` boundary, or whether the most recent tool event still represents a currently active operation. +- Therefore the runtime supports **tool-related activity signals**, but this probe cannot justify text such as `Tool running · shell`. At most, a renderer may report a time-stamped **last observed tool event** after verifying the exact event and safe tool name. + +### Permission and user input + +- **Observed:** permission request/resolution event names during tool execution. No human approval request was left pending, and no permission interaction RPC round trip was tested. +- `runtime.pendingRequestIds` is present in snapshots, but this probe only observed an empty list before/after the completed task. Mapping those IDs to permission or user-input requests is unverified. +- **Not observed:** user-input request events, request IDs/question metadata, pending-state transitions, or resume behavior after a user answer. + +### Phase and progress + +- No runtime phase event or structured analysis/implementation/testing/review phase was observed. +- No generic `current`, `total`, percentage, step, stage, or test-case progress was observed. +- `iteration`, `messageCount`, `toolCount`, token counts, duration, and tool-call counts are runtime metadata, not task completion progress. Do not transform them into percentages or test-case counts. +- Current evidence supports `phase = null` and `progress = null`. + +### Retry and failure + +- `streamRecovery.updated` was observed, but no semantics were proven for model retry, provider retry, tool retry, or task retry. +- No task, tool, or infrastructure failure path was intentionally run. A unified `retry` feedback event is not justified by this probe. + +### Session/state recovery + +- `session/read`, `session/events`, and `session/list` are callable in the tested runtime; `session/resume` is also callable but the unique empty session test returned `-32004` after process restart. +- Same-process event history is queryable by `afterSeq` and `limit`. +- Rebuilding a trustworthy snapshot for a running turn after disconnect, restoring pending interactions, and recovering a completed Bridge task were not established. + +## Evidence table + +| Capability | Observed | Structured | Stable enough for v0.1 | Feedback use | +|---|---|---|---|---| +| Actual selected model | Yes, snapshot | Yes | One-path evidence only | Show with runtime source; omit when unreported | +| Reasoning level | Yes, snapshot | Yes | One-path evidence only | Optional model metadata; never show reasoning content | +| Turn start/completion | Yes | Yes | Useful for this runtime path; cross-version not tested | Lifecycle evidence | +| Tool-related activity | Yes | Yes | State mapping/currentness not established | Last observed activity only, after safe normalization | +| Permission request/resolution event | Yes, paired counts only | Yes | Correlation/pending behavior not established | Do not claim “waiting for permission” from this run | +| User-input request | Not observed | Unknown | No | Keep unknown / not observed | +| Phase | Not observed | No | No | `null` | +| Generic progress | Not observed | No | No | `null` | +| Test-case count | Not observed | No | No | Omit | +| Retry | Stream-recovery event only | Partially | No unified retry semantics | Event-only diagnostic; no generic retry label | +| Session read/replay | Same process observed | Yes | Same-process only | Queryable while session is registered | +| Cross-process session recovery | Empty-session attempt failed; ambiguous older session excluded | Partial | No | Do not promise recovery | +| Final result | Successful turn observed | Yes | This single task only | Runtime turn outcome; keep Agent report separately labeled | + +## Answers to the five questions + +**Q1. Can the runtime provide more structured Native Feedback data than the Bridge currently uses?** +Yes. This probe observed `permission.requested/resolved`, `streamRecovery.updated`, additional `model.streaming` kinds, rich `session.updated` metadata, and snapshot fields such as `runtime.pendingRequestIds`, `runtime.stateRevision`, and `settings.model.current`. Some fields carry sensitive material or lack proven feedback semantics; they should not be exposed without an allowlist and targeted validation. + +**Q2. Can we reliably know whether the agent is waiting, using a tool, or generating output?** +We observed output-stream and tool-related events, plus permission request/resolution notifications. We did not prove a stable current-operation state or a pending human interaction state. `activity` can represent a last observed event, but “tool running” and “waiting for user” are not established by this run. + +**Q3. Can we reliably obtain phase or progress?** +No structured phase or generic progress was observed. Keep both `null`. + +**Q4. Can Bridge rebuild state after disconnect/restart with native queries?** +Same-process `session/read` and `session/events` work. The unique empty-session reconnect attempt could not be read, listed, or resumed after the original process ended. Active/completed-task recovery remains unverified; do not claim full reconstruction. + +**Q5. What is the smallest trustworthy snapshot?** +Freeze only task identity/attempt/status and runtime-reported model metadata plus clearly source-labeled final result/timing. Keep phase/progress null. Represent tool events as last-observed activity only; do not assert currentness. Keep interaction state weak until a real pending interaction and recovery round trip are captured. A runtime turn success must remain distinct from task acceptance: this probe returned success while the requested artifact was missing at host-side inspection. + +## Limitations + +This probe covers one installed runtime version, one provider/model path, one successful task, and one empty-session reconnect attempt. It did not exercise user input, a human permission wait, cancellation, failure, a second model/provider, or reconnect during an active turn. Event counts and ordering are single-run observations, not protocol guarantees. No raw event dump was saved because it could contain reasoning, tool arguments, credentials, or provider telemetry. diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/README.md b/docs/research/native-cli-vs-appserver-2026-10-05/README.md new file mode 100644 index 0000000..f1e3658 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/README.md @@ -0,0 +1,29 @@ +# Native CLI vs app-server 研究(2026-10-05) + +> Status: RESEARCH +> Date: 2026-10-05 +> 研究结论,不代表当前实现。观测版本 ZCode CLI 0.16.9;官方源码提交 `29628c9acdb81b703bbd4080c207a0e7ce5e276e`。 + +本目录是一次完整研究的交付物:6 篇文档、实验脚本、以及证据索引。原始 ledger 不进入仓库,归档位置与 SHA-256 见 [evidence/manifest.json](evidence/manifest.json)。 + +## 结论摘要 + +- Executor 架构方向:**GO WITH CONDITIONS**。引入 NativeCliExecutor 候选、保留 AppServerExecutor、建立统一 Run Ledger 与顺序 Handoff 的方向合理;但没有足够证据把 Native 切成默认 Coding Executor,生产默认继续 app-server。 +- 跨 Host 并发 attach/control:**NO-GO**,除非上游提供明确的 ownership/control 协议或另行证明安全。 +- Session handoff:双向顺序恢复可行;"全部 Runtime 状态无损"未证明。 + +## 文档 + +| 文档 | 内容 | +|---|---| +| [ZCODE_CLI_VS_APPSERVER.md](ZCODE_CLI_VS_APPSERVER.md) | 总报告:两条执行路径的深度审计 | +| [ZCODE_RUNTIME_CALLCHAIN.md](ZCODE_RUNTIME_CALLCHAIN.md) | 官方源码调用链 | +| [ZCODE_EXECUTOR_ARCHITECTURE_RECOMMENDATION.md](ZCODE_EXECUTOR_ARCHITECTURE_RECOMMENDATION.md) | Executor 架构建议与决策门槛 | +| [ZCODE_EXECUTOR_BENCHMARK.md](ZCODE_EXECUTOR_BENCHMARK.md) | 基准与成本审计 | +| [ZCODE_SESSION_HANDOFF.md](ZCODE_SESSION_HANDOFF.md) | Session handoff 审计 | +| [VALIDATION.md](VALIDATION.md) | 交付验证记录 | +| [experiments/README.md](experiments/README.md) | 实验脚本职责与证据文件清单 | + +## 适用范围 + +结论只适用于记录的版本和实验范围,不构成对其他安装版本的保证。文中标注 NOT RUN 的部分保持未验证。 diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/VALIDATION.md b/docs/research/native-cli-vs-appserver-2026-10-05/VALIDATION.md new file mode 100644 index 0000000..768a1b9 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/VALIDATION.md @@ -0,0 +1,28 @@ +# 交付验证记录 + +> Status: RESEARCH +> Date: 2026-10-05 +> 本次研究的交付验证记录。索引见 [README.md](README.md)。 + +2026-10-05,本次工作区基线 `2596759198fa826c2b7ac0478c5682da996e9727`。 + +| 检查 | 结果与范围 | +|---|---| +| `npm ci --ignore-scripts` | 本次成功,用于准备已有Bridge core模块;未修改package/lockfile | +| `npm run build:core` | 本次成功;不等于全项目测试验收 | +| 成功组候选独立重放 | 10/10 artifact、scope、functional acceptance通过;T2两组10单测及3/3 mutation kills | +| 失败组候选独立重放 | 10次执行均未成功终态;T1-N留下修改可通过验收;其余不满足task artifact要求;全部排除性能统计 | +| `.mjs` syntax | 实验脚本逐个 `node --check` 通过 | +| `verify-artifacts.mjs` | 必需5份文档;46个固定源码/本地链接存在及line anchor范围;10成功/10失败ledger、配对promptHash/commit/cwd、全部要求指标字段;敏感对象key白名单检查通过 | +| `git diff --check` | 本次通过 | +| 实际ZCode取消 | Native force tree kill、app-server stop及EOF退出成功;范围仅本次自有helper树 | +| Windows JobObject | 最终自有Node helper树验证成功;实际ZCode集成NOT RUN | +| 全项目 `npm test` / `typecheck` | NOT RUN:本次没有修改生产源码、公共合同或package配置;不引用历史测试冒充本次结果 | +| 默认Executor /发布 | 未更改默认;未提交/推送/创建PR/发布 | +| 临时敏感文件清理 | 已移除专用private-runtime中的credential/provider副本及最后3个自有raw model-I/O文件;保留专用SQLite、fixtures和安全ledger | + +仅 `.gitignore` 增加 `docs/research` 保留规则,其余变化在研究目录。没有修改真实Provider配置、用户Session数据库、任务索引、生产Executor或其他工作树。研究期间只在本次新建fixture执行Git reset/clean。 + +源码worker独立输出位于临时source-review worktree,主审已检查其文档、固定源码与真实探针;没有直接把worker的“无所有权锁”等绝对表述当定论。最终保留absence evidence为SUPPORTED、危险resume为NOT RUN。 + +已知未解决项在总报告、Benchmark和Recommendation明确列出,尤其是严格配置对齐后的成功编码比较受网络故障影响;不把失败组当质量测量。研究产物可review,尚不构成生产架构实施验收。 diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_CLI_VS_APPSERVER.md b/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_CLI_VS_APPSERVER.md new file mode 100644 index 0000000..d7b0c56 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_CLI_VS_APPSERVER.md @@ -0,0 +1,73 @@ +# ZCode Native CLI 与 app-server 深度审计 + +> Status: RESEARCH +> Date: 2026-10-05 +> 研究结论,不代表当前实现。索引见 [README.md](README.md)。 + +研究日期:2026-10-05。结论适用于下述版本和实验范围,不构成对其他安装版本的保证。 + +## 结论 + +**两条入口共享 Agent Core;决定实际行为的是 Host 的 materialization、配置、能力端口和生命周期管理。** 本次没有证明 Native CLI 普遍更省 Token、代码质量更高或进程控制更可靠。5 类真实模型执行的隔离编码任务,两边各完成 5 次,独立验收均通过;观察组 app-server 总 Token 比 Native 少 6.65%,但两边 Memory 配置不同,因此不是严格的执行器因果比较。 + +双向、跨进程的**顺序** Session Handoff 已成功。正在运行的 CLI 会话能被另一 app-server 列出,但该 app-server 的 `read` 和 `stop` 返回未激活错误。没有证据支持安全跨 Host 并发接管。 + +架构建议为 **GO WITH CONDITIONS**:可以引入双 Executor 的实验实现与顺序交接;目前维持已有 app-server 默认。Native CLI 是否应成为默认 Coding Executor,证据不足,应通过配置对齐后的扩展任务验证决定。详见 [架构建议](ZCODE_EXECUTOR_ARCHITECTURE_RECOMMENDATION.md)。 + +## 版本与证据边界 + +| 项目 | 本次记录 | +|---|---| +| Bridge 工作区 | `C:/Users/Sandy/.codex/worktrees/f7bc/codex-zcode-bridge` | +| Bridge 基线 | `2596759198fa826c2b7ac0478c5682da996e9727`,package 1.0.5 | +| 研究分支 | `codex/research-native-cli-appserver-20261005` | +| 官方源码 | [zai-org/ZCode 固定提交](https://github.com/zai-org/ZCode/tree/29628c9acdb81b703bbd4080c207a0e7ce5e276e),`29628c9acdb81b703bbd4080c207a0e7ce5e276e` | +| 本机执行文件 | `C:/Users/Sandy/AppData/Local/Programs/ZCode/resources/glm/zcode.cjs`,CLI 0.16.9 | +| Node | 24.16.0,Windows | +| 模型 / Provider | `GLM-5.3-Flash` / `account:bigmodel-individual-coding-plan` | +| 编码实验 | low、yolo;工具限制为 Bash/Edit/Read/Write | +| 身份、存储 | 同一授权账户;复制到专用临时配置及加密凭据目录,独立 SQLite;未修改真实用户配置 | + +**UNKNOWN:公开源码提交与已安装 0.16.9 的完整构建对应关系。** 发现了明确差异:公开 `runPrompt` 源码默认关闭动态 Workflow;安装版未经规范化的 CLI 请求却注册了 Workflow 工具。下文 SOURCE 结论指固定源码,RUNTIME 结论指实测安装版,不能互相替代。 + +证据标签:`CONFIRMED — SOURCE` = 已定位的源代码事实;`CONFIRMED — RUNTIME` = 本次实机证明;`SUPPORTED` = 源码和实验支持、尚非完整证明;`HYPOTHESIS` = 待验证解释;`UNKNOWN` = 不确定。未执行的危险或范围外操作写 `NOT RUN`。 + +## Q1–Q8 对照 + +| 问题 | 结果与证据强度 | +|---|---| +| Q1 共享 Core? | **CONFIRMED — SOURCE**:两边 `createZCodeApp` 创建 `AgentRuntime`,经不同输入 facade 汇合于 `executeTurn`,使用相同 turn loop / model runner。见 [调用链](ZCODE_RUNTIME_CALLCHAIN.md)。 | +| Q2 System Prompt 一致? | **CONFIRMED — RUNTIME**:默认观察组不一致;Memory 对齐后,同 cwd 同请求的 System Prompt 哈希可以一致。即使 System 相同,meta user context 仍可能不同,不能宣称整个 request 等价。 | +| Q3 Tools 一致? | **CONFIRMED — RUNTIME**:自然配置 CLI 28 个、app-server 22 个,分别含 Workflow / Cron 特有工具;规范化到 4 工具后 schema 哈希相同。注册表相同也不意味着权限、浏览器端口、MCP 或实际可执行性相同。 | +| Q4 Reasoning 等价? | **CONFIRMED — RUNTIME**:low/high/max 均映射到 `thinking.type=enabled` 和对应 `output_config.effort`;实际 body 中 model 相同。Catalog 默认 max 不等于已配置会话默认,本次明确指定 low。无显式选择的真实用户默认行为未覆盖。 | +| Q5 Memory 额外调用? | **CONFIRMED — SOURCE**:use 与 extraction 独立,成功 main turn 后可触发额外 `project_memory_extract`;CLI 默认 extraction=false,Bridge preferences 直接关闭 Memory。小型开启探针未观察到抽取请求,不能证明抽取永远不发生。 | +| Q6 Subagent 策略? | **CONFIRMED — SOURCE**:共享 subagent runner,背景执行受 profile、runInBackground、autoBackgroundMs 和模型选择覆盖控制。没有两入口必然不同的默认模型证据。编码组禁止 Agent 等工具,app-server 查询子会话为空;未做允许后台 subagent 的配对性能实验。 | +| Q7 Permission 造成 replanning? | **CONFIRMED — RUNTIME**:build 模式 CLI Write 被拒绝;app-server 有 Host 往返并允许指定实验文件。两边均 2 个 model requests,本探针没有证明拒绝增加 request 数。 | +| Q8 Task Terminal 相同? | **CONFIRMED — SOURCE**:CLI 有 Workflow settle 和可选 Memory drain;Bridge 以一个 `turn.completed` 收口 prompt,随后 checkpoint / cleanup。**SUPPORTED**:开启背景能力时,仅首个 turn terminal 不足以定义完整 Task 完成;完整 Workflow 实机结算 NOT RUN。 | + +## Context 与模型执行 + +Context builder 共同组织 stable identity、动态行为、session guidance、Memory、环境信息、skills、workspace instructions。AGENTS、skills/MCP、custom commands 与 mid-conversation system 都有配置入口;共同 builder 只能证明处理机制共享,不能证明输入配置相同。Browser 一边可注入 CLI headless runtime,另一边可转发 Host browser interaction。见 [源码索引](ZCODE_RUNTIME_CALLCHAIN.md#context与能力注入)。 + +实际请求摘要只保留 model、thinking/effort、工具名、System / message 哈希和长度、usage;未保存 headers、API key、原始请求或隐藏推理。reasoning 探针中工具 schema SHA-256 均为 `477fe506ecaa4167913c555851f18d94879e91902491fe79081dc262105afd35`。 + +观察组 Native 的 system_prompt 为 9,058 字符,app-server 为 6,899;首次 input 每任务 Native 多 504 tokens。Memory 同时开启的探针,两边 System 哈希同为 `de8af4f2c9058feebfd5d498d271c787bc5e2b845f6cb4ba835dad4dee2a6ce7`,但 meta user context 摘要长度仍不同。补充关闭两边 Memory 的 10 次运行,System 哈希均为 `356698f8c09a223757c95d26f7aa1505d66e15d032486eaec1d85f75fe0c320c`;因网络故障没有成功完成的配对编码结果。 + +Provider Registry 是两边都有的初始化组件,不能简化为“CLI 有 provider、app-server 没有”。Bridge 需要正确处理 account snapshot、builtin revision、模型 ID 映射和 runtime auth。最初使用真实配置路径的 smoke,app-server 成功、CLI resume 后没有有效默认模型;在**临时副本**补齐 default selection 与官方加密凭据后同会话顺序执行成功。这证明配置完整性会影响观测,不能归因于 Agent Core 缺陷。 + +## 实验覆盖与不确定项 + +已完成:源码路径复核、真实 Flash 源码研究 worker、5 类任务 × 2 Executor 的成功观察组、另一组 10 次 Memory 对齐但失败的执行、6 次 reasoning 探针、build 权限探针、Memory 开启探针、A/B 顺序 Handoff、C 运行中只读探针、两种实际 ZCode 进程取消、Windows Job Object 专用 helper 树实验。 + +未覆盖:官方 Desktop 实际启动及其 preferences;完整 MCP/浏览器/skills/custom command 等价 A/B;允许 subagent / Workflow 的负载;项目 Memory 已存在内容的完整跨端恢复;真实 ZCode Job Object 集成;Windows 控制台 Ctrl+C/Ctrl+Break;CPU 与峰值内存;重复种子、统计显著性或计费金额。对这些结论不确定。 + +研究任务最初通过已安装 Bridge 提交,但排队未启动,已取消,不能记为运行成功;随后直接使用同一安装版 stdio app-server 完成授权源码任务和探针。独立审查与验收由 Codex 完成,worker completed 不等于接受。没有修改生产 Executor、默认值或公共协议,没有提交、推送、发布。 + +## 交付索引 + +- [实际调用链与固定源码引用](ZCODE_RUNTIME_CALLCHAIN.md) +- [Session 顺序交接、并发控制边界](ZCODE_SESSION_HANDOFF.md) +- [Benchmark、质量验收、Token 拆分与失败组](ZCODE_EXECUTOR_BENCHMARK.md) +- [架构决策、进程管理与下一步门槛](ZCODE_EXECUTOR_ARCHITECTURE_RECOMMENDATION.md) +- [可复核证据与实验操作说明](experiments/README.md) + diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_EXECUTOR_ARCHITECTURE_RECOMMENDATION.md b/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_EXECUTOR_ARCHITECTURE_RECOMMENDATION.md new file mode 100644 index 0000000..fabd0b7 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_EXECUTOR_ARCHITECTURE_RECOMMENDATION.md @@ -0,0 +1,97 @@ +# Executor 架构建议与决策门槛 + +> Status: RESEARCH +> Date: 2026-10-05 +> 研究方案,不是已批准决策。若据此改变执行路径,需要单独 ADR。索引见 [README.md](README.md)。 + +**决策:GO WITH CONDITIONS。** 引入 NativeCliExecutor 的候选实现、保留 AppServerExecutor、建立统一 Run Ledger 与顺序 Handoff,方向合理;本次仅给研究方案。当前没有足够证据把 Native 切成默认 Coding Executor,生产默认继续 app-server。跨 Host concurrent attach/control 为 **NO-GO**,除非上游提供明确 ownership/control 协议或另行证明安全。 + +## 15 个问题的直接回答 + +| # | 问题 | 回答 | +|---|---|---| +| 1 | 共享 Agent Core? | 是,固定官方源码两条路径汇合于 AgentRuntime.executeTurn / 同一 loop / model runner;SOURCE | +| 2 | 真正分叉点? | Host bootstrap、runtime materialization、输入 admission、能力 broker、preferences、收口与进程管理;不是一个独立 Coding engine | +| 3 | System Prompt 一致? | 自然配置不一致;对齐 Memory 后所测 system 哈希相同。完整 context 未证明普遍一致;RUNTIME / UNKNOWN | +| 4 | Tools 一致? | 自然列表不同;规范化 4 工具 schema 相同。可执行 policy、Browser/MCP/Workflow ports 仍须 separately 校准 | +| 5 | Model Execution 一致? | 所测 Flash + low/high/max 的 body selection/effort 映射一致;auth、背景模型、上下文及辅助调用不保证一致 | +| 6 | Native 为什么可能省 Token? | 某任务少一次模型调用可节省大 context 的重复输入;复用官方 Host 可能减少错误重建。前者 T3 观测成立,普遍解释仍 HYPOTHESIS;总体观察组 A 更少 | +| 7 | Native 为什么可能质量更高? | 尚不确定;此样本均通过、评分相同。permission/MCP/context 等对齐前不能归因于入口 | +| 8 | 已证实 vs 猜测? | shared Core、自然工具/Memory/policy差异、档位映射、顺序恢复已证实;整体优劣、资源优劣、后台差异因果、完整无损仍未知 | +| 9 | CLI → app-server Handoff? | 是,成功完成的所测会话跨进程恢复、marker/历史/Read-Edit state/选择保持;RUNTIME | +| 10 | app-server → CLI? | 是,close 后仍可恢复;本次显式 yolo override;RUNTIME | +| 11 | running CLI 能 attach? | 另一 Host 能 list,read/stop 不激活;安全并发 resume NOT RUN。没有已验证 attach/control 通道,不启用 | +| 12 | 应采用双 Executor? | GO WITH CONDITIONS:作为能力路由候选,须完成版本门槛、normalization、terminal、cleanup、lease、真实负载验证 | +| 13 | 默认 Coding Executor? | 现在继续 app-server;Native 先 opt-in batch pilot。不能凭未成立的成本/质量优势直接切换 | +| 14 | app-server 职责? | 长会话、运行中交互与审批、动态模型/思考档位、steering、子会话检查、事件回放与细粒度读取 | +| 15 | 保留现有 Bridge 能力? | Provider/account/revision/ID mapping、模型目录预检、workspace身份/执行目录、Task/Attempt隔离、规范报告、事件seq/turn去重与回放、timeout/cancel、outcome checkpoint、cleanupVerified、usage隐私白名单 | + +## 候选职责边界 + +```mermaid +flowchart TD + TM[Task Manager: 目标与验收] --> RM[Run Manager: attempt / lease / deadline] + RM --> ROUTE[Executor Router: 显式能力路由] + ROUTE --> N[NativeCliExecutor: 独立 batch process] + ROUTE --> A[AppServerExecutor: Session Programmability] + N --> EN[Event Normalizer: allowlisted events / terminal facts] + A --> EN + EN --> LEDGER[Run Ledger: outcome / usage / cleanup / acceptance] + LEDGER --> H[顺序交接 gate] + H --> ROUTE +``` + +路由按任务能力选择:确定需求且无需执行中审批/steering 的 acceptance-driven batch 可以试用 Native;需要保留活动会话或动态控制时 app-server 更合适。不要把“任务叫 bug fix”直接当必须 Native 的规则,也不要在同一 attempt 未知状态时启动另一 Executor 重跑。 + +Native stream-json 是结构化 Runtime 输出,有实现基础;实际流可能包含 Provider 请求相关字段,所以要像现有 Bridge 一样做安全白名单,不原样转发 stdout。不能因“结构化”便假定没有 headers、隐私或隐藏推理风险。最终 result、process exit、cleanup 与验收分别入账。 + +## 正确的完成状态 + +建议 Run Manager 用事实门槛,而非仅事件名: + +1. 主输入有已关联的 terminal result;event seq、turnId 不把回放旧事件当新结果。 +2. 若允许 Workflow/subagent/background,确认该 run 所拥有工作均完成、失败或被明确取消。无可靠观测时标 incomplete,不能只等待固定几秒冒充 settle。 +3. 显式决定 Memory extraction / title / summary 等辅助请求是否属于 run 成本及 terminal 门槛;按 querySource 分账。 +4. 保存 outcome、报告与已知 usage,之后释放 Runtime/进程;cleanupUnknown 不清空 workspace occupied。 +5. Reviewer 另行验收。模型 success、进程 exit0、cleanupVerified、acceptancePassed 是四个不同事实。 + +Native 官方 settle 有条件且没有内部总 deadline;app-server `session/send` ACK 更不能当完成。现有 Bridge 已有 checkpoint 与 cleanupVerified,扩展时应保留,不另造丢结果的收口逻辑。 + +## Windows 进程控制 + +| 操作 | Native CLI 实测 | app-server 实测 | 结论边界 | +|---|---|---|---| +| spawn / PID / streams | 独立 Node→CLI,采集 stream-json;prompt经文件loader避免argv泄露 | 独立 Node→app-server stdio JSON-RPC | 两者都可被外部 Supervisor 管理 | +| Agent cancellation | 本次未测试 Native 优雅交互取消 | `session/stop` ACK 36ms,随后 terminal=cancelled,1秒观察后 60秒 helper 已死 | stop成功不终止 server;ACK 不等于进程树清理 | +| OS force cancel | 自建 Bash→Node 60秒 helper;`taskkill /PID /T /F` 359ms,helper 死、无 done marker | stop后 server仍活;关闭stdin后 parent exit0,parent/helper均死 | 一个特定树成功,不保证所有任意 detached child | +| Ctrl+C / Ctrl+Break | NOT RUN | NOT RUN | Node SIGINT/SIGTERM 名字不能替代 Windows Console Control Event 语义 | +| Timeout | Native 300秒研究deadline触发taskkill;保留未完成ledger | 协议失败/停止与父进程释放分开记录 | 全生命周期必须有budget与终止升级 | +| Job Object | 专用 helper 树:assign成功、kill-on-close、parent/child均死 | 未集成实际 ZCode | 证明 Windows机制可行,不是生产ZCode已验证 | +| CPU / RAM | 未测 | 未测 | 不确定哪边更省操作系统资源 | + +Windows Job Object 可约束子进程并支持 kill-on-job-close、资源 accounting;适用于两种 Executor,不专属于 CLI。[Microsoft Job Objects](https://learn.microsoft.com/en-us/windows/win32/procthread/job-objects)。实际集成应在 root spawn前后提供可靠 containment,最好 suspended create→assign→resume 或受控握手,以免先创建的孩子逃逸;检查 nested Job、breakaway、权限和实际 shell/browser 派生。工具根退出不保证 orphan 已被杀,须验证 Job 内外进程状态。 + +本次 Job helper 首轮 P/Invoke harness 的嵌套 struct 赋值没持久化,导致 kill-on-close 未生效;已修正,并以 finally 清理自有树。最终成功实验保存错误与修正范围,不掩盖初次失败。Windows Ctrl+C / Break 需要 console process-group 与特定条件,[Microsoft GenerateConsoleCtrlEvent](https://learn.microsoft.com/en-us/windows/console/generateconsolectrlevent);不能把 Node `kill('SIGTERM')` 宣称为同等优雅控制。 + +现有 Bridge 的 app-server 适配器本身每次 run 启动进程并带 deadline / process-tree cleanup。不存在“app-server先天不能受OS管理”的证据。差别主要是 Native batch 的自然进程终态与 app-server 长会话的 Agent-level 终态,需要 Host 对齐生命周期。 + +## GO 的前置条件 + +| 门槛 | 必须交付的证据 | +|---|---| +| 版本与 capabilities | 固定实际bundle/version/hash,入口flags/RPC能力预检;源码与安装版差异有明确fallback | +| Model normalization | Provider、model、reasoning、auth、default选取、辅助模型来源记录;不能只比较modelId | +| Context / tool normalization | 保存各分段及schema安全哈希,显式Memory use/extraction、AGENTS、skills、MCP、Browser、Workflow/subagent policy | +| terminal | 主turn与run完成分离;背景负载真实探针;失联后unknown而非盲重跑 | +| cleanup | 两种实际ZCode进程包含shell/helper/browser并完成deadline、取消、强杀、orphan检查;JobObject集成NOT RUN待完成 | +| handoff | 唯一session/workspace lease、旧Host释放后恢复;持久化及文件状态复核;不并发resume | +| quality / economics | 至少10个来自真实生产任务的扩展负载,多次重复/顺序平衡;配置对齐;含资源/完整usage,Reviewer独立验收 | +| regression | 保留现有事件、报告、usage隐私、错误码、cancel/continue、attempt evidence及恢复行为;公共合同变更另行评审 | + +当前状态:源码与基础真实探针已完成;双向顺序恢复已完成;严格对齐完整编码比较被网络故障阻断;CPU/RAM、完整背景负载及ZCode Job集成未完成。因此 **GO WITH CONDITIONS** 是候选方向,**不是实施已完成/默认切换获验收**。 + +## 后续工作包 + +先做一个不改变默认的 Native batch pilot,与现有 AppServerExecutor 使用相同 RunSpec/Outcome;接入stream-json白名单、结果checkpoint、deadline与统一cleanup,明确foreground-only profile。再扩展背景terminal与租约交接。最后用真实任务重复实验决定默认值,而非把本次5个小fixture的结果当发布依据。 + +本次交付只有研究文档、可复核实验脚本与安全摘要;没有重构生产代码、创建发布PR或发布版本。无需在本研究期间继续派发Coding Agent改生产架构。 diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_EXECUTOR_BENCHMARK.md b/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_EXECUTOR_BENCHMARK.md new file mode 100644 index 0000000..3bc69dd --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_EXECUTOR_BENCHMARK.md @@ -0,0 +1,98 @@ +# Executor Benchmark 与成本审计 + +> Status: RESEARCH +> Date: 2026-10-05 +> 基准数据只适用于记录的版本与场景。索引见 [README.md](README.md)。 + +## 范围与方法 + +用户授权使用 GLM-5.3-Flash。实际完成 5 类编码任务 × 2 Executor 的成功观察组;另尝试 10 次 Memory 对齐补充组,受网络故障全部没有成功执行终态。不是生产 Bridge 的修改验收,也不是统计显著的通用 Coding 排名。 + +fixture 为本次新建、可运行的独立 JS 仓库,代表 Bridge 常见逻辑;**不是宣称当前生产代码存在同样 bug**。包含真实 Read/Write/Bash、生成代码、执行测试、独立持有验收。中等功能实现仍只有几十行,不能外推大型真实工程。任务分别为: + +| ID | 类型 | 验收 | +|---|---|---| +| T1 | timeout parser 小 bug fix | decimal-string / safe integer 边界;拒绝 hex、指数、符号、空值等 | +| T2 | 新增 path overlap 单测 | 至少 8 个命名单测;POSIX/Windows;额外 mutation 验证 prefix/case/equality 三类错误 | +| T3 | retry 功能 | attempts、backoff cap、不重试、原错误 identity、参数验证与 hooks | +| T4 | cancelled 多文件行为 | 5×5 transition 矩阵、unknown 状态拒绝、报表计数 | +| T5 | aggregation 重构 | 内部共享 helper;公共语义、空输入、缺 usage、冻结输入不变 | + +相同 prompt、acceptance criteria、cwd、Git commit、Provider/Flash、low、yolo、OS、4 工具。每次从 fixture 基线 reset;交替执行器次序;禁止网络、安装、delegation、commit(模型 API 网络仍必需)。观察组基线 `30f3a1bee6018ee36aa1c344dea70c586cdf7699`。**未对齐项:Native Memory use 默认开、extraction 关;app-server Memory 整体关闭。** background/Workflow 工具已 deny,MCP/browser 无可用模型工具;标题生成请求在 app-server 编码组关闭。 + +低档位由会话 snapshot / configured selection 记录,并以专项请求 body 探针核验。观察组旧模型 I/O 受安装版日志轮转,未声称每个原始 request body 都被保存;保留 Provider runtime events 与完整 turn usage。每个 Agent turn 可能含多个 model requests,不能混用二者。 + +## 成功观察组结果 + +下表 `N=Native`,`A=app-server`。Token 为 Provider 报告 usage,不是币值;cached input 已包含在 input,不能再加一次。时长为 Runtime `turn.completed.duration`,单位秒。 + +| 任务 | N requests | A requests | N input / output / total | A input / output / total | N / A cached input | N / A turn 秒 | 验收 | +|---|---:|---:|---|---|---|---|---| +| T1 | 4 | 4 | 16,364 / 730 / 17,094 | 14,316 / 634 / 14,950 | 8,960 / 7,616 | 33.897 / 33.487 | 两边通过 | +| T2 | 4 | 4 | 17,332 / 845 / 18,177 | 15,331 / 844 / 16,175 | 11,648 / 11,392 | 34.783 / 32.461 | 两边 10 单测、3/3 mutants killed | +| T3 | 5 | 6 | 22,908 / 1,628 / 24,536 | 24,678 / 1,632 / 26,310 | 17,280 / 19,648 | 60.586 / 81.229 | 两边通过 | +| T4 | 4 | 4 | 17,328 / 1,005 / 18,333 | 15,382 / 1,008 / 16,390 | 12,352 / 8,512 | 37.008 / 28.230 | 两边通过 | +| T5 | 4 | 4 | 16,706 / 561 / 17,267 | 14,688 / 553 / 15,241 | 12,032 / 10,560 | 29.131 / 27.308 | 两边通过 | +| 合计 | 21 | 22 | **90,638 / 4,769 / 95,407** | **84,395 / 4,671 / 89,066** | **62,272 / 57,728** | **195.405 / 202.715** | 各 5/5 | + +本样本 app-server 总 Token 少 6.65%;Native 累计 Runtime turn 秒少约 3.61%,主要来自 T3。样本量、缓存、时序、Memory 混杂使这些数字不支持普遍优劣,也不支持切换默认。 + +| Run | tool sequence | +行 / -行 | Reviewer quality | +|---|---|---|---:| +| T1-N / T1-A | Read → Write → Bash | 12/1;12/1 | 4 / 4 | +| T2-N / T2-A | Read → Write → Bash | 51/0;53/0 | 4 / 4 | +| T3-N | Read → Write → Bash → Bash | 35/1 | 4 | +| T3-A | Read → Write → Bash → Bash → Bash | 34/1 | 4 | +| T4-N / T4-A | Read → Read → Write → Write → Bash | 23/3;29/4 | 4 / 4 | +| T5-N / T5-A | Read → Write → Bash | 3/2;3/2 | 4 / 4 | + +质量分为 Codex **非盲评**:4=good,完成要求、修改范围合理、独立验收通过。T1 两边验证方式不同但语义符合;T2 两边 tests 杀死三种缺陷;T3 都正确保持错误 identity 与 backoff;T4 都覆盖 transition;T5 都引入内部 helper、延续 fixture 紧凑风格。没有足够证据给一边更高分。没有把 worker 自报的测试当独立验收。 + +每 run 的 sessionId、traceId、promptHash、Git commit、原始 task prompt、mode、完整字段、候选 diff 和独立验证结果在 observed ledger(`benchmark-observed.json`)。该 ledger 与其余原始证据不进入仓库;归档位置和 SHA-256 见 [evidence manifest](evidence/manifest.json)。T1/T3/T4/T5 的 testsPassed=1 指一个独立复合断言检查脚本,不是只有一个断言;T2 是 TAP 单测数量。 + +## 补充组:配置对齐但执行失败 + +新的 fixture 基线 `090b0afc708c6e40ba94b7cdd9f03f86406e1fc1`,增加 `.zcode/config.json`:`{"features":{"memory":false}}`。组内继续固定同 cwd / commit / prompt / selection / policy / tools。 + +两边 System body 哈希相同、4 工具 schema 相同、effort=low:可确认 Memory 是 System 差异来源之一。T1-N 首个 Provider input=3,333,system_prompt=6,910 字符;路径与观察组不同,不能直接逐字符比较到观察组。 + +之后遭遇 `ECONNRESET` / `ENOTFOUND open.bigmodel.cn`。N 的 T1 在 300 秒 supervisor deadline 强杀;其余返回 `turn.failed` 或非零进程退出。其他 run 的初始 request 多次重试到 attempt=11;没有成功的配对任务终态,完整 Token usage 缺失记 null。T1-N 留下的修改独立验收可通过,但执行没有完成,所以仍排除性能组;T5 未修改代码时基线语义测试能通过,这不满足“完成重构”的验收,已补上 artifact produced 检查。 + +**这组不合并到成功统计,不据此推断编码质量低、app-server 较贵或 Native 较慢。** 尚不确定断连来源是本机 DNS/proxy、网络还是远端;无需凭错误猜测服务故障。详见 controlled-failed ledger(`benchmark-controlled-failed.json`,位置同上 manifest)。 + +## Reasoning 配对探针 + +相同简单任务:求小于 30 的素数总和,禁止工具,两边输出 129。每 run 一个 model request。 + +| 档位 | N input / output / total | A input / output / total | +|---|---|---| +| low | 3,663 / 25 / 3,688 | 3,157 / 24 / 3,181 | +| high | 3,663 / 89 / 3,752 | 3,157 / 89 / 3,246 | +| max | 3,663 / 89 / 3,752 | 3,157 / 89 / 3,246 | + +body 均为相同 Flash,`thinking={type:"enabled"}`;effort 对应档位。low/high/max 的 adapter 映射等价 **CONFIRMED — RUNTIME**。高档 output 在小探针增加,不证明大型编码质量收益。Provider reasoningTokens=0 不代表没有隐藏推理,可能只是未分列。目录默认 max、私有 configured default low、请求值、实际 body 值要分开。未指定 default 的真实用户会话行为 **UNKNOWN**。 + +## Permission / Memory / 成本拆分 + +| 成本分量 | 本次证据 | +|---|---| +| 初始化 context | 观察组 Native system 多 2,159 字符,首 request 多 504 input tokens;Memory 对齐可使 system 哈希相同 | +| model request 数 | T3-A 比 N 多一次 Bash 与一次模型请求,这一任务 A 更贵;其余任务相同请求数且 A 较少 Token | +| reasoning | 六次实际 body 映射相同;高档 tiny probe output 更多;编码主组 low | +| replanning / permissions | build CLI 拒绝 Write 后报告 DENIED;app-server 指定文件审批允许后 DONE;均两请求。没有证实必然额外 replanning | +| Workflow | 主组工具禁止;完整 Workflow settle 实验 NOT RUN,源码有生命周期差异 | +| Memory extraction | 主组 N extraction 关 / A Memory 关;开启 tiny probe各一 main request,无抽取记录;不能外推没有后台调用 | +| Subagents | 主组禁 Agent;A childSessionIds=[];N 没有可查询实时子会话证据,不能填伪造 0 | +| retries | 成功组观测到的 request attempt>1 数为 0;失败组大量网络重试,足以淹没 executor 耗时差异 | +| session resume | A/B 证明有效恢复历史与指定选择;错误 Provider materialization 初始 smoke 确实阻止 CLI执行 | +| usage口径 | main turn、辅助请求、累计 context-cache、session/usage 产品计量不同;详见 Handoff 文档 | + +“app-server 更贵”在本次观察组总体不成立。Native 可能在特定任务更省请求,或复用官方 Host 减少配置偏差,这是 **HYPOTHESIS**;更高代码质量仍 **UNKNOWN**。没有金额、CPU、RSS 数据,不能把 Token 总量说成全部资源成本。 + +## 完整性、修正与复核 + +所有运行均记录要求的指标字段。未取得的字段保持 null:CPU、峰值内存;N 子会话数量;workflowCount(没有通用可靠活动计数)。toolCallCount 按 toolCallId 去重,不计 streaming / updated 事件条数;retryCount 计已观测 attempt>1 的 request_started,不能等同业务重试 feature。主组 permission events=0、A interaction=0。finalStatus、executionCompleted、acceptancePassed 分开保存。 + +最初验收 harness 有两个输出解释错误:对 `git status` 使用 trim 导致首条路径截断;Node 24 默认非 TAP reporter 导致 test count 未解析。已改为 trimEnd 与显式 TAP,并从每个已保存 diff/untracked 重建候选、重新独立验收。初始结果只保留在 temp 原始 ledger;交付以复核结果为准。T5 后来增加“产生修改”条件,避免失败运行的基线测试被误当成功。 + +wallTime 不同口径:N 为 spawn→exit;A 为 session/send→terminal + 1.5 秒观察,不含预创建时间。两者不能直接比较;上述表用 Runtime duration。完整源研究 worker 的 102 个 model requests、6,178,317 tokens 另列为研究成本,不掺入 Benchmark。Codex 主会话 Token 未取得:当前宿主未提供本次调用统计。 diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_RUNTIME_CALLCHAIN.md b/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_RUNTIME_CALLCHAIN.md new file mode 100644 index 0000000..1537fc7 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_RUNTIME_CALLCHAIN.md @@ -0,0 +1,76 @@ +# ZCode Runtime 调用链 + +> Status: RESEARCH +> Date: 2026-10-05 +> 基于固定官方提交的源码阅读结果。索引见 [README.md](README.md)。 + +来源固定为官方提交 `29628c9acdb81b703bbd4080c207a0e7ce5e276e`。以下源码链接均指该提交;安装版行为另见 [总报告](ZCODE_CLI_VS_APPSERVER.md)。 + +## 实际汇合点 + +```mermaid +flowchart TD + CLI[run.ts --prompt / --target] --> RP[runPrompt] + RP --> HP[Provider Registry / Headless Ports / Config] + HP --> APP[createZCodeApp] + APP --> SUB[submitPrompt] + SUB --> RPT[InputFacade.runPromptTurn] + RPT --> TURN[AgentRuntime.executeTurn] + AS[run.ts app-server] --> ENTRY[runZCodeProtocolAgent] + ENTRY --> CTX[SQLite / Provider / Protocol Server Context] + CTX --> CREATE[session/create or resume] + CREATE --> APP2[materializeSessionRecord / createZCodeApp] + APP2 --> SEND[session/send] + SEND --> BG[runPromptTurnInBackground] + BG --> INPUT[InputFacade.sendInput / admitPrompt] + INPUT --> TURN + TURN --> LOOP[runRegularTurnLoop] + LOOP --> STEP[runModelBackedTurnStep] + STEP --> MODEL[runModelTextRequest] + MODEL --> RUNNER[model runner / executor.prepareRequest] + RUNNER --> SDK[generateText or streamText] +``` + +**CONFIRMED — SOURCE:共同 Runtime/Core。** 一个易误读细节:CLI `submitPrompt → runPromptTurn` 直接进入 `executeTurn`;协议 `sendInput → admitPrompt → executeTurn`。不能把两入口都写成 `admitPrompt`。 + +## 逐级源码位置 + +| 层 | Native | app-server / 共享 | +|---|---|---| +| CLI 入口 | [run.ts L498](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/cli/src/run.ts#L498) prompt 分派 | 同文件 [L236](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/cli/src/run.ts#L236) `runZCodeProtocolCommand`、L537 server 分派 | +| 进程 bootstrap | [prompt-command.ts L194](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/cli/src/prompt-command.ts#L194) Provider / App 配置 | [zcode-protocol-entrypoint.ts L82](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/zcode-protocol-entrypoint.ts#L82),SQLite L139、Provider L156、App 工厂 L249 | +| 创建 Runtime | [create-app.ts L726](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/app/create-app.ts#L726) | 相同 `new AgentRuntime` | +| 输入提交 | [input-facade.ts L369](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/app/input-facade.ts#L369) submitPrompt → L99 runPromptTurn | 同文件 [L193](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/app/input-facade.ts#L193) sendInput → admitPrompt | +| 协议 admission | 不经过此 Host 分派 | [server-operations.ts L1918](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/zcode-protocol/server-operations.ts#L1918) sendPrompt、L2360 background、L2410 sendInput | +| admitPrompt | CLI 本链不经过 | [prompt-admission.ts L20](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/core/src/runtime/methods/prompt-admission.ts#L20) → executeTurn | +| 共享 turn | [turn.ts L70](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/core/src/runtime/methods/turn.ts#L70) executeTurn | L614 regular loop;[turn-loop.ts L43](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/core/src/runtime/methods/turn-loop.ts#L43)、L205 model step | +| 共享 model step | [turn-model-step.ts L192](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/core/src/runtime/methods/turn-model-step.ts#L192) | model request、L235 runModelTextRequest | +| 模型执行 | [model.ts L36](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/core/src/runtime/methods/model.ts#L36) | messages/tools/signal/options L119;generate L142 / stream L228 | +| Provider/SDK | [adapters/model/model.ts L72](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/adapters/src/model/model.ts#L72) | executor.prepareRequest;[runner-runtime.ts L50](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/adapters/src/model/runner-runtime.ts#L50) Vercel AI SDK | + +## Runtime materialization 的分叉 + +Native [prompt-command.ts L212–243](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/cli/src/prompt-command.ts#L212) 注入 browser、Provider Registry、default model、runtime auth headers、headless permission broker、runtimeConfig。公开源码设置 Workflow 为 `options.enableWorkflow===true`、Memory extraction 为 `options.memoryBench===true`、streaming=on。 + +协议 [server-operations.ts L3279](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/zcode-protocol/server-operations.ts#L3279) 的 materialize/createRecord 在 L3345+ 应用 startup preferences、session allow/disallowlist、Memory enabled override、协议 permission/browser broker、automation / subagent 端口。请求 `session/requestRuntimePreferences` 是 Host 往返;未回答或 answered preferences 不同,会改变 materialization。 + +**CONFIRMED — SOURCE:** Bridge 基线 [model-settings.ts](../../src/runtime/model-settings.ts) L339 固定关闭 Memory、native search enhancements、自动回答 AskUserQuestion。实际执行适配器也有同样回调,见本仓库 [zcode-app-server-adapter.ts](../../src/adapters/zcode-app-server-adapter.ts) L759 一带。权威是上述本机固定 HEAD 源码。 + +两边的 Provider Registry 启动来源共享;account access 还依赖 credential identity、snapshot revision、runtime headers。Browser、MCP、permissions 是能力端口,不是仅由注册工具名字决定的能力。 + +## Context与能力注入 + +- [core/context/builder.ts L149](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/core/src/context/builder.ts#L149):Memory 注入;L177 skills,L193 workspace instructions;前面依次组织 stable identity / dynamic behavior / session guidance。系统、meta user、普通 history 必须分别比较。 +- [adapters/context/index.ts](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/adapters/src/context/index.ts):用户 `~/.zcode/AGENTS.md` 与祖先 workspace instructions 发现;同 cwd 也可能受用户目录和 storage/config 环境影响。 +- [server-operations.ts L3348](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/zcode-protocol/server-operations.ts#L3348):工具边界、startup preferences、MCP、能力 broker;工具注册与 permission policy 分开。 +- [subagent/runner.ts L147](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/core/src/subagent/runner.ts#L147):背景运行条件;L214 inactivity timeout,L653 autoBackground 配置。foreground/background 模型应查子会话实际 selection,不应从父会话模型推定。 +- [project-memory-extraction.ts L32](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/core/src/runtime/helpers/project-memory-extraction.ts#L32):disabled / memory root 等门槛,L115 extraction operation,L156 memory agent loop。**可能**产生额外模型请求;并非每个 turn 必然抽取。 +- [headless-workflow.ts L31](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/cli/src/headless-workflow.ts#L31):headless broker,仅特定 Workflow 工具自动允许;普通审批在 build 下可拒绝。 + +## 输出与结束判定 + +CLI 使用 Runtime Event subscriber 输出结构化 stream-json,最终 `result` 与进程退出构成 batch 执行的可观察边界。不是 terminal scraping。`runPrompt` 提交后还调用 [headless-workflow.ts L334](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/cli/src/headless-workflow.ts#L334) settle:观察到 Workflow activity 后等待背景任务与 active/queued turn work,无内部固定总超时,应由外层 supervisor 约束。Memory bench 另行 drain。 + +协议 `session/send` ACK 仅表明 admission;terminal 从 session/event 读取。Bridge 现有适配器 L804–831 按 session/turn/seq 过滤,在 `turn.completed` resolve 当前 prompt;L580+ checkpoint 后执行 cleanup。其资源清理已有设计,不能把现状描述成“没有进程控制”。未发现与 CLI 相同的 Workflow settle 收口;这支持背景能力开启时存在风险,尚未复现漏收完整 Workflow 的实际故障。 + +Core shared、Host distinct 的结论并不推出默认工具、prompt、model options、资源成本或 task terminal 一定相同。 diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_SESSION_HANDOFF.md b/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_SESSION_HANDOFF.md new file mode 100644 index 0000000..56ad266 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/ZCODE_SESSION_HANDOFF.md @@ -0,0 +1,82 @@ +# Session Handoff 审计 + +> Status: RESEARCH +> Date: 2026-10-05 +> 研究结论,不代表当前实现。索引见 [README.md](README.md)。 + +**结论:双向顺序恢复可行;“全部 Runtime 状态无损”未证明;正在运行的跨 Host attach/control 不应启用。** 所有实验只使用专用私有 SQLite、独立 fixture 和本次生成的 Session。模型均为 GLM-5.3-Flash / low。 + +## A — CLI → app-server + +Session:`sess_861d65fd-78de-4076-9dc8-bd78d813df6f`。 + +CLI Read `handoff.txt`,记住第一行 `ALPHA_731`,正常退出。另起 app-server:`session/list` 可见;未 resume 的 `session/read` 报 `-32004 Session is not active`;`session/resume` 成功。续作能回忆 marker,并直接 Edit 把 `stage=one` 改为 `stage=two`,没有再次 Read。文件最终为 `ALPHA_731\nstage=two\n`。 + +| 检查项 | 实测结果 / 边界 | +|---|---| +| History | 3 → 6 条持久化消息,前 3 条内容摘要一致;本探针的历史保留 **CONFIRMED — RUNTIME** | +| Model / reasoning | 恢复快照保持指定 Provider、Flash、low | +| Mode | `settings.mode.current=yolo`;session 元数据却为 build,不能把 list 字段当当前权限真值 | +| Workspace | 保持原 fixture 路径 / workspace key | +| Tool state | 没有重读就 Edit 成功;与源码 read-file-state hydration 一致。只证明所测 Read/Edit 状态 | +| Usage | 累计 context-cache input 38,378 → 70,411;usage 查询 request count 2 → 4。跨轮延续成立,但 accounting 数值不是所有网络请求简单相加 | +| Memory | 不是已有 project memory 内容的恢复实验;此项 **UNKNOWN** | +| 续作质量 | marker 正确、Edit 正确、终态成功。只证明此小型续作 | + +## B — app-server → CLI + +Session:`sess_49430bee-2fd3-4c4a-b5da-31f0281cc412`。 + +app-server 完成 Read `BETA_842` 的首轮;对空闲会话 `stop` 返回 `{}`,`close` 返回 `{closed:true}`。关闭后 list 仍可见,进程退出。CLI 用同一 ID `--resume` 并明确传 `--mode yolo`,成功回忆 marker,直接 Edit 完成 stage=two。第三个 app-server 再 resume 验证:历史 3 → 6、前 3 条摘要保持,模型与 low 保持,当前 yolo,workspace 保持;累计 input 31,758 → 70,411,request count 2 → 4。 + +**CONFIRMED — RUNTIME:session/close 不删除这次已持久化的历史。** 此处 yolo 有显式参数覆盖,不能声称不传 mode 时也一定恢复相同权限。session title 从 first_input 变为 generated;标题等辅助状态可能另外更新,不在“主 turn 的历史内容一致”保证之内。 + +## C — 运行中的 CLI → 另一 app-server + +Session:`sess_d168cbe0-5ad0-4d3f-b7e7-88680013ddaf`。CLI 在 Bash 中执行本次创建的 15 秒 helper;启动标记证明工具仍在运行。 + +| 操作 | 结果 | +|---|---| +| session/list | 可发现持久化记录,但 status=idle,与运行实况不同 | +| session/read | `-32004 Session is not active` | +| session/stop | 同样错误;无法停止另一进程持有的 Runtime | +| session/resume | **NOT RUN — safety stop**:源码显示冷 resume 会 materialize 新 Runtime;未发现跨进程执行租约。用户原任务要求出现数据风险即停止 | +| CLI 自身完成 | exit 0、完成标记存在、工具正常结束 | +| ownership error / global lock | 没有实测到,因为没有执行危险的并发 resume;**UNKNOWN** | +| history corruption / duplicate tool | 未制造;不能写成已复现或必然发生 | + +**CONFIRMED — RUNTIME:持久化可见 ≠ 活动 Runtime 可控制。SUPPORTED:第二个 Host 的冷 resume 应视为新 Runtime 的恢复,不能当成连接现有 PID 的 attach。** 不支持把未执行的 concurrent mutation 误记成一次“安全 attach 失败”的完整实验。 + +## 持久化、恢复、所有权 + +官方 [server.ts L133 / L260](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/zcode-protocol/server.ts#L133) 的 session registry 是进程内 Map;[sendPrompt L1929](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/zcode-protocol/server-operations.ts#L1929) 检查该 record 的 activeAbortController。SQLite 的 WAL / busy locking 保护存储访问,不等于跨 Runtime 的会话执行锁。审读 bootstrap/session-store 及相关检索未发现通用 session-owner 租约;这是范围内的 absence evidence,标 **SUPPORTED**,不是对所有未审代码“绝无锁”的证明。 + +[resume L1411](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/zcode-protocol/server-operations.ts#L1411) 读取持久化消息、推导 mode、创建 record,再 app.resume;model 从持久化 selection entry 恢复。[core resume.ts L137](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/core/src/runtime/methods/resume.ts#L137) 恢复 ReadFileState 与 execution state。history hydrator 在 [L188](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/core/src/agent/session-history-hydrator.ts#L188) 把未完成 tool 转为 interrupted 结果;若另一进程仍真实执行该 tool,就存在状态理解分歧的风险,不是安全接管协议。 + +## close、stop、EOF 的区别 + +| 信号 / 方法 | 源码语义与本次证据 | +|---|---| +| `session/stop` | [L2592](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/zcode-protocol/server-operations.ts#L2592) abort 当前回合,可暂停 active goal;ACK 不证明 tool tree 已结束 | +| `session/close` | [L2716](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/zcode-protocol/server-operations.ts#L2716) app.close、移除内存 record/event store;不等于删除 SQLite 历史;B 已证实 | +| `App.close` | [session-facade.ts L264](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/app/session-facade.ts#L264) drain memory、停 Workflow、清资源;drain 有等待边界 | +| stdin EOF | transport disconnect 与进程 shutdown 不同于一条 `session/close`;本探针 EOF 后进程退出且历史可恢复 | +| 强杀进程树 | 操作系统取消,不保证正常 flush 或干净持久化;须验证后续 resume,而非自动假定可恢复 | + +另一个边界:最初 smoke 对刚 create、未 admission prompt 的“immediate”空会话,进程退出后 resume 报 Session not found。完成首轮后才观察到可恢复持久化。不能只凭 `session/create` 的返回 ID 保证已有 durable history。 + +## Usage 不应宣称完全无损 + +A 的 `session/usage` totalTokens 从 19,268 到 19,492;同一会话 runtime context-cache 的累计 input 从 38,378 到 70,411。官方 [getTaskTokenUsage L2908](https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/apps/zcode-cli/packages/bootstrap/src/zcode-protocol/server-operations.ts#L2908) 使用 usage store task-accounting,并返回 inputBaselineBySource。不同字段有不同口径。恢复 request count 连续不意味着所有历史 Provider Token、缓存字段、标题/后台请求都可由一个数重建。 + +建议 Run Ledger 保存每个已完成 request/turn 的 Provider usage,保留 querySource、logicalCallId、requestId、attempt,标完整性;session/usage 作为独立产品计量来源,禁止混加或重复计费。未完成受网络影响的请求记 UNKNOWN,不补 0。 + +## 建议的顺序交接门槛 + +1. Run Manager 为 workspace + session 持有唯一执行租约。 +2. 当前输入回合与允许的背景活动已收口,记录 final result、usage 完整性和持久化摘要。 +3. 旧 Host 正常释放、其工具进程已清理,确认旧进程退出;不能只等 stop ACK。 +4. 新 Host 使用相同存储、身份、workspace,显式配置 policy;resume 后核对历史、model、reasoning、mode,再提交新输入。 +5. 任一步未知就保留 occupied / recovery_required,不开启并行写入。 + +该流程是架构建议,**没有在本次修改生产 Bridge 实现**。跨机器 handoff、后台 subagent/Workflow 完整恢复、外部 Memory 对齐均仍不确定。 diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/evidence/manifest.json b/docs/research/native-cli-vs-appserver-2026-10-05/evidence/manifest.json new file mode 100644 index 0000000..24c386f --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/evidence/manifest.json @@ -0,0 +1,43 @@ +{ + "date": "2026-10-05", + "bridgeCommit": "2596759198fa826c2b7ac0478c5682da996e9727", + "officialCommit": "29628c9acdb81b703bbd4080c207a0e7ce5e276e", + "runtimeVersion": "0.16.9", + "runtimePath": "C:/Users/Sandy/AppData/Local/Programs/ZCode/resources/glm/zcode.cjs", + "runtimeSHA256": "fad4c35c4c36ec210d8a06d3fa0e77de23c8545e2eb6ff90aea1eb38d1e6275f", + "nodeVersion": "v24.16.0", + "model": "GLM-5.3-Flash", + "provider": "account:bigmodel-individual-coding-plan", + "storageScope": "own private SQLite and fixtures; credential/config copies excluded", + "exportPolicy": "allowlisted runtime fields; request/header/hidden reasoning/raw prompt logs excluded; benchmark task prompts and fixture diffs intentionally included", + "evidenceArchive": { + "reason": "Raw ledgers are large; only this manifest is versioned. The ledgers themselves stay in the local archive.", + "location": "C:/Users/Sandy/.codex/archived_docs/codex-zcode-bridge/2026-10-05/evidence/native-cli-vs-appserver-20261005", + "locationNote": "The archived_docs root is documented in docs/README.md.", + "files": [ + { "name": "benchmark-controlled-failed.json", "bytes": 950188, "sha256": "126cdb8d3bcc5e676d8961b682efa66130fe95bef95fd0571d4ece415d3a5ff1" }, + { "name": "benchmark-observed.json", "bytes": 527901, "sha256": "b541219006b8e7d16710dcc55bf5b48a0aaff85d2206c2e3e291ec4058c89907" }, + { "name": "handoff.json", "bytes": 88971, "sha256": "c9162516eff2b90b2cb747f33f1985a133280b427a1e35a6c9e0f286f561d85d" }, + { "name": "job-object.json", "bytes": 243, "sha256": "a23c3a0e0f61a7963e7de8ed89fb399550001e6e0aab4a51e8df7f409f171c6e" }, + { "name": "policy.json", "bytes": 59072, "sha256": "f42d748b4d325aa7d664265b00e0c16d8fb29d9b30602bfe10baa64f2d9683f1" }, + { "name": "process-control.json", "bytes": 10332, "sha256": "c6ae462091f3524cef709a1ba37b810cffe9ae1cadce1146a5f64defd5bc8a3a" }, + { "name": "reasoning.json", "bytes": 78306, "sha256": "b2d2be58f79e80c9aba73d7d7aedffcf7180682fc85654fc2b0812e8ba1c2f86" }, + { "name": "smoke-isolated.json", "bytes": 16864, "sha256": "9a8e03c25f0e5890f19a2d68b688d7349cd73fbf1e45db2e039a50ca1a7d2b29" }, + { "name": "source-task-summary.json", "bytes": 5842, "sha256": "b3cd2939c040886390fcff89e6808e8cb02130c2bae2e26dd6b2d413f9e26fd9" } + ] + }, + "observedRuns": 10, + "observedCompleted": 10, + "controlledRuns": 10, + "controlledCompleted": 0, + "controlledDisposition": "network failures/timeout; excluded from coding comparison", + "notRun": [ + "concurrent mutating session resume", + "real ZCode Job Object containment", + "Windows Console Ctrl+C/Ctrl+Break", + "CPU/RAM accounting", + "Desktop runtime preferences", + "full background workflow/subagent performance", + "existing project memory handoff" + ] +} diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/experiments/README.md b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/README.md new file mode 100644 index 0000000..26eba31 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/README.md @@ -0,0 +1,57 @@ +# 实验与证据复核 + +这些脚本为 2026-10-05 研究专用,不是生产 Executor。会调用真实模型并产生费用/配额用量;只有模型执行脚本需要授权账户配置。本次用户已授权 GLM-5.3-Flash。脚本内的固定 Windows 路径用于重现本机环境;迁移时先适配,不应直接指向真实项目执行 reset/clean。 + +## 证据 + +进入仓库的只有 `../evidence/manifest.json`:它记录版本、范围、完成与 NOT RUN 清单,以及各证据文件的归档位置与 SHA-256。其余原始 ledger 不进入仓库,归档在 `C:/Users/Sandy/.codex/archived_docs/codex-zcode-bridge/2026-10-05/evidence/native-cli-vs-appserver-20261005`;`archived_docs` 根位置见 `docs/README.md`。下表描述这些 ledger 的内容: + +| 文件 | 内容 | +|---|---| +| manifest.json | 版本、范围、完成与NOT RUN清单 | +| benchmark-observed.json | 10个成功运行:每run指标、真实task prompt及hash、候选diff、复核验收、规范事件 | +| benchmark-controlled-failed.json | Memory同时关闭的10次失败运行;完整usage未知、明确排除性能比较 | +| reasoning.json | low/high/max六次request body安全digest | +| policy.json | build权限拒绝/允许、Memory开启小探针 | +| handoff.json | A/B顺序续接,C只读发现/控制失败与危险resume NOT RUN | +| process-control.json | 两种实际ZCode执行器取消与自有工具helper存活检查 | +| job-object.json | 自有Node helper树的最终JobObject验证,不是ZCode集成 | +| smoke-isolated.json | 临时身份/config materialization后的顺序smoke | +| source-task-summary.json | 源码研究worker的会话选择与terminal usage | + +临时完整allowlist ledger、fixture仓库、官方clone及session SQLite保留在 `C:/Users/Sandy/.codex/tmp/zcode-executor-audit-20261005`。原始model-I/O可能轮转,且含敏感headers/隐藏推理,**不复制入仓库、不以截图展示**;仅本次自有Session输出哈希、长度和usage。研究结束已移除临时配置、加密凭据副本和专用rollout中最后3个自有raw model-I/O文件;真实用户文件未改。 + +## 脚本职责 + +- `probe.mjs`:读取Bridge resolver的本机路径,在临时副本配置Flash/low,按官方cipher建立临时credential store;stdio RPC、事件白名单、Native stream-json与进程终止。 +- `run-source.mjs`:独立research worktree中的真实源码审计worker。不能替代主审。 +- `run-benchmark.mjs`:新建fixture,只在该专用Git仓库reset/clean;按5类任务交替执行器顺序。 +- `review-benchmark.mjs`:不调用模型;重建候选diff/新文件并重新运行独立验收,生成指标。 +- `run-handoff.mjs`:A/B顺序恢复,C运行中只读探针;不会做并发mutating resume。 +- `run-reasoning.mjs`:相同无工具任务的三个effort档位配对。 +- `run-policy.mjs`:build权限与Memory开启探针;仅本次指定文件的Write允许。 +- `run-process-control.mjs`:执行本次自有60秒Node helper,分别验证taskkill与session/stop。 +- `job-object.ps1`:P/Invoke helper树,kill-on-job-close,finally清理自有进程。 +- `summarize-model-io.mjs`:精确Session ID读自有文件;仅输出安全digest,不输出headers/原始content。 +- `export-evidence.mjs`:从已白名单ledger导出,移除敏感字段和高频streaming事件。 + +## 本机复核命令 + +先 `npm ci --ignore-scripts` 与 `npm run build:core`;确认resolver只读取正确本机安装与账户身份。官方clone必须固定到上述commit,路径为临时目录下 `official`,因为probe复用其credential cipher。目录 `.zcode/config.json` 使用正式项目配置发现规则。 + +```powershell +# 只重放已有观察组候选,不重新调用模型 +node docs/research/experiments/review-benchmark.mjs + +# 只重放失败补充组 +$env:AUDIT_BENCHMARK_PHASE = 'controlled' +node docs/research/experiments/review-benchmark.mjs +Remove-Item Env:AUDIT_BENCHMARK_PHASE + +# 导出白名单证据 +node docs/research/experiments/export-evidence.mjs +``` + +重新执行 `run-benchmark.mjs` 会调用真实模型并覆盖同名临时ledger;应另建日期目录并保留旧证据。`AUDIT_ROOT` 只用于指向**已经准备好的专用审计目录**,不可指向项目根、用户Home或共享数据库。更换日期目录需复制/固定official clone并检查resolver配置。`AUDIT_BENCHMARK_PHASE=controlled` 新建 Memory-off fixture,而非更改真实用户配置。 + +独立验证的known limitations:只在Windows/Node24、安装版0.16.9、一个账户、一次样本上运行;T2是mutation testing,其余为复合断言脚本;Reviewer质量分非盲评。成功组Memory混杂,补充组网络失败。字段null表示未知,不表示0。 diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/experiments/export-evidence.mjs b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/export-evidence.mjs new file mode 100644 index 0000000..5b8c8d2 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/export-evidence.mjs @@ -0,0 +1,26 @@ +// Export only already-allowlisted, audit-owned evidence; never raw model I/O. +import {readFileSync,writeFileSync,mkdirSync} from 'node:fs'; +import path from 'node:path'; +import {fileURLToPath} from 'node:url'; +import {createHash} from 'node:crypto'; +const base=process.env.AUDIT_ROOT||'C:/Users/Sandy/.codex/tmp/zcode-executor-audit-20261005'; +const out=fileURLToPath(new URL('../evidence/',import.meta.url));mkdirSync(out,{recursive:true}); +const read=n=>JSON.parse(readFileSync(path.join(base,n+'.json'),'utf8')); +const write=(n,x)=>writeFileSync(path.join(out,n+'.json'),JSON.stringify(x,null,2)+'\n'); +const important=e=>e.type!=='model.streaming'&&(e.requestType||e.contextUsageBreakdown||e.tool||e.type?.startsWith('turn.')||e.type?.startsWith('permission.')||e.type==='session.resumed'); +function compact(x){ + if(Array.isArray(x))return x.map(compact); + if(!x||typeof x!=='object')return x; + const result={};for(const [k,v]of Object.entries(x)){ + if(['requestHeaders','responseHeaders','apiKey','accessToken','refreshToken','credentials','requestAuth','initialAcceptance','killResult'].includes(k))continue; + if(k==='events'){result.events=Array.isArray(v)?v.filter(important).map(compact):v;continue;} + result[k]=compact(v); + }return result; +} +for(const [from,to]of [['benchmark-reviewed','benchmark-observed'],['benchmark-controlled-reviewed','benchmark-controlled-failed'],['handoff','handoff'],['reasoning','reasoning'],['policy','policy'],['process-control','process-control'],['job-object','job-object'],['smoke-isolated','smoke-isolated']])write(to,compact(read(from))); +const source=read('source-task'); +write('source-task-summary',{sessionId:source.created?.session?.sessionId||source.sid,created:compact(source.created),terminal:compact(source.run?.terminal),error:source.error||null,independentReview:'Source paths and runtime claims cross-checked by Codex; worker drafts not accepted as authority; runtime ownership absence narrowed to SUPPORTED.'}); +const runtimePath='C:/Users/Sandy/AppData/Local/Programs/ZCode/resources/glm/zcode.cjs'; +const runtimeSHA256=createHash('sha256').update(readFileSync(runtimePath)).digest('hex'); +write('manifest',{date:'2026-10-05',bridgeCommit:'2596759198fa826c2b7ac0478c5682da996e9727',officialCommit:'29628c9acdb81b703bbd4080c207a0e7ce5e276e',runtimeVersion:'0.16.9',runtimePath,runtimeSHA256,nodeVersion:process.version,model:'GLM-5.3-Flash',provider:'account:bigmodel-individual-coding-plan',storageScope:'own private SQLite and fixtures; credential/config copies excluded',exportPolicy:'allowlisted runtime fields; request/header/hidden reasoning/raw prompt logs excluded; benchmark task prompts and fixture diffs intentionally included',observedRuns:10,observedCompleted:10,controlledRuns:10,controlledCompleted:0,controlledDisposition:'network failures/timeout; excluded from coding comparison',notRun:['concurrent mutating session resume','real ZCode Job Object containment','Windows Console Ctrl+C/Ctrl+Break','CPU/RAM accounting','Desktop runtime preferences','full background workflow/subagent performance','existing project memory handoff']}); +console.log('Exported audit-owned sanitized evidence.'); diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/experiments/job-object.ps1 b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/job-object.ps1 new file mode 100644 index 0000000..f191c2d --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/job-object.ps1 @@ -0,0 +1,60 @@ +# Research-only Windows process-tree containment probe, using owned helper processes. +$ErrorActionPreference = 'Stop' +$auditBase = 'C:\Users\Sandy\.codex\tmp\zcode-executor-audit-20261005' +Add-Type -TypeDefinition @' +using System; using System.Runtime.InteropServices; +public static class AuditJob { + [StructLayout(LayoutKind.Sequential)] public struct Basic { public long PerProcessUserTimeLimit,PerJobUserTimeLimit; public uint LimitFlags; public UIntPtr MinimumWorkingSetSize,MaximumWorkingSetSize; public uint ActiveProcessLimit; public UIntPtr Affinity; public uint PriorityClass,SchedulingClass; } + [StructLayout(LayoutKind.Sequential)] public struct IO { public ulong ReadOperationCount,WriteOperationCount,OtherOperationCount,ReadTransferCount,WriteTransferCount,OtherTransferCount; } + [StructLayout(LayoutKind.Sequential)] public struct Extended { public Basic BasicLimitInformation; public IO IoInfo; public UIntPtr ProcessMemoryLimit,JobMemoryLimit,PeakProcessMemoryUsed,PeakJobMemoryUsed; } + [DllImport("kernel32.dll",CharSet=CharSet.Unicode,SetLastError=true)] public static extern IntPtr CreateJobObject(IntPtr attrs,string name); + [DllImport("kernel32.dll",SetLastError=true)] public static extern bool AssignProcessToJobObject(IntPtr job,IntPtr process); + [DllImport("kernel32.dll",SetLastError=true)] public static extern bool SetInformationJobObject(IntPtr job,int cls,ref Extended data,uint length); + [DllImport("kernel32.dll")] public static extern bool CloseHandle(IntPtr handle); +} +'@ +$helper = Join-Path $auditBase 'job-parent.cjs' +$permit = Join-Path $auditBase 'job-permit.txt' +$pidFile = Join-Path $auditBase 'job-child-pid.txt' +foreach ($f in @($permit,$pidFile)) { if (Test-Path -LiteralPath $f) { Remove-Item -LiteralPath $f } } +@' +const fs=require('node:fs'),cp=require('node:child_process'); +const timer=setInterval(()=>{if(fs.existsSync(process.argv[2])){clearInterval(timer);const child=cp.spawn(process.execPath,['-e','setInterval(()=>{},1000)'],{windowsHide:true,stdio:'ignore'});fs.writeFileSync(process.argv[3],String(child.pid));}},50); +setInterval(()=>{},1000); +'@ | Set-Content -LiteralPath $helper -Encoding utf8 +$job = [AuditJob]::CreateJobObject([IntPtr]::Zero,$null) +if ($job -eq [IntPtr]::Zero) { throw 'CreateJobObject failed' } +$limits = New-Object AuditJob+Extended +$basicLimits = New-Object AuditJob+Basic +$basicLimits.LimitFlags = 0x2000 # KILL_ON_JOB_CLOSE +$limits.BasicLimitInformation = $basicLimits +$size = [Runtime.InteropServices.Marshal]::SizeOf($limits) +if (-not [AuditJob]::SetInformationJobObject($job,9,[ref]$limits,$size)) { throw 'SetInformationJobObject failed' } +$psi = New-Object Diagnostics.ProcessStartInfo +$psi.FileName = 'C:\Program Files\nodejs\node.exe' +$psi.UseShellExecute = $false +$psi.CreateNoWindow = $true +$psi.ArgumentList.Add($helper); $psi.ArgumentList.Add($permit); $psi.ArgumentList.Add($pidFile) +$parent = [Diagnostics.Process]::Start($psi) +$result = @{ killOnJobClose = $true; parentPid = $parent.Id; scope = 'owned Node helper tree, not actual ZCode Job integration' } +try { + $result.assigned = [AuditJob]::AssignProcessToJobObject($job,$parent.Handle) + if (-not $result.assigned) { throw 'AssignProcessToJobObject failed' } + 'go' | Set-Content -LiteralPath $permit + $sw=[Diagnostics.Stopwatch]::StartNew() + while (-not (Test-Path -LiteralPath $pidFile) -and $sw.ElapsedMilliseconds -lt 5000) { [Threading.Thread]::Sleep(50) } + if (-not (Test-Path -LiteralPath $pidFile)) { throw 'helper child not started' } + $ownedChildId=[int](Get-Content -LiteralPath $pidFile) + $result.childPid=$ownedChildId + $result.childAliveBefore=[bool](Get-Process -Id $ownedChildId -ErrorAction SilentlyContinue) + [void][AuditJob]::CloseHandle($job); $job=[IntPtr]::Zero + [void]$parent.WaitForExit(5000) + $result.parentExited=$parent.HasExited + $result.childAliveAfter=[bool](Get-Process -Id $ownedChildId -ErrorAction SilentlyContinue) + $result | ConvertTo-Json | Set-Content -LiteralPath (Join-Path $auditBase 'job-object.json') -Encoding utf8 + $result | ConvertTo-Json +} finally { + if($job -ne [IntPtr]::Zero){[void][AuditJob]::CloseHandle($job)} + if(-not $parent.HasExited){$parent.Kill($true)} + $parent.Dispose() +} diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/experiments/probe.mjs b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/probe.mjs new file mode 100644 index 0000000..0ea6051 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/probe.mjs @@ -0,0 +1,129 @@ +// Research only. No production executor changes. Requires npm run build:core. +import { spawn, execFileSync } from 'node:child_process'; +import { readFileSync, writeFileSync, mkdirSync, copyFileSync } from 'node:fs'; +import path from 'node:path'; +import { fileURLToPath, pathToFileURL } from 'node:url'; +import { createHash } from 'node:crypto'; +import { NodeRuntimeResolver, loadPersistedRuntimeEnvironment } from '../../../dist/src/runtime/resolver.js'; +import { buildAccountProviderPayload, runtimeAuthReply, zcodeDataBaseDir } from '../../../dist/src/runtime/account-provider.js'; +const base = process.env.AUDIT_ROOT || 'C:/Users/Sandy/.codex/tmp/zcode-executor-audit-20261005'; +mkdirSync(base, {recursive:true}); +const realCfg = await new NodeRuntimeResolver().resolve(); +// Private, reversible credential materialization using official encryption. +// Never modify the user's provider configs or credential store. +const privateDir=path.join(base,'private-runtime');mkdirSync(path.join(privateDir,'.zcode','v2'),{recursive:true}); +const cfg={...realCfg,providerBuiltinConfigFile:path.join(privateDir,'zcode-builtin.json'),providerPersonalConfigFile:path.join(privateDir,'.zcode','v2','provider_config.json')}; +copyFileSync(realCfg.providerBuiltinConfigFile,cfg.providerBuiltinConfigFile); +copyFileSync(realCfg.providerPersonalConfigFile,cfg.providerPersonalConfigFile); +const personal=JSON.parse(readFileSync(cfg.providerPersonalConfigFile,'utf8')); +personal.config.defaultModelSelection={providerId:'account:bigmodel-individual-coding-plan',modelId:'GLM-5.3-Flash',options:{reasoningLevel:'low'}}; +writeFileSync(cfg.providerPersonalConfigFile,JSON.stringify(personal),{mode:0o600}); +const cipherModule=path.join(base,'official/apps/zcode-cli/packages/adapters/src/auth/credential-cipher.ts'); +const {createZCodeCredentialCipher}=await import(pathToFileURL(cipherModule).href); +const cipher=createZCodeCredentialCipher(); +const auth=runtimeAuthReply('account:bigmodel-individual-coding-plan',realCfg); +if(!auth.requestAuth?.apiKey)throw new Error('No authorized coding-plan credential available'); +const identity='key-'+createHash('sha256').update(auth.requestAuth.apiKey).digest('hex').slice(0,24); +const providerId='account:bigmodel-individual-coding-plan'; +writeFileSync(path.join(privateDir,'.zcode','v2','credentials.json'),JSON.stringify({ + [`account-provider:${providerId}:identity`]:cipher.encrypt(identity), + [`account-provider:coding-plan:${providerId}:account:${encodeURIComponent(identity)}:api-key`]:cipher.encrypt(auth.requestAuth.apiKey) +}),{mode:0o600}); +const persisted = loadPersistedRuntimeEnvironment(process.env); +const env = {...process.env, ZCODE_BUILTIN_PROVIDER_CONFIG_FILE:cfg.providerBuiltinConfigFile, + ZCODE_PERSONAL_PROVIDER_CONFIG_FILE:cfg.providerPersonalConfigFile, + ZCODE_DATA_BASE_DIR:privateDir,ZCODE_STORAGE_DIR:path.join(privateDir,'.zcode','cli'), + ZCODE_SESSION_DB_PATH:path.join(privateDir,'.zcode','cli','db','sessions.db')}; +delete env.ZCODE_HOME; +const model = {providerId:'account:bigmodel-individual-coding-plan',modelId:'GLM-5.3-Flash',options:{reasoningLevel:'low'}}; +const denyTools=['Agent','AskUserQuestion','CronCreate','CronDelete','CronList','CronUpdate','EnterPlanMode','ExitPlanMode','Skill','TaskOutput','TaskStop','TodoRead','TodoWrite','WebFetch','WebSearch','SendMessage','ReadSessionContext','CreateWorkflow','AmendWorkflow','SaveWorkflow','EvalWorkflowSnippet','ListWorkflowRuns','GetWorkflowRun','ResumeWorkflowRun','ResolveWorkflowQuestion','ListSavedWorkflows','ListModels','mcp__node_repl__js']; +const hash = x => createHash('sha256').update(typeof x==='string'?x:JSON.stringify(x)).digest('hex'); +const save = (name, x) => writeFileSync(path.join(base,name+'.json'),JSON.stringify(x,null,2)+'\n'); +const delay = ms => new Promise(r=>setTimeout(r,ms)); +function snapshot(x) { + return {session:x?.session, runtime:x?.runtime, + model:x?.settings?.model?.current, modelSelection:x?.settings?.model?.selection, + thought:x?.settings?.thoughtLevel, mode:x?.settings?.mode, + workspace:x?.workspace, messageCount:x?.messages?.length, + messageShapes:x?.messages?.map(m=>({id:m.id,role:m.role,kind:m.kind,keys:Object.keys(m)})), + topKeys:Object.keys(x||{}),settingsKeys:Object.keys(x?.settings||{})}; +} +function eventSafe(e) { + const p=e.payload||{}; + const tool=p.toolName||p.name||p.tool?.name; + return {type:e.type,seq:e.seq,sessionId:e.sessionId,turnId:e.turnId, + at:new Date().toISOString(),payloadKeys:Object.keys(p),tool, + toolCallId:p.toolCallId||p.callId,resultType:p.resultType,usage:p.usage, + response:p.response,workflowCount:p.workflowCount, + providerId:p.providerId,modelId:p.modelId,toolCount:p.toolCount,iteration:p.iteration, + requestId:p.requestId,traceId:p.traceId,attempt:p.attempt,modelCall:p.modelCall, + requestType:p.type,toolCallCount:p.toolCallCount,historyRoundCount:p.historyRoundCount, + contextUsageBreakdown:p.contextUsageBreakdown,duration:p.duration, + interruptedToolCount:p.interruptedToolCount,messageCount:p.messageCount,partCount:p.partCount, + kind:p.kind}; +} +export class Server { + constructor(cwd) { + this.events=[];this.calls=[];this.interactions=[];this.pending=new Map();this.id=0;this.errorText=''; + this.child=spawn(cfg.nodeExecutable,[cfg.zcodeEntrypoint,'app-server','--stdio'],{cwd,env,windowsHide:true,stdio:['pipe','pipe','pipe']}); + this.closed=false;let buffer=''; + this.child.stdout.setEncoding('utf8');this.child.stderr.setEncoding('utf8'); + this.child.stderr.on('data',s=>{this.errorText+=s;}); + this.child.stdout.on('data',s=>{buffer+=s;let pos;while((pos=buffer.indexOf('\n'))>=0){const line=buffer.slice(0,pos);buffer=buffer.slice(pos+1);let m;try{m=JSON.parse(line);}catch{continue;}this.handle(m);}}); + this.child.on('close',(code,signal)=>{this.closed=true;this.exit={code,signal};for(const p of this.pending.values()){clearTimeout(p.timer);p.reject(new Error('server exited'));}this.pending.clear();}); + } + write(m){if(!this.closed)this.child.stdin.write(JSON.stringify(m)+'\n');} + handle(m){ + if(m.method==='session/event'){this.events.push(eventSafe(m.params));return;} + if(m.method && m.id!==undefined){ + this.interactions.push({method:m.method,at:new Date().toISOString(),paramKeys:Object.keys(m.params||{})}); + if(m.method==='session/requestRuntimePreferences')this.write({id:m.id,result:{nativeSearchEnhancementsEnabled:false,memoryEnabled:false,askUserQuestionAutoResolutionEnabled:false}}); + else if(m.method==='interaction/requestProviderRuntimeHeaders')this.write({id:m.id,result:runtimeAuthReply(m.params?.modelSelection?.providerId||m.params?.providerId,realCfg)}); + else this.write({id:m.id,error:{code:-32601,message:'Research host does not approve unexpected interactions'}}); + return; + } + const p=this.pending.get(m.id);if(p){clearTimeout(p.timer);this.pending.delete(m.id);m.error?p.reject(new Error(JSON.stringify(m.error))):p.resolve(m.result);} + } + request(method,params={}) { + this.calls.push({method,at:new Date().toISOString()});const id=++this.id; + return new Promise((resolve,reject)=>{const timer=setTimeout(()=>{this.pending.delete(id);reject(new Error('RPC timeout '+method));},30000);this.pending.set(id,{resolve,reject,timer});this.write({id,method,params});}); + } + async init(){const payload=buildAccountProviderPayload(realCfg);if(payload){const a=buildAccountProviderPayload(cfg);payload.basedOnZCodeBuiltinRevision=a.basedOnZCodeBuiltinRevision;this.sync=await this.request('provider/updateAccountConfig',payload);}this.capabilities=await this.request('runtime/capabilities').catch(e=>({error:e.message}));} + async create(cwd,normalized=false){const s=await this.request('session/create',{workspace:{workspacePath:cwd,workspaceKey:cwd},mode:'yolo',persistence:'immediate',...(normalized?{toolDenylist:denyTools,titleGenerationEnabled:false}: {})});this.sid=s.session.sessionId;return this.request('session/setModel',{sessionId:this.sid,model,persistAsWorkspaceLastUsed:false});} + async resume(sid,cwd){const x=await this.request('session/resume',{sessionId:sid,workspace:{workspacePath:cwd,workspaceKey:cwd}});this.sid=sid;return x;} + async run(prompt,timeout=300000){ + const start=Date.now();await this.request('session/subscribe',{sessionId:this.sid,deliveryKind:'desktop-continuous',includeSnapshot:false}); + const from=this.events.length;await this.request('session/send',{sessionId:this.sid,content:prompt}); + while(Date.now()-start['turn.completed','turn.failed'].includes(e.type));if(end){await delay(1500);return {wallTimeMs:Date.now()-start,terminal:end,events:this.events.slice(from)};}if(this.closed)throw new Error('server exited');await delay(200);} + await this.request('session/stop',{sessionId:this.sid}).catch(()=>{});throw new Error('turn timeout'); + } + async finish(){this.child.stdin.end();for(let i=0;i<40&&!this.closed;i++)await delay(100);if(!this.closed)execFileSync('taskkill',['/PID',String(this.child.pid),'/T','/F'],{windowsHide:true});return this.exit;} +} +export async function native(cwd,prompt,sid,timeout=300000,extra=[]){ + // Prompt is a synthetic experiment, never credentials. Keep out of argv. + const pf=path.join(base,'prompt-'+Date.now()+'.txt');writeFileSync(pf,prompt); + const loader=fileURLToPath(new URL('../../../src/adapters/zcode-loader.cjs',import.meta.url)); + const args=[loader,cfg.zcodeEntrypoint,pf,'--cwd',cwd,'--mode','yolo','--output-format','stream-json',...extra]; + // An explicit experiment mode replaces the default rather than duplicating it. + const modeAt=extra.indexOf('--mode'); + if(modeAt>=0){args[args.indexOf('--mode')+1]=extra[modeAt+1];args.splice(args.lastIndexOf('--mode'),2);} + if(sid)args.push('--resume',sid); + const start=Date.now();const child=spawn(cfg.nodeExecutable,args,{cwd,env,windowsHide:true,stdio:['ignore','pipe','pipe']}); + let stdout='',stderr='';child.stdout.on('data',s=>stdout+=s);child.stderr.on('data',s=>stderr+=s); + const exit=await new Promise((r,j)=>{const timer=setTimeout(()=>{execFileSync('taskkill',['/PID',String(child.pid),'/T','/F'],{windowsHide:true});},timeout);child.on('error',j);child.on('close',(code,signal)=>{clearTimeout(timer);r({code,signal});});}); + const parsed=[];for(const line of stdout.split('\n')){try{parsed.push(JSON.parse(line));}catch{}} + return {executor:'native',wallTimeMs:Date.now()-start,pid:child.pid,exit, + events:parsed.filter(x=>x.type&&x.type!=='result').map(eventSafe),summary:parsed.findLast(x=>!x.type||x.type==='result'), + stdoutBytes:Buffer.byteLength(stdout),stderrBytes:Buffer.byteLength(stderr),stderrHash:hash(stderr), + // Safe diagnostics only, no raw log persistence. + error:stderr.split('\n').filter(l=>/Error|Unsupported|Unknown|not found|unavailable|No model|No provider|login|required|must be/.test(l)).map(l=>l.slice(0,300)).slice(0,5)}; +} +export {base,env,cfg,model,save,snapshot,hash,delay,denyTools}; +if(process.argv[1]===fileURLToPath(import.meta.url)){ + const cwd=path.join(base,'smoke-workspace');mkdirSync(cwd,{recursive:true}); + const s=new Server(cwd);let out={runtimeVersion:'0.16.9',model,executor:'smoke'}; + try{await s.init();const created=await s.create(cwd);out.created=snapshot(created);out.sid=s.sid;out.capabilities=s.capabilities;out.seed=await s.run('Reply exactly AUDIT_FLASH_SEED. Do not use tools or modify files.');out.seedAfter=snapshot(await s.request('session/read',{sessionId:s.sid}));await s.finish(); + out.native=await native(cwd,'Reply exactly AUDIT_FLASH_SMOKE. Do not use tools or modify files.',s.sid); + const t=new Server(cwd);try{await t.init();out.resumed=snapshot(await t.resume(s.sid,cwd));out.app=await t.run('Reply exactly AUDIT_FLASH_CONTINUED. Do not use tools or modify files.');out.after=snapshot(await t.request('session/read',{sessionId:s.sid}));}finally{await t.finish();} + }catch(e){out.error=e.message;await s.finish();}save('smoke-isolated',out);console.log(JSON.stringify({sid:out.sid,model:out.created?.model,nativeExit:out.native?.exit,nativeError:out.native?.error,nativeSummary:out.native?.summary,appUsage:out.app?.terminal?.usage,error:out.error},null,2)); +} diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/experiments/review-benchmark.mjs b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/review-benchmark.mjs new file mode 100644 index 0000000..08e7a00 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/review-benchmark.mjs @@ -0,0 +1,37 @@ +import {readFileSync,writeFileSync,mkdirSync} from 'node:fs'; +import {execFileSync,spawnSync} from 'node:child_process'; +import path from 'node:path'; +process.env.AUDIT_REVIEW_ONLY='1'; +const {verify,tasks}=await import('./run-benchmark.mjs'); +const base=process.env.AUDIT_ROOT||'C:/Users/Sandy/.codex/tmp/zcode-executor-audit-20261005'; +const phase=process.env.AUDIT_BENCHMARK_PHASE==='controlled'?'benchmark-controlled':'benchmark'; +const ledger=JSON.parse(readFileSync(path.join(base,phase+'.json'),'utf8')); +const cwd=ledger.cwd; +for(const r of ledger.pairs){ + execFileSync('git',['reset','--hard',ledger.commit],{cwd});execFileSync('git',['clean','-fd'],{cwd});mkdirSync(path.join(cwd,'test'),{recursive:true}); + if(r.acceptance.diff)execFileSync('git',['apply','--ignore-space-change','-'],{cwd,input:r.acceptance.diff}); + for(const f of r.acceptance.untracked||[])if(f.content!==null){mkdirSync(path.dirname(path.join(cwd,f.path)),{recursive:true});writeFileSync(path.join(cwd,f.path),f.content);} + r.initialAcceptance=r.acceptance; + r.acceptance=verify(tasks.find(t=>t.id===r.task)); + r.acceptance.reviewer='Codex independent replay of stored candidate diff, corrected path and TAP parsing'; + const events=r.run.events; + const tools=new Map();for(const e of events)if(e.toolCallId&&e.tool&&!tools.has(e.toolCallId))tools.set(e.toolCallId,e.tool); + const terminal=events.findLast(e=>['turn.completed','turn.failed'].includes(e.type)); + r.executionCompleted=terminal?.type==='turn.completed'&&terminal.resultType==='success'&&(r.executor!=='native'||r.run.exit?.code===0); + r.metrics={executor:r.executor,sessionId:r.sessionId,runId:r.runId, + traceId:events.find(e=>e.traceId)?.traceId||r.created?.session?.traceId||r.run.summary?.traceId||null, + model:[...new Set(events.filter(e=>e.modelId).map(e=>e.modelId))],provider:[...new Set(events.filter(e=>e.providerId).map(e=>e.providerId))], + reasoningLevel:r.created?.model?.options?.reasoningLevel||r.modelIO?.records?.[0]?.outputConfig?.effort||'low (private configured default; request verification tracked separately)',mode:'yolo', + wallTimeMs:r.run.wallTimeMs,wallTimeInterval:r.executor==='native'?'CLI spawn to exit':'session/send to turn.completed plus 1.5s observation', + runtimeTurnDurationMs:events.findLast(e=>e.type==='turn.completed')?.duration||null, + turnCount:events.filter(e=>e.type==='turn.started').length,modelRequestCount:r.usage?.modelRequestCount||null, + inputTokens:r.usage?.inputTokens??null,cachedInputTokens:r.usage?.cacheReadTokens??null,outputTokens:r.usage?.outputTokens??null,totalTokens:r.usage?.totalTokens??null, + toolCallCount:tools.size,toolSequence:[...tools.values()],permissionRequestCount:r.executor==='app-server'?r.permissionRequestCount:events.filter(e=>e.type==='permission.requested').length, + subagentCount:r.subagents?.childSessionIds?.length??null,workflowCount:null,filesChanged:r.acceptance.filesChanged, + linesAdded:r.acceptance.linesAdded,linesDeleted:r.acceptance.linesDeleted,testsPassed:r.acceptance.testsPassed,testsFailed:r.acceptance.testsFailed, + acceptanceCriteriaPassed:r.acceptance.passed,retryCount:events.filter(e=>e.attempt>1&&e.requestType?.includes('started')).length, + finalStatus:terminal?.type==='turn.failed'?'failed':r.finalStatus??null,qualityScore:r.executionCompleted&&r.acceptance.passed?4:null,qualityReview:r.executionCompleted&&r.acceptance.passed?'independent acceptance and scoped diff review; non-blinded; 4=good, not proof of executor superiority':'execution failed/incomplete; no coding quality inference',cpuTimeMs:null,peakWorkingSetBytes:null, + contextUsageBreakdown:events.filter(e=>e.contextUsageBreakdown).map(e=>e.contextUsageBreakdown)}; + console.log(JSON.stringify({runId:r.runId,passed:r.acceptance.passed,tests:r.acceptance.testsPassed,mutants:r.acceptance.mutantsKilled,files:r.acceptance.filesChanged})); +} +writeFileSync(path.join(base,phase+'-reviewed.json'),JSON.stringify(ledger,null,2)); diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-benchmark.mjs b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-benchmark.mjs new file mode 100644 index 0000000..c9d2619 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-benchmark.mjs @@ -0,0 +1,73 @@ +// Five real model-executed coding workloads in an isolated, runnable fixture. +// These are representative Bridge-domain tasks, not asserted production bugs. +import {spawnSync,execFileSync} from 'node:child_process'; +import {mkdirSync,writeFileSync,readFileSync,existsSync,readdirSync} from 'node:fs'; +import path from 'node:path'; +import {Server,native,base,save,snapshot,hash,denyTools} from './probe.mjs'; +import {modelIO} from './summarize-model-io.mjs'; +const phase=process.env.AUDIT_BENCHMARK_PHASE==='controlled'?'benchmark-controlled':'benchmark'; +const cwd=path.join(base,phase,'workspace');mkdirSync(path.join(cwd,'src'),{recursive:true});mkdirSync(path.join(cwd,'test'),{recursive:true}); +const files={ + 'package.json':JSON.stringify({name:'isolated-executor-benchmark',type:'module',scripts:{test:'node --test'}},null,2), + 'AGENTS.md':'Work only on the files listed in the task. Use built-in Node APIs. Do not install packages, delegate, access network, or commit.\n', + 'src/timeout.mjs':`export const DEFAULT=3600000;\nexport function parseTimeout(raw){const n=Number(raw);return Number.isSafeInteger(n)&&n>=60000&&n<=14400000?n:DEFAULT;}\n`, + 'src/paths.mjs':`import path from 'node:path';\nexport function canonical(raw,windows=false){const p=windows?path.win32:path.posix;const n=p.normalize(raw).replace(/[\\\\/]+$/,'');return windows?n.toLowerCase():n;}\nexport function overlaps(a,b,windows=false){a=canonical(a,windows);b=canonical(b,windows);const sep=windows?'\\\\':'/';return a===b||a.startsWith(b+sep)||b.startsWith(a+sep);}\n`, + 'src/retry.mjs':`export async function withRetry(operation,options={}){throw new Error('not implemented');}\n`, + 'src/ledger.mjs':`export const terminalStatuses=['completed','failed'];\nexport function isTerminal(status){return terminalStatuses.includes(status);}\nexport function transition(current,next){if(isTerminal(current)&&next!==current)throw new Error('terminal');if(current==='queued'&&next==='completed')throw new Error('invalid');return next;}\n`, + 'src/report.mjs':`import {isTerminal} from './ledger.mjs';\nexport function summarize(runs){return {completed:runs.filter(x=>x.status==='completed').length,failed:runs.filter(x=>x.status==='failed').length,active:runs.filter(x=>!isTerminal(x.status)).length};}\n`, + 'src/summary.mjs':`export function summarizeEvents(events){let n=0,input=0,output=0;for(const e of events){if(e.type==='turn.completed'){n++;input+=e.usage?.inputTokens??0;output+=e.usage?.outputTokens??0;}}return {count:n,inputTokens:input,outputTokens:output,totalTokens:input+output};}\nexport function summarizeRuns(runs){let n=0,input=0,output=0;for(const r of runs){if(r.status==='completed'){n++;input+=r.usage?.inputTokens??0;output+=r.usage?.outputTokens??0;}}return {count:n,inputTokens:input,outputTokens:output,totalTokens:input+output};}\n` +}; +if(phase==='benchmark-controlled')files['.zcode/config.json']=JSON.stringify({features:{memory:false}},null,2); +if(!existsSync(path.join(cwd,'.git'))){for(const [f,t]of Object.entries(files)){mkdirSync(path.dirname(path.join(cwd,f)),{recursive:true});writeFileSync(path.join(cwd,f),t);}execFileSync('git',['init'],{cwd});execFileSync('git',['add','.'],{cwd});execFileSync('git',['-c','user.name=Executor Audit','-c','user.email=audit@example.invalid','commit','-m','Isolated benchmark baseline'],{cwd});} +const commit=execFileSync('git',['rev-parse','HEAD'],{cwd,encoding:'utf8'}).trim(); +const common=`Work in this isolated repository. Do not change AGENTS.md/package.json, install packages, access network, delegate, or commit. Run relevant checks with node. End with a concise report of changes and tests.`; +const tasks=[ + {id:'T1',kind:'small bug fix',allowed:['src/timeout.mjs'],prompt:`Fix src/timeout.mjs parseTimeout. Accept only string values containing trimmed decimal digits, and safe integer numbers, in the inclusive range 60000..14400000. Hexadecimal, exponent, fractional, signed string, empty, missing, objects, boolean, Infinity and NaN inputs must return DEFAULT. Preserve exports and DEFAULT. Only edit src/timeout.mjs.`}, + {id:'T2',kind:'add unit tests',allowed:['test/paths.test.mjs'],prompt:`Add meaningful node:test unit tests for canonical and overlaps in src/paths.mjs, only writing test/paths.test.mjs. Cover POSIX normalization, equal and nested overlap both directions, sibling prefix false positives, trailing separators, and Windows case-insensitivity and slash normalization. Include at least 8 independently named test cases. Do not modify implementation. Run node --test test/paths.test.mjs.`}, + {id:'T3',kind:'medium feature',allowed:['src/retry.mjs'],prompt:`Implement withRetry(operation, options={}) in src/retry.mjs using built-in JS. Defaults maxAttempts=3, baseDelayMs=10, maxDelayMs=1000, sleep=ms=>new Promise(r=>setTimeout(r,ms)), shouldRetry=()=>true, onAttempt=()=>{}. Call operation(attempt) with 1-based attempt; return success value. On failure only retry when shouldRetry(error,attempt) returns true and attempts remain. Await sleep(min(maxDelayMs,baseDelayMs*2**(attempt-1))) only between attempts. Call onAttempt(attempt) before each operation. Throw last error unchanged when exhausted or not retryable. Validate maxAttempts positive integer, delay values finite nonnegative, before invoking operation. Preserve export. Only edit src/retry.mjs.`}, + {id:'T4',kind:'multi-file behavior',allowed:['src/ledger.mjs','src/report.mjs'],prompt:`Add cancelled lifecycle state across src/ledger.mjs and src/report.mjs. terminalStatuses and isTerminal must recognize cancelled. transition must implement this exact graph: queued -> queued/running/cancelled; running -> running/completed/failed/cancelled; terminal status -> same status only. Reject unknown current or next statuses and every other transition. summarize must return completed,failed,cancelled,active counts, including cancelled separately and exclude it from active. Only edit these two files. Preserve named exports.`}, + {id:'T5',kind:'refactor',allowed:['src/summary.mjs'],prompt:`Refactor src/summary.mjs to remove duplicate aggregation logic shared by summarizeEvents and summarizeRuns. Preserve exact public exports, accepted inputs and output values: count completed entries, sum inputTokens and outputTokens with missing fields as zero, total as their sum. Inputs must not be mutated. Introduce one internal helper for the common accumulation. Only edit src/summary.mjs.`} +]; +function checkScript(id){const pre=`import assert from 'node:assert/strict';\n`; + if(id==='T1')return pre+`import {parseTimeout,DEFAULT} from './src/timeout.mjs';for(const v of ['0x10000','6e4','60000.0','+60000','-60000','',undefined,null,true,false,{},Infinity,NaN,59999,14400001,60000.1])assert.equal(parseTimeout(v),DEFAULT,String(v));for(const [v,n]of [['60000',60000],[' 60000 ',60000],['00060000',60000],['14400000',14400000],[60000,60000],[14400000,14400000]])assert.equal(parseTimeout(v),n);`; + if(id==='T3')return pre+`import {withRetry} from './src/retry.mjs';let attempts=[],delays=[];const sentinel=new Error('sentinel');assert.equal(await withRetry(async n=>{attempts.push(n);if(n<4)throw sentinel;return 'ok';},{maxAttempts:4,baseDelayMs:5,maxDelayMs:12,sleep:async x=>delays.push(x)}),'ok');assert.deepEqual(attempts,[1,2,3,4]);assert.deepEqual(delays,[5,10,12]);attempts=[];delays=[];await assert.rejects(withRetry(async n=>{attempts.push(n);throw sentinel;},{shouldRetry:()=>false,sleep:async n=>delays.push(n)}),e=>e===sentinel);assert.deepEqual(attempts,[1]);assert.deepEqual(delays,[]);await assert.rejects(withRetry(()=>{throw sentinel},{maxAttempts:1}),e=>e===sentinel);for(const opts of [{maxAttempts:0},{maxAttempts:1.5},{baseDelayMs:-1},{maxDelayMs:Infinity}])await assert.rejects(withRetry(()=>assert.fail('invoked'),opts));const seen=[];assert.equal(await withRetry(n=>n,{onAttempt:n=>seen.push(n)}),1);assert.deepEqual(seen,[1]);`; + if(id==='T4')return pre+`import {terminalStatuses,isTerminal,transition} from './src/ledger.mjs';import {summarize} from './src/report.mjs';const ss=['queued','running','completed','failed','cancelled'];const graph={queued:['queued','running','cancelled'],running:['running','completed','failed','cancelled'],completed:['completed'],failed:['failed'],cancelled:['cancelled']};for(const a of ss)for(const b of ss){if(graph[a].includes(b))assert.equal(transition(a,b),b);else assert.throws(()=>transition(a,b));}assert.throws(()=>transition('unknown','running'));assert.throws(()=>transition('running','unknown'));assert.equal(isTerminal('cancelled'),true);assert.ok(terminalStatuses.includes('cancelled'));assert.deepEqual(summarize(ss.map(status=>({status}))),{completed:1,failed:1,cancelled:1,active:2});`; + if(id==='T5')return pre+`import {summarizeEvents,summarizeRuns} from './src/summary.mjs';for(const [fn,key,val]of [[summarizeEvents,'type','turn.completed'],[summarizeRuns,'status','completed']]){assert.deepEqual(fn([]),{count:0,inputTokens:0,outputTokens:0,totalTokens:0});const entries=Object.freeze([Object.freeze({[key]:val,usage:Object.freeze({inputTokens:3,outputTokens:2})}),Object.freeze({[key]:'ignore',usage:{inputTokens:99}}),Object.freeze({[key]:val})]);assert.deepEqual(fn(entries),{count:2,inputTokens:3,outputTokens:2,totalTokens:5});}`; +} +function verify(t){ + let result; + if(t.id==='T2'){ + const file=path.join(cwd,'test/paths.test.mjs');if(!existsSync(file)){result={passed:false,testsPassed:0,testsFailed:1,details:'test file missing'};}else{ + const normal=spawnSync(process.execPath,['--test','--test-reporter=tap','test/paths.test.mjs'],{cwd,encoding:'utf8'});const count=Number(normal.stdout.match(/# tests (\d+)/)?.[1]||0); + const pf=path.join(cwd,'src/paths.mjs');const original=readFileSync(pf,'utf8');const mutants=[original.replace('a.startsWith(b+sep)','a.startsWith(b)').replace('b.startsWith(a+sep)','b.startsWith(a)'),original.replace('return windows?n.toLowerCase():n','return n'),original.replace('a===b||','')];const kills=[]; + try{for(const m of mutants){writeFileSync(pf,m);kills.push(spawnSync(process.execPath,['--test','--test-reporter=tap','test/paths.test.mjs'],{cwd,encoding:'utf8'}).status!==0);}}finally{writeFileSync(pf,original);} + result={passed:normal.status===0&&count>=8&&kills.every(Boolean),testsPassed:count,testsFailed:Number(normal.stdout.match(/# fail (\d+)/)?.[1]||0),mutantsKilled:kills,details:normal.stdout.slice(-1000)};} + }else{ + const r=spawnSync(process.execPath,['--input-type=module','-e',checkScript(t.id)],{cwd,encoding:'utf8'});result={passed:r.status===0,testsPassed:r.status===0?1:0,testsFailed:r.status===0?0:1,details:(r.stdout+r.stderr).slice(-1800)}; + } + const changed=execFileSync('git',['status','--porcelain'],{cwd,encoding:'utf8'}).trimEnd().split('\n').filter(Boolean).map(l=>l.slice(3)); + // git status may collapse an untracked test directory; independently inspect it. + if(changed.includes('test/')){changed.splice(changed.indexOf('test/'),1,...readdirSync(path.join(cwd,'test')).map(f=>'test/'+f));} + result.filesChanged=changed;result.scopePassed=changed.every(f=>t.allowed.includes(f));result.artifactProduced=changed.length>0;result.passed&&=result.scopePassed&&result.artifactProduced; + const stat=execFileSync('git',['diff','--numstat'],{cwd,encoding:'utf8'}).trim().split('\n').filter(Boolean); + result.linesAdded=stat.reduce((n,l)=>n+Number(l.split('\t')[0]||0),0);result.linesDeleted=stat.reduce((n,l)=>n+Number(l.split('\t')[1]||0),0); + for(const f of changed)if(existsSync(path.join(cwd,f))&&spawnSync('git',['ls-files','--error-unmatch',f],{cwd,encoding:'utf8'}).status!==0)result.linesAdded+=readFileSync(path.join(cwd,f),'utf8').trimEnd().split('\n').length; + result.diff=execFileSync('git',['diff'],{cwd,encoding:'utf8'});result.untracked=changed.filter(f=>!result.diff.includes('b/'+f)).map(f=>({path:f,content:existsSync(path.join(cwd,f))?readFileSync(path.join(cwd,f),'utf8'):null}));return result; +} +export {verify,tasks}; +if(!process.env.AUDIT_REVIEW_ONLY){ +const ledger={date:'2026-10-05',commit,cwd,phase,kind:'isolated representative coding tasks',pairs:[],controls:{reasoning:'low',mode:'yolo',memory:phase==='benchmark-controlled'?'both disabled by project config':'native default, app-server disabled',tools:'Read Write Edit Bash',denyTools}}; +for(const [i,t] of tasks.entries())for(const executor of i%2===0?['native','app-server']:['app-server','native']){ + // Safe: reset only the newly initialized, dedicated fixture checkout above. + execFileSync('git',['reset','--hard',commit],{cwd});execFileSync('git',['clean','-fd'],{cwd});mkdirSync(path.join(cwd,'test'),{recursive:true}); + const runId=t.id+'-'+executor;const prompt=t.prompt+'\n'+common;console.log('START '+runId); + const entry={runId,task:t.id,kind:t.kind,executor,prompt,promptHash:hash(prompt),commit,workspace:cwd,acceptanceCriteria:t.prompt,startedAt:new Date().toISOString(),permissionRequestCount:0,retryCount:0}; + let s; + try{if(executor==='native'){entry.run=await native(cwd,prompt,null,300000,['--disallowed-tools',denyTools.join(',')]);entry.sessionId=entry.run.summary?.sessionId||entry.run.events[0]?.sessionId;} + else{s=new Server(cwd);await s.init();entry.created=snapshot(await s.create(cwd,true));entry.sessionId=s.sid;console.log(JSON.stringify({runId,sessionId:s.sid,model:entry.created.model,mode:entry.created.mode}));entry.run=await s.run(prompt);entry.after=snapshot(await s.request('session/read',{sessionId:s.sid}));entry.interactions=s.interactions;entry.permissionRequestCount=s.interactions.filter(x=>x.method==='interaction/requestPermission').length;entry.subagents=await s.request('session/subagents',{sessionId:s.sid}).catch(e=>({error:e.message}));} + entry.acceptance=verify(t);entry.usage=entry.run.events.findLast(e=>e.type==='turn.completed')?.usage;entry.finalStatus=entry.run.events.findLast(e=>['turn.completed','turn.failed'].includes(e.type))?.resultType||entry.run.exit?.code; + }catch(e){entry.error=e.message;entry.acceptance=verify(t);}finally{if(s)await s.finish();if(entry.sessionId)entry.modelIO=modelIO(entry.sessionId);entry.finishedAt=new Date().toISOString();ledger.pairs.push(entry);save(phase,ledger);} + console.log(JSON.stringify({runId,sessionId:entry.sessionId,usage:entry.usage,passed:entry.acceptance?.passed,files:entry.acceptance?.filesChanged,error:entry.error})); +} +console.log('BENCHMARK COMPLETE'); +} diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-handoff.mjs b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-handoff.mjs new file mode 100644 index 0000000..4c14f4f --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-handoff.mjs @@ -0,0 +1,34 @@ +import {spawn,execFileSync} from 'node:child_process'; +import {mkdirSync,writeFileSync,readFileSync,existsSync} from 'node:fs'; +import path from 'node:path'; +import {fileURLToPath} from 'node:url'; +import {Server,native,base,env,cfg,save,snapshot,delay,denyTools,hash} from './probe.mjs'; +const cwd=path.join(base,'handoff-workspace');mkdirSync(cwd,{recursive:true}); +const out={date:'2026-10-05',mode:'yolo',model:'GLM-5.3-Flash',reasoning:'low'}; +const file=path.join(cwd,'handoff.txt'); +const terminal=x=>x.events.findLast(e=>e.type==='turn.completed'); +const snap=x=>({...snapshot(x),messages:x.messages?.map(m=>({id:m.info?.id,role:m.info?.role,hash:hash(m)}))}); +const tryRequest=async(s,m,p)=>{try{return await s.request(m,p);}catch(e){return {error:e.message};}}; +writeFileSync(file,'ALPHA_731\nstage=one\n'); +out.A={};out.A.native=await native(cwd,'Use Read to read handoff.txt. Remember its first line for the next turn. Reply only that first line. Do not modify files or use other tools.',null); +out.A.sid=out.A.native.summary?.sessionId||out.A.native.events[0]?.sessionId; +let s=new Server(cwd); +try{await s.init();out.A.list=await tryRequest(s,'session/list',{sessionIds:[out.A.sid]});out.A.coldRead=await tryRequest(s,'session/read',{sessionId:out.A.sid});out.A.resumed=snap(await s.resume(out.A.sid,cwd)); +out.A.usageBefore=await tryRequest(s,'session/usage',{sessionId:out.A.sid}); +out.A.continue=await s.run('State the remembered first line from the previous turn. Then use Edit to replace stage=one with stage=two in handoff.txt without reading the file again. Do not use Bash.');out.A.after=snap(await s.request('session/read',{sessionId:out.A.sid}));out.A.fileAfter=readFileSync(file,'utf8');out.A.usageAfter=await tryRequest(s,'session/usage',{sessionId:out.A.sid});} +catch(e){out.A.error=e.message;}finally{await s.finish();}save('handoff',out); +writeFileSync(file,'BETA_842\nstage=one\n');out.B={};s=new Server(cwd); +try{await s.init();out.B.created=snap(await s.create(cwd));out.B.sid=s.sid;out.B.first=await s.run('Use Read to read handoff.txt. Remember its first line for the next turn. Reply only that first line. Do not modify files or use other tools.');out.B.before=snap(await s.request('session/read',{sessionId:s.sid}));out.B.stop=await s.request('session/stop',{sessionId:s.sid});out.B.close=await s.request('session/close',{sessionId:s.sid});out.B.listAfterClose=await tryRequest(s,'session/list',{sessionIds:[s.sid]});} +catch(e){out.B.error=e.message;}finally{await s.finish();} +if(out.B.sid){out.B.native=await native(cwd,'State the remembered first line from the previous turn. Then use Edit to replace stage=one with stage=two in handoff.txt without reading the file again. Do not use Bash.',out.B.sid);out.B.fileAfter=readFileSync(file,'utf8');s=new Server(cwd);try{await s.init();out.B.after=snap(await s.resume(out.B.sid,cwd));}catch(e){out.B.afterError=e.message;}finally{await s.finish();}}save('handoff',out); +// C: observe an owned disposable Native session. Never cold-resume it live. +out.C={resumeStatus:'NOT RUN — safety stop: no cross-process ownership lease found; resume rehydrates a second runtime'}; +writeFileSync(path.join(cwd,'wait.cjs'),`require('node:fs').writeFileSync('running-marker.txt','started');setTimeout(()=>{require('node:fs').writeFileSync('finished-marker.txt','finished');},15000);`); +const pf=path.join(base,'concurrent-prompt.txt');writeFileSync(pf,'Use Bash to run node wait.cjs in the current directory and wait for it to finish. Do not modify or read files. Then reply DONE.'); +const loader=fileURLToPath(new URL('../../../src/adapters/zcode-loader.cjs',import.meta.url)); +const c=spawn(cfg.nodeExecutable,[loader,cfg.zcodeEntrypoint,pf,'--cwd',cwd,'--mode','yolo','--output-format','stream-json','--disallowed-tools',denyTools.join(',')],{cwd,env,windowsHide:true,stdio:['ignore','pipe','pipe']}); +out.C.pid=c.pid;let sid,raw='',buf='',closed=false;c.stderr.resume();c.stdout.setEncoding('utf8');c.stdout.on('data',chunk=>{raw+=chunk;buf+=chunk;let i;while((i=buf.indexOf('\n'))>=0){let x;try{x=JSON.parse(buf.slice(0,i));}catch{}buf=buf.slice(i+1);if(x?.sessionId)sid=x.sessionId;}});c.on('close',code=>{closed=true;out.C.exitCode=code;}); +try{const start=Date.now();while(!closed&&!existsSync(path.join(cwd,'running-marker.txt'))&&Date.now()-start<120000)await delay(100);out.C.sid=sid;out.C.runningConfirmed=!closed&&existsSync(path.join(cwd,'running-marker.txt'));if(out.C.runningConfirmed&&sid){s=new Server(cwd);try{await s.init();out.C.list=await tryRequest(s,'session/list',{sessionIds:[sid]});out.C.read=await tryRequest(s,'session/read',{sessionId:sid});out.C.stopForeign=await tryRequest(s,'session/stop',{sessionId:sid});}finally{await s.finish();}} +const end=Date.now()+120000;while(!closed&&Date.now(){ + if(m.method==='interaction/requestPermission'&&m.id!==undefined){const p=m.params||{};const f=p.input?.file_path||p.input?.filePath||p.input?.path;const allowed=p.toolName==='Write'&&typeof f==='string'&&path.resolve(permissionDir,f)===path.join(permissionDir,'approval.txt');s.interactions.push({method:m.method,tool:p.toolName,path:f,decision:allowed?'allow':'deny',authorizedScope:'requested permission roundtrip experiment, owned approval.txt only'});s.write({id:m.id,result:{decision:allowed?'allow':'deny',reason:allowed?'User-authorized isolated permission experiment':'Outside experiment scope'}});return;}original(m);}; + out.appBuild=await s.run(prompt,120000);out.appInteractions=s.interactions;out.appFile=existsSync(path.join(permissionDir,'approval.txt'))?readFileSync(path.join(permissionDir,'approval.txt'),'utf8'):null; +}catch(e){out.appError=e.message;}finally{await s.finish();}save('policy',out); +const memoryDir=path.join(base,'memory-workspace');mkdirSync(memoryDir,{recursive:true}); +const memoryPrompt='For this disposable project, remember the agreed convention: session files use UTC timestamps. Reply ACK only. Do not use tools or modify project files.'; +out.nativeMemory=await native(memoryDir,memoryPrompt,null,180000,['--memory-bench','--disallowed-tools',denyTools.join(',')]); +out.nativeMemory.sid=out.nativeMemory.summary?.sessionId||out.nativeMemory.events[0]?.sessionId;out.nativeMemory.modelIO=modelIO(out.nativeMemory.sid); +const t=new Server(memoryDir);try{const original=t.handle.bind(t);t.handle=m=>{if(m.method==='session/requestRuntimePreferences'&&m.id!==undefined){t.write({id:m.id,result:{nativeSearchEnhancementsEnabled:false,memoryEnabled:true,askUserQuestionAutoResolutionEnabled:false}});return;}original(m);};await t.init();out.appMemoryCreated=snapshot(await t.create(memoryDir,true));out.appMemory=await t.run(memoryPrompt,120000);await delay(8000);out.appMemoryClose=await t.request('session/close',{sessionId:t.sid}).catch(e=>({error:e.message}));out.appMemoryIO=modelIO(t.sid); +}catch(e){out.appMemoryError=e.message;}finally{await t.finish();save('policy',out);} +console.log(JSON.stringify({nativeBuild:out.nativeBuild.exit,nativeBuildTools:out.nativeBuild.events.filter(e=>e.tool).map(e=>e.tool),appRequests:out.appInteractions?.filter(x=>x.method==='interaction/requestPermission'),appFile:out.appFile,nativeMemoryCalls:out.nativeMemory.modelIO,appMemoryCalls:out.appMemoryIO},null,2)); diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-process-control.mjs b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-process-control.mjs new file mode 100644 index 0000000..6176c51 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-process-control.mjs @@ -0,0 +1,20 @@ +import {spawn,execFileSync} from 'node:child_process'; +import {mkdirSync,writeFileSync,readFileSync,existsSync,unlinkSync} from 'node:fs'; +import path from 'node:path'; +import {fileURLToPath} from 'node:url'; +import {Server,base,env,cfg,save,delay,denyTools} from './probe.mjs'; +const cwd=path.join(base,'process-workspace');mkdirSync(cwd,{recursive:true}); +writeFileSync(path.join(cwd,'wait.cjs'),`require('node:fs').writeFileSync('child-pid.txt',String(process.pid));setTimeout(()=>require('node:fs').writeFileSync('done.txt','done'),60000);`); +const alive=pid=>{try{process.kill(pid,0);return true;}catch{return false;}}; +const out={};const prompt='Use Bash to run node wait.cjs in this directory, wait for it, then reply DONE. Do not modify or read any files.'; +for(const executor of ['native','app-server']){ + for(const f of ['child-pid.txt','done.txt'])if(existsSync(path.join(cwd,f)))unlinkSync(path.join(cwd,f)); + let server,child,closed=false,runPromise;const r={executor};let lines=''; + if(executor==='native'){const pf=path.join(base,'process-prompt.txt');writeFileSync(pf,prompt);const loader=fileURLToPath(new URL('../../../src/adapters/zcode-loader.cjs',import.meta.url));child=spawn(cfg.nodeExecutable,[loader,cfg.zcodeEntrypoint,pf,'--cwd',cwd,'--mode','yolo','--output-format','stream-json','--disallowed-tools',denyTools.join(',')],{cwd,env,windowsHide:true,stdio:['ignore','pipe','pipe']});child.stdout.on('data',s=>lines+=s);child.stderr.resume();child.on('close',code=>{closed=true;r.exitCode=code;});} + else{server=new Server(cwd);await server.init();await server.create(cwd,true);r.sessionId=server.sid;child=server.child;runPromise=server.run(prompt,120000).catch(e=>({error:e.message}));} + r.pid=child.pid;const start=Date.now();while(!existsSync(path.join(cwd,'child-pid.txt'))&&!closed&&Date.now()-start<60000)await delay(100); + r.toolStarted=existsSync(path.join(cwd,'child-pid.txt'));if(r.toolStarted){r.toolPid=Number(readFileSync(path.join(cwd,'child-pid.txt'),'utf8'));r.childAliveBefore=alive(r.toolPid);const stopAt=Date.now();if(server){r.stopReply=await server.request('session/stop',{sessionId:server.sid});r.stopAckMs=Date.now()-stopAt;r.run=await runPromise;r.parentAliveAfterStop=alive(child.pid);}else{r.killResult=execFileSync('taskkill',['/PID',String(child.pid),'/T','/F'],{windowsHide:true,encoding:'utf8'});for(let i=0;i<30&&!closed;i++)await delay(100);r.treeKillMs=Date.now()-stopAt;} + await delay(1000);r.childAliveAfter=alive(r.toolPid);r.doneMarker=existsSync(path.join(cwd,'done.txt'));if(r.childAliveAfter){r.cleanup='owned helper child explicitly terminated after observation';execFileSync('taskkill',['/PID',String(r.toolPid),'/T','/F'],{windowsHide:true});}} + if(server){r.parentExit=await server.finish();r.parentAliveAfterFinish=alive(child.pid);}else if(!closed)execFileSync('taskkill',['/PID',String(child.pid),'/T','/F'],{windowsHide:true}); + out[executor]=r;save('process-control',out);console.log(JSON.stringify({executor,sessionId:r.sessionId,toolStarted:r.toolStarted,childAliveAfter:r.childAliveAfter,parentAliveAfterStop:r.parentAliveAfterStop,stopAckMs:r.stopAckMs,treeKillMs:r.treeKillMs})); +} diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-reasoning.mjs b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-reasoning.mjs new file mode 100644 index 0000000..15e8816 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-reasoning.mjs @@ -0,0 +1,15 @@ +import {readFileSync,writeFileSync,mkdirSync} from 'node:fs'; +import path from 'node:path'; +import {Server,native,base,cfg,save,snapshot,denyTools} from './probe.mjs'; +import {modelIO} from './summarize-model-io.mjs'; +const cwd=path.join(base,'reasoning-workspace');mkdirSync(cwd,{recursive:true}); +const personal=JSON.parse(readFileSync(cfg.providerPersonalConfigFile,'utf8'));const original=personal.config.defaultModelSelection; +const out=[];const prompt='Compute the sum of all prime numbers less than 30. Reply with only the integer answer. Do not use tools, delegate, modify files, or access network.'; +try{for(const level of ['low','high','max'])for(const executor of ['native','app-server']){ + const r={level,executor};let s; + try{if(executor==='native'){personal.config.defaultModelSelection={...original,options:{reasoningLevel:level}};writeFileSync(cfg.providerPersonalConfigFile,JSON.stringify(personal),{mode:0o600});r.run=await native(cwd,prompt,null,120000,['--disallowed-tools',denyTools.join(',')]);r.sid=r.run.summary?.sessionId||r.run.events[0]?.sessionId;} + else{s=new Server(cwd);await s.init();await s.create(cwd,true);const selected=await s.request('session/setThoughtLevel',{sessionId:s.sid,thoughtLevel:level});r.selected=snapshot(selected);r.sid=s.sid;r.run=await s.run(prompt,120000);} + r.terminal=r.run.events.findLast(e=>e.type==='turn.completed');r.correct=r.terminal?.response?.trim()==='129';r.modelIO=modelIO(r.sid); + }catch(e){r.error=e.message;}finally{if(s)await s.finish();out.push(r);save('reasoning',out);} + console.log(JSON.stringify({executor,level,sid:r.sid,selected:r.selected?.thought?.current,correct:r.correct,usage:r.terminal?.usage,error:r.error})); +}}finally{personal.config.defaultModelSelection=original;writeFileSync(cfg.providerPersonalConfigFile,JSON.stringify(personal),{mode:0o600});} diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-source.mjs b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-source.mjs new file mode 100644 index 0000000..9239306 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/run-source.mjs @@ -0,0 +1,16 @@ +import path from 'node:path'; +import {Server,base,save,snapshot} from './probe.mjs'; +const cwd=path.join(base,'source-review'); +const s=new Server(cwd); +const result={}; +try{ + await s.init();result.created=snapshot(await s.create(cwd)); + console.log(JSON.stringify({sessionId:s.sid,model:result.created.model,thought:result.created.thought,mode:result.created.mode})); + result.run=await s.run(`研究任务,仅生成源码证据文档。官方源码位于 ${base}/official,commit 29628c9acdb81b703bbd4080c207a0e7ce5e276e,CLI 安装版 0.16.9 不保证与其对应。只读官方源码。当前 Bridge worktree HEAD 2596759198fa826c2b7ac0478c5682da996e9727。 +只允许写当前目录 docs/research/source-audit/CALLCHAIN.md、RUNTIME_DIFFERENCES.md、HANDOFF_SOURCE.md。禁止修改 Bridge实现/test/plugins/package或官方源码;禁止启动其他zcode/模型、git提交、网络、读取私人配置凭据DB日志memory。用文件读取、搜索工具追踪源码即可。 +CALLCHAIN: Native packages/cli/src/run.ts -> prompt-command.ts runPrompt/createZCodeApp/submitPrompt,与 bootstrap/src/zcode-protocol-entrypoint.ts -> zcode-protocol/server-operations.ts materializeSessionRecord/sessionSend 进入 model runtime 的实际汇合点。追踪到模型request执行函数。所有引用有实际文件行号、固定commit GitHub链接。 +RUNTIME_DIFFERENCES: system prompt、MCS、workspace AGENTS/skills/MCP、memory use vs extraction、tools registry vs permission policy、browser、provider account auth、model reasoning low/high/max/default、foreground/background subagent、workflow terminal settle。记录每个默认值和覆盖位置。当前 Bridge 的请求RuntimePreferences与turn.completed收口也列出。 +HANDOFF_SOURCE: 追踪持久化 session entries/model/mode/workspace/read state/usage 与 resume;close是否删除、EOF是否保留、stop做什么;检查跨进程 session ownership/lock,不把SQLite migration/file锁当作session执行锁。运行时注册Map只在同进程还是跨进程。发现双Runtime并发写风险需明确,但不要运行探针。 +用中文。直接源码事实标CONFIRMED — SOURCE;不能确定的写不确定/UNKNOWN。不要假定benchmark或runtime验证结果,不编造行号。尽可能深入检查实际实现。完成返回简短报告列文件、关键事实、未知项。`,900000); + console.log(JSON.stringify({completed:result.run.terminal?.resultType,usage:result.run.terminal?.usage,tools:result.run.events.filter(e=>e.type.startsWith('tool.')).length})); +}catch(e){result.error=e.message;console.log(result.error);}finally{await s.finish();save('source-task',result);} diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/experiments/summarize-model-io.mjs b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/summarize-model-io.mjs new file mode 100644 index 0000000..07ae887 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/summarize-model-io.mjs @@ -0,0 +1,21 @@ +// Allowlisted digest of this audit's owned model-I/O files. Never export headers, +// raw requests, reasoning text, prompts, or unrelated sessions. +import {readFileSync,existsSync,writeFileSync} from 'node:fs'; +import {createHash} from 'node:crypto'; +import path from 'node:path'; +const base=process.env.AUDIT_ROOT||'C:/Users/Sandy/.codex/tmp/zcode-executor-audit-20261005'; +const hash=x=>createHash('sha256').update(typeof x==='string'?x:JSON.stringify(x)).digest('hex'); +export function modelIO(sid){ + const candidates=[path.join(base,'private-runtime','.zcode','cli','rollout',`model-io-${sid}.jsonl`),path.join(base,'private-runtime','.zcode','cli','cli','rollout',`model-io-${sid}.jsonl`),path.join('C:/Users/Sandy/.zcode/cli/rollout',`model-io-${sid}.jsonl`)]; + const f=candidates.find(x=>existsSync(x));if(!f)return {available:false,reason:'owned model-I/O file not found, possibly rotated'}; + const records=[];for(const line of readFileSync(f,'utf8').split('\n')){if(!line.trim())continue;let x;try{x=JSON.parse(line);}catch{continue;}if(x.sessionId!==sid)continue;const b=x.request?.body||{};const msgs=x.request?.messages||[]; + records.push({sessionId:sid,traceId:x.traceId,turnId:x.turnId,requestId:x.requestId,attempt:x.attempt,querySource:x.querySource,model:x.model, + bodyKeys:Object.keys(b),modelInBody:b.model,thinking:b.thinking,outputConfig:b.output_config,maxTokens:b.max_tokens, + systemHash:b.system?hash(b.system):null,systemLength:b.system?JSON.stringify(b.system).length:null, + systemSegmentDigests:Array.isArray(b.system)?b.system.map(s=>({type:s.type,hash:hash(s.text||''),length:s.text?.length||0})):null, + toolsHash:b.tools?hash(b.tools):null,toolNames:x.request?.toolNames||null,messageCount:x.request?.messageCount, + messagesKind:x.request?.messagesKind,messageOffset:x.request?.messageOffset, + messageDigests:msgs.map(m=>({role:m.role,hash:hash(m.content),length:JSON.stringify(m.content)?.length||0})),usage:x.response?.usage||null});} + return {available:true,records}; +} +if(process.argv[2]){const value=modelIO(process.argv[2]);if(process.argv[3])writeFileSync(process.argv[3],JSON.stringify(value,null,2));else console.log(JSON.stringify(value,null,2));} diff --git a/docs/research/native-cli-vs-appserver-2026-10-05/experiments/verify-artifacts.mjs b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/verify-artifacts.mjs new file mode 100644 index 0000000..7e842e6 --- /dev/null +++ b/docs/research/native-cli-vs-appserver-2026-10-05/experiments/verify-artifacts.mjs @@ -0,0 +1,30 @@ +// Read-only artifact validation. Does not import probe or call a model. +import {readFileSync,readdirSync,existsSync} from 'node:fs'; +import path from 'node:path'; +import {fileURLToPath} from 'node:url'; +import assert from 'node:assert/strict'; +const root=fileURLToPath(new URL('../',import.meta.url)); +const base=process.env.AUDIT_ROOT||'C:/Users/Sandy/.codex/tmp/zcode-executor-audit-20261005'; +const official=path.join(base,'official'); +const required=['ZCODE_CLI_VS_APPSERVER.md','ZCODE_RUNTIME_CALLCHAIN.md','ZCODE_SESSION_HANDOFF.md','ZCODE_EXECUTOR_BENCHMARK.md','ZCODE_EXECUTOR_ARCHITECTURE_RECOMMENDATION.md']; +const files=required.map(f=>path.join(root,f));files.push(path.join(root,'experiments/README.md')); +let links=0; +for(const file of files){const text=readFileSync(file,'utf8');assert.ok(text.length>500,file); + for(const m of text.matchAll(/\]\(([^)]+)\)/g)){ + const target=m[1];if(target.startsWith('https://github.com/zai-org/ZCode/blob/29628c9acdb81b703bbd4080c207a0e7ce5e276e/')){ + const relative=target.split('29628c9acdb81b703bbd4080c207a0e7ce5e276e/')[1].split('#')[0];const source=path.join(official,relative);assert.ok(existsSync(source),'missing source '+relative); + const line=Number(target.match(/#L(\d+)/)?.[1]||0);if(line)assert.ok(readFileSync(source,'utf8').split('\n').length>=line,'invalid anchor '+target);links++; + }else if(!/^(https?:|#)/.test(target)){assert.ok(existsSync(path.resolve(path.dirname(file),target.split('#')[0])),'missing local link '+target);links++;} + } +} +const metrics=['executor','sessionId','runId','traceId','model','provider','reasoningLevel','mode','wallTimeMs','turnCount','modelRequestCount','inputTokens','cachedInputTokens','outputTokens','totalTokens','toolCallCount','toolSequence','permissionRequestCount','subagentCount','workflowCount','filesChanged','linesAdded','linesDeleted','testsPassed','testsFailed','acceptanceCriteriaPassed','retryCount','finalStatus','qualityScore']; +for(const [name,completed]of [['benchmark-observed',10],['benchmark-controlled-failed',0]]){ + const ledger=JSON.parse(readFileSync(path.join(root,'evidence',name+'.json'),'utf8'));assert.equal(ledger.pairs.length,10);assert.equal(ledger.pairs.filter(r=>r.executionCompleted).length,completed); + for(const r of ledger.pairs){for(const k of metrics)assert.ok(k in r.metrics,r.runId+' missing '+k);assert.equal(r.commit,ledger.commit);assert.equal(r.workspace,ledger.cwd);if(completed){assert.equal(r.acceptance.passed,true);assert.equal(r.metrics.qualityScore,4);assert.equal(r.metrics.model[0],'GLM-5.3-Flash');}} + for(const id of ['T1','T2','T3','T4','T5']){const pairs=ledger.pairs.filter(r=>r.task===id);assert.equal(pairs.length,2);assert.equal(pairs[0].promptHash,pairs[1].promptHash);} +} +let objects=0; +const prohibited=new Set(['requestHeaders','responseHeaders','apiKey','accessToken','refreshToken','requestAuth','credentials']); +function walk(x){if(!x||typeof x!=='object')return;objects++;for(const[k,v]of Object.entries(x)){assert.ok(!prohibited.has(k),'sensitive key exported '+k);walk(v);}} +for(const name of readdirSync(path.join(root,'evidence'))){if(name.endsWith('.json'))walk(JSON.parse(readFileSync(path.join(root,'evidence',name),'utf8')));} +console.log(JSON.stringify({requiredDocuments:required.length,sourceAndLocalLinksChecked:links,observedRuns:10,controlledFailedRuns:10,privacyObjectsChecked:objects,result:'PASS'})); diff --git a/probe/evidence/run-001-summary.json b/probe/evidence/run-001-summary.json new file mode 100644 index 0000000..39e9a8f --- /dev/null +++ b/probe/evidence/run-001-summary.json @@ -0,0 +1,146 @@ +{ + "evidence_kind": "sanitized runtime probe summary", + "observed_at": "2026-10-05", + "runtime_version": "0.16.9", + "bridge_commit": "2596759198fa826c2b7ac0478c5682da996e9727", + "os": "Windows", + "node_version": "v24.16.0", + "provider_id": "account:bigmodel-individual-coding-plan", + "model_id": "GLM-5.3", + "reasoning_level": "max", + "rpc_probes": { + "runtime/capabilities": { + "ok": true, + "returned": { + "independentPlanState": true + } + }, + "session/create": { + "ok": true, + "snapshot_groups": [ + "messages", + "projection", + "protocol", + "runtime", + "session", + "settings", + "slashCommands", + "todoGroups", + "todos" + ], + "session_fields": [ + "createdAt", + "mode", + "model", + "sessionId", + "sessionKind", + "status", + "target", + "title", + "traceId", + "updatedAt", + "workspace" + ], + "runtime_fields": [ + "eventSeq", + "goalVerificationTimeline", + "goalVerifications", + "pendingRequestIds", + "stateRevision" + ], + "settings_fields": [ + "mode", + "model", + "permission", + "thoughtLevel" + ] + }, + "session/read_same_process": { + "ok": true, + "status_after_completion": "idle", + "event_seq_after_completion": 261, + "pending_request_count_after_completion": 0 + }, + "session/events_same_process": { + "ok": true, + "after_seq_and_limit": "accepted" + }, + "session/subagents_empty_session": { + "ok": false, + "error_code": -32004, + "interpretation": "no subagent data from this empty-session probe; not proof of unsupported capability" + }, + "session/list": { + "ok": true, + "returned_entries": 50 + } + }, + "normal_task": { + "scope": "asked runtime to create and verify one line in a temporary isolated workspace", + "send_accepted": true, + "host_side_artifact_check": { + "file": "probe.txt", + "exists_after_run": false, + "interpretation": "runtime turn success did not establish objective satisfaction" + }, + "terminal_event": { + "type": "turn.completed", + "sequence": 261, + "resultType": "success" + }, + "live_event_count": 261, + "event_counts": { + "session.titleUpdated": 1, + "turn.started": 1, + "session.updated": 36, + "model.streaming": 177, + "tool.updated": 26, + "permission.requested": 5, + "permission.resolved": 5, + "streamRecovery.updated": 9, + "turn.completed": 1 + }, + "observed_model_stream_kinds": [ + "reasoning_delta", + "text_delta", + "tool_input_start", + "tool_input_delta", + "tool_input_end", + "tool_call" + ], + "observed_tool_name_example": "Write", + "tool_call_count_from_terminal_metadata": 9, + "permission_human_roundtrip": false, + "user_input_request": false, + "phase_or_progress_event": false, + "failure_or_cancellation_path": false, + "retry_semantics_established": false + }, + "reconnect_probe": { + "kind": "new idle session in unique temporary workspace", + "initial_status": "idle", + "initial_event_seq": 0, + "session_read_after_process_restart": { + "ok": false, + "error_code": -32004 + }, + "session_events_after_process_restart": { + "ok": false, + "error_code": -32004 + }, + "session_list_found_exact_session": false, + "session_resume_exact_session": { + "ok": false, + "error_code": -32004 + }, + "active_or_completed_task_recovery_tested": false + }, + "privacy": { + "model_text_saved": false, + "reasoning_text_saved": false, + "tool_arguments_saved": false, + "headers_or_credentials_saved": false, + "raw_rpc_dump_saved": false + }, + "caveat": "Single runtime version, provider/model path, successful task, and empty-session reconnect attempt; not a compatibility guarantee." +}