Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
50 changes: 50 additions & 0 deletions apps/desktop/src/main/__tests__/executor-selection.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -28,6 +28,54 @@ import { useExecutorSelection, newTaskConfiguration, ConversationServicesProvide

const entry: ExecutorCatalogEntry = { id: 'external', displayName: 'External', readiness: 'ready', models: [{ id: 'selected', name: 'Selected' }], supportsAttachments: false, supportsModelChange: true };

test('explicit restore confirms the saved model while preserving the current task', async () => {
const { document, window } = parseHTML('<html><body><div id="root"></div></body></html>');
const values = { document, window, HTMLElement: window.HTMLElement, Node: window.Node, IS_REACT_ACT_ENVIRONMENT: true };
const originals = new Map(Object.keys(values).map(key => [key, Object.getOwnPropertyDescriptor(globalThis, key)]));
for (const [key, value] of Object.entries(values)) Object.defineProperty(globalThis, key, { configurable: true, writable: true, value });
const root = createRoot(document.getElementById('root')!);
let latest!: ReturnType<typeof useExecutorSelection>;
let ready = false;
let finishRestore!: () => void;
const confirmation = new Promise<void>((resolve) => { finishRestore = resolve; });
const writes: Array<{ sessionId: string; model: string | undefined }> = [];
const services = {
subscribeChanges: () => () => {},
newTasks: { subscribeChanges: () => () => {}, getExecutors: async () => [entry] },
sessions: {
getExecutorState: async () => [{ ...entry, readiness: ready ? 'ready' : 'restorable', currentModel: 'selected' }],
setExecutorModelConfiguration: async (sessionId: string, config: { model?: string }) => {
writes.push({ sessionId, model: config.model });
await confirmation;
ready = true;
return { ok: true, session: { executorConfig: config } };
},
},
} as unknown as ConversationServices;
function Probe() {
latest = useExecutorSelection({
key: 'saved', cwd: '/fixture',
target: { hostId: 'host', profileId: 'profile', projectId: null },
session: { id: 'saved', executorId: 'external', executorConfig: { model: 'selected' } } as SessionSummary,
});
return null;
}
try {
await act(async () => root.render(createElement(ConversationServicesProvider, { services, children: createElement(Probe) })));
assert.equal(latest.entry?.readiness, 'restorable');
let restoring!: Promise<void>;
await act(async () => { restoring = latest.restore(); });
assert.equal(latest.entry?.readiness, 'restoring');
await act(async () => { finishRestore(); await restoring; });
assert.deepEqual(writes, [{ sessionId: 'saved', model: 'selected' }]);
assert.equal(latest.entry?.readiness, 'ready');
assert.equal(latest.selection?.executorId, 'external');
} finally {
await act(async () => root.unmount());
for (const [key, descriptor] of originals) { if (descriptor) Object.defineProperty(globalThis, key, descriptor); else Reflect.deleteProperty(globalThis, key); }
}
});

test('a catalog change during failed discovery retries after the in-flight result settles', async () => {
const { document, window } = parseHTML('<html><body><div id="root"></div></body></html>');
const values = { document, window, HTMLElement: window.HTMLElement, Node: window.Node, IS_REACT_ACT_ENVIRONMENT: true };
Expand Down Expand Up @@ -147,6 +195,8 @@ test('late draft discovery cannot replace the current target; existing tasks ins
assert.equal(latest.changing, true);
await act(async () => {
await assert.rejects(latest.select({ executorId: 'external', configuration: { model: 'other' } }), /pending/);
await assert.rejects(latest.restore(), /pending/);
assert.equal(latest.entry?.readiness, 'ready', 'a rejected restore is not shown as active');
serverModel = 'fast';
confirm({ ok: true, session: { executorConfig: { model: 'fast' } } });
await pending;
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -185,11 +185,29 @@ export function useExecutorSelection(input: {
}
};
const entry = catalog.find((candidate) => candidate.id === selection?.executorId);
const restore = async () => {
if (!sessionId || !executorId) throw new Error('Executor Session is unavailable');
if (inFlight.current === key) throw new Error('Executor configuration is pending');
const model = input.session?.executorConfig?.model ?? inspected?.currentModel;
if (!model) throw new Error('Executor model is unavailable');
setSnapshot((previous) =>
previous?.key === key
? {
...previous,
catalog: previous.catalog.map((entry) =>
entry.id === executorId ? { ...entry, readiness: 'restoring' } : entry,
),
}
: previous,
);
await select({ executorId, configuration: { model } });
};
return {
selection,
catalog,
entry,
select,
restore,
refresh,
changing: changingKey === key,
loading: snapshot?.key !== key || snapshot.loading,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -43,6 +43,7 @@ export function executorComposerProps(
loading: executor.loading,
error: executor.error,
onSelect: (selection) => executor.select(selection),
onRestore: () => executor.restore(),
onRetry: () => {
void executor.refresh();
},
Expand Down
26 changes: 24 additions & 2 deletions docs/antigravity-acp-plugin-rebuild.md
Original file line number Diff line number Diff line change
Expand Up @@ -69,8 +69,9 @@ policy, initial model configuration, and Antigravity question/failure recognitio
- cancellation drain with the actual provider stop reason retained in the durable runtime ledger;
- bounded, cached model discovery and Agent-confirmed idle model changes;
- generic initial ACP configuration validation/application;
- durable history-only detection so a Host/Plugin restart cannot silently fork an existing external
conversation into a new ACP Session.
- Plugin-private external Session identity and write-ahead prompt state; a clean
completed task can resume the same ACP Session after a Host/Plugin restart,
while uncertain history stays readable without creating a replacement Session.

The Antigravity adapter owns:

Expand All @@ -83,6 +84,27 @@ The Host owns only executor visibility and binding, Maka Session/run identity, c
persistence, hosted form admission, and Plugin retirement. No ACP process, credential, or external
Session identifier crosses the Plugin boundary.

## Restoring an existing task

After a restart, open the Antigravity task and choose **Restore** in the model
menu or the notice beside the composer. Maka starts the Agent only when you
choose to restore or continue. The Agent must support ACP `session/resume`; a
successful restore uses the original external Session and keeps the saved task
and model. No earlier prompt is sent again.

If restoration fails, the task history remains readable. Check the installed
Agent and helper, Google sign-in, and network access in **Settings → External
Agents**, then retry Restore. Maka will not create a new external Session for
that task. An executable or helper change also requires a new task because the
saved Session is bound to the previous installation.

**History gap** means a prompt may have reached the Agent before Maka finished
saving its terminal event. Maka inspects ACP load replay separately and does
not append possibly duplicate output to the saved history or resend that
prompt. Start a new task if you need to continue working. Tasks created before
this restoration version have no saved external Session ID and remain
history-only.

## Deliberately not ported

The rebuild does not carry forward the PR #5224 provider catalog protocol, external-Agent Session
Expand Down
6 changes: 3 additions & 3 deletions docs/archive/antigravity-acp-pr2-acceptance.md
Original file line number Diff line number Diff line change
Expand Up @@ -287,7 +287,7 @@ The earlier completed-task capture remains
[historical execution evidence](https://github.com/user-attachments/assets/4fc2827b-feed-4639-9a6a-d50caf203055),
not a claim about the current picker or a repeated coding run.

Public CI, independent human approval and merge remain repository gates. This document does not
mark issue #5103's PR 2 checkbox complete. Unknown future label shapes remain raw models until
Public CI and independent human approval passed before PR 2 merged as #5224 on 2026-09-23.
Issue #5103's PR 2 checklist is complete. Unknown future label shapes remain raw models until
verified. Cross-process restoration belongs to PR 3; modes and expanded catalog lifecycle belong
to PR 4. Neither is claimed by this PR.
to PR 4. Neither is claimed by PR 2.
148 changes: 148 additions & 0 deletions docs/archive/antigravity-acp-pr3-acceptance.md
Original file line number Diff line number Diff line change
@@ -0,0 +1,148 @@
<!--
Licensed to the Apache Software Foundation (ASF) under one
or more contributor license agreements. See the NOTICE file
distributed with this work for additional information
regarding copyright ownership. The ASF licenses this file
to you under the Apache License, Version 2.0 (the
"License"); you may not use this file except in compliance
with the License. You may obtain a copy of the License at

http://www.apache.org/licenses/LICENSE-2.0

Unless required by applicable law or agreed to in writing,
software distributed under the License is distributed on an
"AS IS" BASIS, WITHOUT WARRANTIES OR CONDITIONS OF ANY
KIND, either express or implied. See the License for the
specific language governing permissions and limitations
under the License.
-->

# Antigravity ACP PR 3 acceptance

PR 3 is tracked by [issue #5103](https://github.com/apache/maka/issues/5103).
The results below are from a new probe, separate from the successful PR 2
execution and earlier cross-process feasibility checks in
[the PR 2 record](antigravity-acp-pr2-acceptance.md).

## Real Agent prerequisite probe, 2026-09-24

- Official macOS arm64 `agy_acp_server.par` **1.1.1** with its matching
`localharness_external`, ACP SDK **1.4.0**.
- The probe used a temporary toy project with a randomly generated synthetic
token. An isolated temporary home copied only the existing OAuth token and
ACP settings into mode-restricted files, but stalled before `session/new`
completed. The method results below used the existing authenticated Agent
home with the toy project. No credentials, Session IDs, token values, or
project contents were recorded.
- `initialize` returned protocol version 1, `loadSession: true`, and
`sessionCapabilities.resume`. `session/new` returned an external Session ID
and confirmed `gemini-3.7-flash-high`.
- A fresh process accepted `session/resume` with the same ID and project
directory and returned the same model. A second independent process also
accepted `session/resume` for that ID. `session/load` accepted it and sent
two `user_message_chunk` notifications and an
`available_commands_update`. No successful Agent output existed to replay
in this run.
- The first prompt and post-resume prompt each ended with `end_turn` but
emitted an Agent execution error: the model request returned HTTP 403,
reporting that the account was not eligible for Gemini Code Assist for
individuals in the current location. The temporary profile with and without
explicit HTTP(S) proxy variables stalled before Session creation. Direct access with
the machine's SOCKS proxy configuration failed because the bundled Python
lacked `python-socks`. Explicit local HTTP(S) proxy variables allowed
Session creation but produced the same model 403.
- All probe-owned Agent process groups and temporary profiles were removed.

This initial probe **did not pass the PR 3 prerequisite gate**. It confirms
cross-process method acceptance, but cannot establish continued model context,
successful replay shape, duplicate or reordered replay behavior, or a crash
window where the Agent advanced beyond Maka's durable history. The 2026-09-12
PR 2 feasibility probe observed one successful output chunk and tool replay;
it did not send a post-resume prompt or establish reconciliation semantics.
The later route-change probe below supplied the missing successful turns.

## Post-login recheck, 2026-09-24

After the user completed login in a locally built Maka, the official 1.1.1
Agent was probed again with the authenticated home and a new temporary toy
project. It negotiated the same resume/load capabilities, created a Session,
and confirmed `gemini-3.7-flash-high`. After a synthetic-token prompt, a fresh
process resumed the same Session ID and accepted a follow-up asking for that
token. Both prompts emitted the same location-eligibility HTTP 403 Agent error,
so the follow-up did not recall the token. A third process loaded the Session;
it replayed two user-message chunks and an available-commands update, with no
successful assistant or tool output to reconcile. The configured local HTTP(S)
proxy's observed egress region was Singapore. All probe processes and the toy
project were removed. Successful login therefore has not cleared the real-model
prerequisite gate.

## Route-change verification, 2026-09-24

After the user changed the proxy exit, a new probe used the same official
macOS arm64 1.1.1 server and matching helper, the user's authenticated Agent
home, and a temporary toy project. Synthetic random values were used only to
test context; no token values, credentials, external Session IDs, or private
project content are recorded here. Intermittent HTTP 403 responses still
occurred between successful requests, so the probe retried affected tool turns.

- `initialize` again negotiated protocol 1, `loadSession: true`, and
`sessionCapabilities.resume`. `session/new` returned a Session ID and
`gemini-3.7-flash-high`.
- A successful prompt gave the Agent one synthetic token. A new process used
`session/resume` with that same ID and project path, returned the same model,
and a follow-up answer recalled the exact token. The Session was not
replaced and the prompt was not resent.
- A successful tool turn read a second synthetic token from a toy file. The
file was removed, then a new process used `session/load` and replayed user,
Agent, and tool updates. A follow-up recalled the file-only token after the
file was gone, proving that tool context survived the process change.
- Two independent `session/load` calls produced the same update counts and
replay tool IDs for the completed history. A failed tool turn showed that
live tool IDs can differ from replayed IDs. Repeated identical user prompts
appeared as distinct replayed user chunks. Thus neither prompt text nor
live tool ID alone is a safe general deduplication key.
- The probe terminated all child process groups and removed its toy project.

These observations support `session/resume` for a fully committed Maka turn.
For an interrupted turn, Maka captures `session/load` replay separately and
reports a history gap. It does not append unaligned replay to the canonical
conversation or submit another prompt. This is a conservative reconciliation
decision because the official replay did not provide stable canonical event
identities across every observed turn. The user can read the saved history and
start a new task; the external Session ID is never silently replaced.

## Built Plugin smoke, 2026-09-24

The built production ACP Runtime and Antigravity adapter bundles were loaded
with the official 1.1.1 executable/helper and the authenticated Agent home.
A temporary toy task completed a synthetic-token prompt. The Runtime
acknowledged that turn, disposed its process, and a fresh `AcpExecutor`
instance reported `restorable` from the saved Plugin-private record. The
explicit restore changed readiness to `ready`; a follow-up prompt completed
and recalled the synthetic token. The test removed the toy project and did
not print the token or external Session ID. This exercises the shipped
Plugin code path in addition to the direct protocol probe. It does not stand
in for a full Desktop UI restart test.

## Implementation checks

On the branch tested at `bd8661f3a` and rebased without conflicts to
`b62ca805e` before opening the draft PR:

- `npm run build`, `npm run typecheck`, `npm run lint`, and
`npm run format:check` passed.
- Renderer architecture, locale hygiene, and the protocol compatibility epoch
guard passed; the epoch advanced from 183 to 184 for the new readiness values.
- Every workspace test suite passed in a final serial run using
`node scripts/run-workspace-tests-parallel.mjs --concurrency=1`. Controlled
ACP tests include a forcibly killed, durably acknowledged child process
restored under the same external Session in a fresh process; uncertain load
replay remains outside the canonical turn.
- An earlier serial run had one intermittent Runtime Host Goal handoff timing
assertion. The full Host workspace passed independently, that exact test
passed in isolation, and the final serial run passed. The failing test does
not touch ACP restoration.

A full Desktop UI restart with a signed-in real Agent was not executed. The
production Plugin bundle smoke, Desktop picker/controller tests, and the real
Agent protocol probes cover the corresponding layers separately.
Loading