Skip to content

Commit 1ab9b79

Browse files
committed
feat: standard tools and software engineer agent
- Implemented standard file system and shell tools in `src/runner/standard-tools.ts` - Added `software-engineer` agent definition - Added integration tests for standard tools - Updated `keystone-architect` agent - Added `memory-service` and `robust-automation` templates - Updated documentation and schema for new tools
1 parent 7f92c75 commit 1ab9b79

15 files changed

Lines changed: 749 additions & 20 deletions

‎README.md‎

Lines changed: 34 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -260,6 +260,23 @@ finally:
260260
type: shell
261261
run: echo "Workflow finished"
262262
263+
### Expression Syntax
264+
265+
Keystone uses `${{ }}` syntax for dynamic values. Expressions are evaluated using a safe AST parser.
266+
267+
- `${{ inputs.name }}`: Access workflow inputs.
268+
- `${{ steps.id.output }}`: Access the raw output of a previous step.
269+
- `${{ steps.id.outputs.field }}`: Access specific fields if the output is an object.
270+
- `${{ steps.id.status }}`: Get the execution status of a step (`'success'`, `'failed'`, etc.).
271+
- `${{ item }}`: Access the current item in a `foreach` loop.
272+
- `${{ args.name }}`: Access tool arguments (available ONLY inside agent tool execution steps).
273+
- `${{ secrets.NAME }}`: Access redacted secrets.
274+
- `${{ env.NAME }}`: Access environment variables.
275+
276+
Standard JavaScript-like expressions are supported: `${{ steps.build.status == 'success' ? '🚀' : '❌' }}`.
277+
278+
---
279+
263280
outputs:
264281
slack_message: ${{ steps.notify.output }}
265282
```
@@ -274,8 +291,11 @@ Keystone supports several specialized step types:
274291
- `llm`: Prompt an agent and get structured or unstructured responses. Supports `schema` (JSON Schema) for structured output.
275292
- `allowClarification`: Boolean (default `false`). If `true`, allows the LLM to ask clarifying questions back to the user or suspend the workflow if no human is available.
276293
- `maxIterations`: Number (default `10`). Maximum number of tool-calling loops allowed for the agent.
294+
- `allowInsecure`: Boolean (default `false`). Set `true` to allow risky tool execution.
295+
- `allowOutsideCwd`: Boolean (default `false`). Set `true` to allow tools to access files outside of the current working directory.
277296
- `request`: Make HTTP requests (GET, POST, etc.).
278297
- `file`: Read, write, or append to files.
298+
- `allowOutsideCwd`: Boolean (default `false`). Set `true` to allow reading/writing files outside of the current working directory.
279299
- `human`: Pause execution for manual confirmation or text input.
280300
- `inputType: confirm`: Simple Enter-to-continue prompt.
281301
- `inputType: text`: Prompt for a string input, available via `${{ steps.id.output }}`.
@@ -352,6 +372,8 @@ You are a technical communications expert. Your goal is to take technical output
352372

353373
Agents can be equipped with tools, which are essentially workflow steps they can choose to execute. You can define tools in the agent definition, or directly in an LLM step within a workflow.
354374

375+
Tool arguments are passed to the tool's execution step via the `args` variable.
376+
355377
**`.keystone/workflows/agents/developer.md`**
356378
```markdown
357379
---
@@ -363,6 +385,18 @@ tools:
363385
id: list-files-tool
364386
type: shell
365387
run: ls -F
388+
- name: read_file
389+
description: Read a specific file
390+
parameters:
391+
type: object
392+
properties:
393+
path: { type: string }
394+
required: [path]
395+
execution:
396+
id: read-file-tool
397+
type: file
398+
op: read
399+
path: ${{ args.path }}
366400
---
367401
You are a software developer. You can use tools to explore the codebase.
368402
```

‎package.json‎

Lines changed: 1 addition & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -1,6 +1,6 @@
11
{
22
"name": "keystone-cli",
3-
"version": "0.6.1",
3+
"version": "0.7.0",
44
"description": "A local-first, declarative, agentic workflow orchestrator built on Bun",
55
"type": "module",
66
"bin": {

‎src/expression/evaluator.ts‎

Lines changed: 2 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -29,6 +29,7 @@ export interface ExpressionContext {
2929
secrets?: Record<string, string>;
3030
steps?: Record<string, { output?: unknown; outputs?: Record<string, unknown>; status?: string }>;
3131
item?: unknown;
32+
args?: unknown;
3233
index?: number;
3334
env?: Record<string, string>;
3435
output?: unknown;
@@ -295,6 +296,7 @@ export class ExpressionEvaluator {
295296
secrets: context.secrets || {},
296297
steps: context.steps || {},
297298
item: context.item,
299+
args: context.args,
298300
index: context.index,
299301
env: context.env || {},
300302
stdout: contextAsRecord.stdout, // For transform expressions

‎src/parser/schema.ts‎

Lines changed: 3 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -95,6 +95,9 @@ const LlmStepSchema = BaseStepSchema.extend({
9595
])
9696
)
9797
.optional(),
98+
useStandardTools: z.boolean().optional(),
99+
allowOutsideCwd: z.boolean().optional(),
100+
allowInsecure: z.boolean().optional(),
98101
});
99102

100103
const WorkflowStepSchema = BaseStepSchema.extend({

‎src/runner/llm-executor.ts‎

Lines changed: 39 additions & 3 deletions
Original file line numberDiff line numberDiff line change
@@ -9,13 +9,14 @@ import { RedactionBuffer, Redactor } from '../utils/redactor';
99
import { type LLMMessage, getAdapter } from './llm-adapter';
1010
import { MCPClient } from './mcp-client';
1111
import type { MCPManager, MCPServerConfig } from './mcp-manager';
12+
import { STANDARD_TOOLS, validateStandardToolSecurity } from './standard-tools';
1213
import type { StepResult } from './step-executor';
1314

1415
interface ToolDefinition {
1516
name: string;
1617
description?: string;
1718
parameters: unknown;
18-
source: 'agent' | 'step' | 'mcp';
19+
source: 'agent' | 'step' | 'mcp' | 'standard';
1920
execution?: Step;
2021
mcpClient?: MCPClient;
2122
}
@@ -105,7 +106,24 @@ export async function executeLlmStep(
105106
}
106107
}
107108

108-
// 3. Add MCP tools
109+
// 3. Add Standard tools
110+
if (step.useStandardTools) {
111+
for (const tool of STANDARD_TOOLS) {
112+
allTools.push({
113+
name: tool.name,
114+
description: tool.description,
115+
parameters: tool.parameters || {
116+
type: 'object',
117+
properties: {},
118+
additionalProperties: true,
119+
},
120+
source: 'standard',
121+
execution: tool.execution,
122+
});
123+
}
124+
}
125+
126+
// 4. Add MCP tools
109127
const mcpServersToConnect: (string | MCPServerConfig)[] = [...(step.mcpServers || [])];
110128
if (step.useGlobalMcp && mcpManager) {
111129
const globalServers = mcpManager.getGlobalServers();
@@ -374,10 +392,28 @@ export async function executeLlmStep(
374392
});
375393
}
376394
} else if (toolInfo.execution) {
395+
// Security validation for standard tools
396+
if (toolInfo.source === 'standard') {
397+
try {
398+
validateStandardToolSecurity(toolInfo.name, args, {
399+
allowOutsideCwd: step.allowOutsideCwd,
400+
allowInsecure: step.allowInsecure,
401+
});
402+
} catch (error) {
403+
messages.push({
404+
role: 'tool',
405+
tool_call_id: toolCall.id,
406+
name: toolCall.function.name,
407+
content: `Security Error: ${error instanceof Error ? error.message : String(error)}`,
408+
});
409+
continue;
410+
}
411+
}
412+
377413
// Execute the tool as a step
378414
const toolContext: ExpressionContext = {
379415
...context,
380-
item: args, // Use item to pass args to tool execution
416+
args, // Use args to pass parameters to tool execution
381417
};
382418

383419
const result = await executeStepFn(toolInfo.execution, toolContext);

‎src/runner/shell-executor.ts‎

Lines changed: 40 additions & 12 deletions
Original file line numberDiff line numberDiff line change
@@ -136,14 +136,11 @@ export async function executeShell(
136136
const cwd = step.dir ? ExpressionEvaluator.evaluateString(step.dir, context) : undefined;
137137
const mergedEnv = Object.keys(env).length > 0 ? { ...Bun.env, ...env } : Bun.env;
138138

139-
// Safe Fast Path: If command contains only safe characters (alphanumeric, -, _, ., /) and spaces,
140-
// we can split it and execute directly without a shell.
141-
// This completely eliminates shell injection risks for simple commands.
142-
const isSimpleCommand = /^[a-zA-Z0-9_\-./]+(?: [a-zA-Z0-9_\-./]+)*$/.test(command);
139+
// Shell metacharacters that require a real shell
140+
const hasShellMetas = /[|&;<>`$!]/.test(command);
143141

144142
// Common shell builtins that must run in a shell
145-
const splitArgs = command.split(/\s+/);
146-
const cmd = splitArgs[0];
143+
const firstWord = command.trim().split(/\s+/)[0];
147144
const isBuiltin = [
148145
'exit',
149146
'cd',
@@ -155,19 +152,50 @@ export async function executeShell(
155152
'unalias',
156153
'eval',
157154
'set',
158-
].includes(cmd);
155+
'true',
156+
'false',
157+
].includes(firstWord);
158+
159+
const canUseSpawn = !hasShellMetas && !isBuiltin;
159160

160161
try {
161162
let stdoutString = '';
162163
let stderrString = '';
163164
let exitCode = 0;
164165

165-
if (isSimpleCommand && !isBuiltin) {
166-
// split by spaces
167-
const args = splitArgs.slice(1);
168-
if (!cmd) throw new Error('Empty command');
166+
if (canUseSpawn) {
167+
// Robust splitting that handles single and double quotes
168+
const args: string[] = [];
169+
let current = '';
170+
let inQuote = false;
171+
let quoteChar = '';
172+
173+
for (let i = 0; i < command.length; i++) {
174+
const char = command[i];
175+
if ((char === "'" || char === '"') && (i === 0 || command[i - 1] !== '\\')) {
176+
if (inQuote && char === quoteChar) {
177+
inQuote = false;
178+
quoteChar = '';
179+
} else if (!inQuote) {
180+
inQuote = true;
181+
quoteChar = char;
182+
} else {
183+
current += char;
184+
}
185+
} else if (/\s/.test(char) && !inQuote) {
186+
if (current) {
187+
args.push(current);
188+
current = '';
189+
}
190+
} else {
191+
current += char;
192+
}
193+
}
194+
if (current) args.push(current);
195+
196+
if (args.length === 0) throw new Error('Empty command');
169197

170-
const proc = Bun.spawn([cmd, ...args], {
198+
const proc = Bun.spawn(args, {
171199
cwd,
172200
env: mergedEnv,
173201
stdout: 'pipe',
Lines changed: 147 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,147 @@
1+
import { afterAll, beforeAll, describe, expect, it, mock, spyOn } from 'bun:test';
2+
import type { ExpressionContext } from '../expression/evaluator';
3+
import type { LlmStep, Step } from '../parser/schema';
4+
import { ConsoleLogger } from '../utils/logger';
5+
import { OpenAIAdapter } from './llm-adapter';
6+
import { executeLlmStep } from './llm-executor';
7+
8+
describe('Standard Tools Integration', () => {
9+
const originalOpenAIChat = OpenAIAdapter.prototype.chat;
10+
11+
beforeAll(() => {
12+
// Mocking OpenAI Adapter
13+
});
14+
15+
afterAll(() => {
16+
OpenAIAdapter.prototype.chat = originalOpenAIChat;
17+
});
18+
19+
it('should inject standard tools when useStandardTools is true', async () => {
20+
// biome-ignore lint/suspicious/noExplicitAny: mock
21+
let capturedTools: any[] = [];
22+
23+
OpenAIAdapter.prototype.chat = mock(async (messages, options) => {
24+
capturedTools = options.tools || [];
25+
return {
26+
message: {
27+
role: 'assistant',
28+
content: 'I will read the file',
29+
tool_calls: [
30+
{
31+
id: 'call_1',
32+
type: 'function',
33+
function: {
34+
name: 'read_file',
35+
arguments: JSON.stringify({ path: 'test.txt' }),
36+
},
37+
},
38+
],
39+
},
40+
usage: { prompt_tokens: 10, completion_tokens: 10, total_tokens: 20 },
41+
// biome-ignore lint/suspicious/noExplicitAny: mock
42+
} as any;
43+
});
44+
45+
const step: LlmStep = {
46+
id: 'l1',
47+
type: 'llm',
48+
agent: 'test-agent',
49+
needs: [],
50+
prompt: 'read test.txt',
51+
useStandardTools: true,
52+
maxIterations: 1,
53+
};
54+
55+
const context: ExpressionContext = { inputs: {}, steps: {} };
56+
const executeStepFn = mock(async (s: Step) => {
57+
return { status: 'success', output: 'file content' };
58+
});
59+
60+
// We catch the "Max iterations reached" error because we set maxIterations to 1
61+
// but we can still check if tools were injected and the tool call was made.
62+
try {
63+
// biome-ignore lint/suspicious/noExplicitAny: mock
64+
await executeLlmStep(step, context, executeStepFn as any);
65+
} catch (e) {
66+
if ((e as Error).message !== 'Max ReAct iterations reached') throw e;
67+
}
68+
69+
expect(capturedTools.some((t) => t.function.name === 'read_file')).toBe(true);
70+
expect(executeStepFn).toHaveBeenCalled();
71+
const toolStep = executeStepFn.mock.calls[0][0] as Step;
72+
expect(toolStep.type).toBe('file');
73+
});
74+
75+
it('should block risky standard tools without allowInsecure', async () => {
76+
OpenAIAdapter.prototype.chat = mock(async (messages, options) => {
77+
return {
78+
message: {
79+
role: 'assistant',
80+
content: 'I will run a command',
81+
tool_calls: [
82+
{
83+
id: 'call_2',
84+
type: 'function',
85+
function: {
86+
name: 'run_command',
87+
arguments: JSON.stringify({ command: 'rm -rf /' }),
88+
},
89+
},
90+
],
91+
},
92+
usage: { prompt_tokens: 10, completion_tokens: 10, total_tokens: 20 },
93+
// biome-ignore lint/suspicious/noExplicitAny: mock
94+
} as any;
95+
});
96+
97+
const step: LlmStep = {
98+
id: 'l1',
99+
type: 'llm',
100+
agent: 'test-agent',
101+
needs: [],
102+
prompt: 'run risky command',
103+
useStandardTools: true,
104+
allowInsecure: false, // Explicitly false
105+
maxIterations: 2,
106+
};
107+
108+
const context: ExpressionContext = { inputs: {}, steps: {} };
109+
const executeStepFn = mock(async () => ({ status: 'success', output: '' }));
110+
111+
// The execution should not throw, but it should return a tool error message to the LLM
112+
// However, in our mock, we want to see if executeStepFn was called.
113+
// Actually, in llm-executor.ts, it pushes a "Security Error" message if check fails and continues loop.
114+
115+
let securityErrorMessage = '';
116+
OpenAIAdapter.prototype.chat = mock(async (messages) => {
117+
const lastMessage = messages[messages.length - 1];
118+
if (lastMessage.role === 'tool') {
119+
securityErrorMessage = lastMessage.content;
120+
return {
121+
message: { role: 'assistant', content: 'stop' },
122+
usage: { prompt_tokens: 1, completion_tokens: 1, total_tokens: 2 },
123+
// biome-ignore lint/suspicious/noExplicitAny: mock
124+
} as any;
125+
}
126+
return {
127+
message: {
128+
role: 'assistant',
129+
tool_calls: [
130+
{
131+
id: 'c2',
132+
type: 'function',
133+
function: { name: 'run_command', arguments: '{"command":"rm -rf /"}' },
134+
},
135+
],
136+
},
137+
// biome-ignore lint/suspicious/noExplicitAny: mock
138+
} as any;
139+
});
140+
141+
// biome-ignore lint/suspicious/noExplicitAny: mock
142+
await executeLlmStep(step, context, executeStepFn as any);
143+
144+
expect(securityErrorMessage).toContain('Security Error');
145+
expect(executeStepFn).not.toHaveBeenCalled();
146+
});
147+
});

0 commit comments

Comments
 (0)