feat(tui): enhance LLM interaction with max_tokens and temperature overrides; add tests for parallel delegation and workflow script parsing
This commit is contained in:
@@ -111,13 +111,16 @@ interface ScriptedStep {
|
||||
function makeScriptedProvider(script: ScriptedStep[]): {
|
||||
provider: ProviderService;
|
||||
seen: Array<ChatMessage[]>;
|
||||
chatCalls: Array<ChatMessage[]>;
|
||||
} {
|
||||
const seen: Array<ChatMessage[]> = [];
|
||||
const chatCalls: Array<ChatMessage[]> = [];
|
||||
let step = 0;
|
||||
const chatResponses: string[] = [];
|
||||
const provider: ProviderService = {
|
||||
async chat(_messages, _tools, _maxTokens, _temperature) {
|
||||
async chat(messages, _tools, _maxTokens, _temperature) {
|
||||
// Used by compaction and review. For review tests we script below.
|
||||
chatCalls.push([...messages]);
|
||||
const text = chatResponses.shift() ?? "";
|
||||
return { message: { role: "assistant", content: text || null }, usage: [10, 5] };
|
||||
},
|
||||
@@ -146,7 +149,7 @@ function makeScriptedProvider(script: ScriptedStep[]): {
|
||||
(provider as unknown as { setChatReply: (s: string) => void }).setChatReply = (text: string) => {
|
||||
chatResponses.push(text);
|
||||
};
|
||||
return { provider, seen };
|
||||
return { provider, seen, chatCalls };
|
||||
}
|
||||
|
||||
/** A ToolExecutor that echoes fixed outputs per tool. */
|
||||
@@ -180,33 +183,13 @@ function buildParams(userText: string): {
|
||||
return { params, events };
|
||||
}
|
||||
|
||||
/** Extract the `[Flow]` steering system message from a captured batch. */
|
||||
function flowDirective(batch: ChatMessage[]): string {
|
||||
const m = batch.find((x) => (x.content ?? "").includes("[Flow]"));
|
||||
return m?.content ?? "";
|
||||
}
|
||||
|
||||
describe("runTurn complexity steering", () => {
|
||||
it("injects a complex-request directive for a complex prompt", async () => {
|
||||
const { provider, seen } = makeScriptedProvider([
|
||||
{ content: "final answer" },
|
||||
]);
|
||||
describe("runTurn", () => {
|
||||
it("runs a simple one-shot prompt to completion", async () => {
|
||||
const { provider } = makeScriptedProvider([{ content: "here is the answer" }]);
|
||||
const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({}), [] as never);
|
||||
const { params } = buildParams("Please refactor the architecture across multiple files");
|
||||
const { params } = buildParams("hello");
|
||||
await svc.runTurn(params);
|
||||
// The complexity directive system message must be present before the LLM.
|
||||
const flow = flowDirective(seen[0] ?? []);
|
||||
expect(flow).toContain("flagged as complex");
|
||||
});
|
||||
|
||||
it("injects a keep-it-simple directive instead", async () => {
|
||||
const { provider, seen } = makeScriptedProvider([{ content: "hi" }]);
|
||||
const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({}), [] as never);
|
||||
const { params } = buildParams("what is 2+2");
|
||||
await svc.runTurn(params);
|
||||
const flow = flowDirective(seen[0] ?? []);
|
||||
expect(flow).toContain("looks simple");
|
||||
expect(flow).not.toContain("complex");
|
||||
expect(params.in_flight.value).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
@@ -266,6 +249,25 @@ describe("runTurn self-review", () => {
|
||||
// review_usage emitted.
|
||||
expect(events.some((e) => e.kind === "review_usage")).toBe(true);
|
||||
});
|
||||
|
||||
it("passes the work-in-progress to the reviewer (not an empty slate)", async () => {
|
||||
const { provider, chatCalls } = makeScriptedProvider([
|
||||
{ tools: [{ name: "edit" }] },
|
||||
{ content: "fixed" },
|
||||
]);
|
||||
(provider as unknown as { setChatReply(s: string): void }).setChatReply(
|
||||
"- [PRIORITY: medium] add a null check",
|
||||
);
|
||||
const svc = new AgentTurnServiceImpl(provider as never, fixedExecutor({ edit: "changed" }), [] as never);
|
||||
const { params } = buildParams("harden the parser");
|
||||
await svc.runTurn(params);
|
||||
// The REVIEWER chat call carries the digest in its user message.
|
||||
const reviewerCall = chatCalls.find((c) =>
|
||||
c.some((m) => (m.content ?? "").includes("Review the work-in-progress below")),
|
||||
);
|
||||
expect(reviewerCall).toBeTruthy();
|
||||
expect(reviewerCall!.some((m) => (m.content ?? "").includes("harden the parser"))).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
/* ── Compaction: efficient + accurate ─────────────────────────────── */
|
||||
|
||||
@@ -23,7 +23,6 @@ import {
|
||||
errorRecoveryNote,
|
||||
reviewerPrompt,
|
||||
} from "@zesdex/agent";
|
||||
import { isComplexRequest } from "@zesdex/workflow";
|
||||
import type { ProviderService } from "./ports.ts";
|
||||
import type { ToolExecutor } from "./index.ts";
|
||||
|
||||
@@ -59,6 +58,9 @@ const MAX_NO_PROGRESS_STREAK = 4;
|
||||
/** Self-review is bounded to this many passes per turn. */
|
||||
const MAX_REVIEW_PASSES = 1;
|
||||
|
||||
/** Per-tool-call wall-clock budget before the tool is considered hung. */
|
||||
const TOOL_TIMEOUT_MS = 90_000;
|
||||
|
||||
/** Whether the output string denotes a tool error. */
|
||||
function isErrorOutput(output: string): boolean {
|
||||
return output.startsWith("Error:");
|
||||
@@ -206,16 +208,30 @@ async function executeToolCall(
|
||||
executor: ToolExecutor,
|
||||
sink: TurnEventSink,
|
||||
tc: ToolCall,
|
||||
signal?: AbortSignal,
|
||||
): Promise<string> {
|
||||
const name = tc.function.name;
|
||||
const args = sanitizeToolArguments(tc.function.arguments);
|
||||
|
||||
let output: string;
|
||||
try {
|
||||
output = await executor.execute(name, args as JsonValue);
|
||||
} catch (e) {
|
||||
output = `Error: ${(e as Error).message}`;
|
||||
}
|
||||
const run = executor.execute(name, args as JsonValue).catch((e) => `Error: ${(e as Error).message}`);
|
||||
|
||||
// Correct async cancellation: enforce a per-tool time budget and stop
|
||||
// immediately if the turn is aborted while the tool is still running.
|
||||
let timer: ReturnType<typeof setTimeout> | undefined;
|
||||
const output = await Promise.race([
|
||||
run,
|
||||
new Promise<string>((resolve) => {
|
||||
const onAbort = () => resolve("Error: Turn aborted by user");
|
||||
timer = setTimeout(() => resolve(`Error: Tool timed out after ${TOOL_TIMEOUT_MS / 1000}s`), TOOL_TIMEOUT_MS);
|
||||
signal?.addEventListener("abort", onAbort, { once: true });
|
||||
// Release the abort listener + timer once the tool settles either way.
|
||||
void run.finally(() => {
|
||||
clearTimeout(timer);
|
||||
signal?.removeEventListener("abort", onAbort);
|
||||
});
|
||||
}),
|
||||
]);
|
||||
if (timer) clearTimeout(timer);
|
||||
|
||||
const isError = isToolFailure(name, output);
|
||||
const truncated = truncateToolOutput(output);
|
||||
@@ -237,6 +253,7 @@ async function executeToolCallsInParallel(
|
||||
executor: ToolExecutor,
|
||||
sink: TurnEventSink,
|
||||
toolCalls: ToolCall[],
|
||||
signal?: AbortSignal,
|
||||
): Promise<string[]> {
|
||||
// Simple bounded concurrency preserving input order.
|
||||
const results: string[] = new Array(toolCalls.length);
|
||||
@@ -246,7 +263,7 @@ async function executeToolCallsInParallel(
|
||||
while (true) {
|
||||
const idx = next++;
|
||||
if (idx >= toolCalls.length) return;
|
||||
results[idx] = await executeToolCall(executor, sink, toolCalls[idx]!);
|
||||
results[idx] = await executeToolCall(executor, sink, toolCalls[idx]!, signal);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -445,23 +462,6 @@ export class AgentTurnServiceImpl {
|
||||
// Estimate request complexity from the last user message.
|
||||
const last = params.messages[params.messages.length - 1];
|
||||
const requestLen = last?.content?.length ?? 0;
|
||||
const userText = last?.content ?? "";
|
||||
|
||||
// Complexity steering: a one-shot directive so the model picks the right
|
||||
// depth instead of relying on prose memory. Pure and cheap.
|
||||
const complex = isComplexRequest(userText);
|
||||
this.push(sink, {
|
||||
kind: "system_note",
|
||||
systemKind: "info",
|
||||
message: complex ? "Complex request detected — plan before executing." : "Simple request — keep tool use minimal.",
|
||||
});
|
||||
params.messages.push(
|
||||
systemMessage(
|
||||
complex
|
||||
? "[Flow] This request is flagged as complex. Enter a short plan with `plan_enter` and track steps with `todowrite` before starting edits."
|
||||
: "[Flow] This request looks simple. If you can answer directly without tools, do so — do not spawn agents or workflows for it.",
|
||||
),
|
||||
);
|
||||
|
||||
const errors = new ErrorTracker();
|
||||
let sawToolCalls = false;
|
||||
@@ -507,9 +507,10 @@ export class AgentTurnServiceImpl {
|
||||
// Auto-compact oversized history before the LLM call.
|
||||
await this.autoCompactIfNeeded(params.messages);
|
||||
|
||||
// Adaptive generation parameters.
|
||||
const maxTokens = adaptiveMaxTokens(requestLen);
|
||||
const temperature = sawToolCalls ? 0.2 : 0.7;
|
||||
// Generation parameters: a config override wins, else fall back to the
|
||||
// adaptive defaults (lower temperature once the model is doing tool work).
|
||||
const maxTokens = params.max_tokens ?? adaptiveMaxTokens(requestLen);
|
||||
const temperature = params.temperature ?? (sawToolCalls ? 0.2 : 0.7);
|
||||
|
||||
this.push(sink, { kind: "stream_start" });
|
||||
|
||||
@@ -547,10 +548,10 @@ export class AgentTurnServiceImpl {
|
||||
const parallel = toolCalls.length > 1 && toolCalls.every(isParallelSafe);
|
||||
|
||||
const outputs = parallel
|
||||
? await executeToolCallsInParallel(this.toolExecutor, sink, toolCalls)
|
||||
? await executeToolCallsInParallel(this.toolExecutor, sink, toolCalls, abort.signal)
|
||||
: await (async () => {
|
||||
const seq: string[] = [];
|
||||
for (const tc of toolCalls) seq.push(await executeToolCall(this.toolExecutor, sink, tc));
|
||||
for (const tc of toolCalls) seq.push(await executeToolCall(this.toolExecutor, sink, tc, abort.signal));
|
||||
return seq;
|
||||
})();
|
||||
|
||||
@@ -580,11 +581,12 @@ export class AgentTurnServiceImpl {
|
||||
|
||||
if (mutated) {
|
||||
pendingVerify = true;
|
||||
verifyPrompted = false; // allow a fresh nudge after the next edit batch
|
||||
// Do NOT re-arm verifyPrompted: the verify instruction is already in
|
||||
// context, so re-injecting it on every edit batch only bloats tokens.
|
||||
}
|
||||
if (verified) {
|
||||
pendingVerify = false; // a successful bash run satisfies the verify nudge
|
||||
verifyPrompted = false;
|
||||
verifyPrompted = false; // a later edit batch may nudge again
|
||||
}
|
||||
|
||||
// Bounded self-review pass after file mutations (at most one per turn).
|
||||
@@ -608,9 +610,16 @@ export class AgentTurnServiceImpl {
|
||||
sink: TurnEventSink,
|
||||
): Promise<void> {
|
||||
if (abort.signal.aborted) return;
|
||||
// Reduce the work-in-progress to a focused digest so the reviewer has real
|
||||
// context to critique (not an empty slate that invites hallucinated issues).
|
||||
const digest = reduceMessagesToDigest(messages.slice(-80));
|
||||
if (digest.trim() === "") return;
|
||||
try {
|
||||
const result = await this.provider.chat(
|
||||
[systemMessage(reviewerPrompt())],
|
||||
[
|
||||
systemMessage(reviewerPrompt()),
|
||||
userMessage(`Review the work-in-progress below and list actionable issues:\n\n---\n${digest}`),
|
||||
],
|
||||
undefined,
|
||||
1024,
|
||||
0.3,
|
||||
|
||||
@@ -62,7 +62,6 @@ export function agentStatusDisplay(s: AgentStatus, error?: string): string {
|
||||
|
||||
/** Events emitted onto the turn-event queue while an agent turn runs. */
|
||||
export type TurnEvent =
|
||||
| { kind: "assistant_message"; message: ChatMessage }
|
||||
| { kind: "tool_result"; tool_call_id: string; tool_name: string; output: string; is_error: boolean; path: string | null }
|
||||
| { kind: "system_note"; systemKind: string; message: string }
|
||||
| { kind: "stream_start" }
|
||||
@@ -183,6 +182,10 @@ export interface AgentTurnParams {
|
||||
api_key: string;
|
||||
model: string;
|
||||
api_base?: string;
|
||||
/** Optional override for the LLM max_tokens ceiling (defaults to adaptive). */
|
||||
max_tokens?: number;
|
||||
/** Optional override for the LLM temperature (defaults to 0.7 / 0.2 tooling). */
|
||||
temperature?: number;
|
||||
}
|
||||
|
||||
/** Minimal event-sink abstraction (backs the Rust `Arc<Mutex<VecDeque>>`). */
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
/**
|
||||
* Tests for the LLM-driven parallel-delegation splitter. The decomposition is
|
||||
* decided by the AI (no hardcoded keyword guessing); these tests lock in the
|
||||
* tolerant JSON parsing and the access-tier normalization.
|
||||
*/
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import { parseDirectivesJson } from "./parallel_delegate.ts";
|
||||
|
||||
describe("parseDirectivesJson", () => {
|
||||
test("parses a clean JSON array", () => {
|
||||
const out = parseDirectivesJson(
|
||||
`[{"directive":"Scan the config code","access":"read"},{"directive":"Implement the change","access":"full"}]`,
|
||||
);
|
||||
expect(out.length).toBe(2);
|
||||
expect(out[0]).toEqual({ directive: "Scan the config code", access: "read" });
|
||||
});
|
||||
|
||||
test("extracts the array from markdown fences with a preamble", () => {
|
||||
const out = parseDirectivesJson(
|
||||
'Here are the subtasks:\n```json\n[{"directive":"A","access":"read"},{"directive":"B","access":"full"}]\n```\nDone.',
|
||||
);
|
||||
expect(out.map((d) => d.directive)).toEqual(["A", "B"]);
|
||||
});
|
||||
|
||||
test("normalises unknown access tiers to write", () => {
|
||||
const out = parseDirectivesJson(`[{"directive":"do it","access":"admin"}]`);
|
||||
expect(out[0]!.access).toBe("write");
|
||||
});
|
||||
|
||||
test("drops empty directives and malformed input", () => {
|
||||
expect(parseDirectivesJson(`[{"directive":"","access":"write"}]`)).toEqual([]);
|
||||
expect(parseDirectivesJson("not json at all")).toEqual([]);
|
||||
expect(parseDirectivesJson("")).toEqual([]);
|
||||
});
|
||||
});
|
||||
@@ -50,7 +50,8 @@ export class ParallelDelegate implements Tool {
|
||||
}))
|
||||
.filter((d) => d.directive !== "");
|
||||
} else {
|
||||
directives = fallbackSplit(task, maxParallel);
|
||||
// Let the LLM decide the decomposition instead of guessing keywords.
|
||||
directives = await aiDecomposeTask(task, maxParallel);
|
||||
}
|
||||
|
||||
if (directives.length === 0) {
|
||||
@@ -73,22 +74,67 @@ export class ParallelDelegate implements Tool {
|
||||
}
|
||||
}
|
||||
|
||||
/** Fallback splitting when LLM is unavailable. */
|
||||
function fallbackSplit(task: string, maxParallel: number): Array<{ directive: string; access: string }> {
|
||||
const directives: Array<{ directive: string; access: string }> = [];
|
||||
if (task.includes("backend") || task.includes("api") || task.includes("server")) {
|
||||
directives.push({ directive: `Implement the backend/API components for: ${task}`, access: "write" });
|
||||
/**
|
||||
* Have the LLM decompose a task into focused, non-overlapping parallel
|
||||
* sub-directives. This is the AI deciding how to split the work — no keyword
|
||||
* guessing. On LLM failure (no provider / error), degrades to a single honest
|
||||
* full-task directive rather than fabricated "part N" stubs.
|
||||
*/
|
||||
async function aiDecomposeTask(
|
||||
task: string,
|
||||
maxParallel: number,
|
||||
): Promise<Array<{ directive: string; access: string }>> {
|
||||
try {
|
||||
const { resolveConfig } = await import("../../../subagent/infrastructure/config_resolver.ts");
|
||||
const { buildProviderService } = await import("../../../subagent/infrastructure/http_provider.ts");
|
||||
const { baseUrl, apiKey, model } = await resolveConfig();
|
||||
const svc = buildProviderService(baseUrl, apiKey, model);
|
||||
|
||||
const systemPrompt =
|
||||
"You decompose a large task into independent, non-overlapping subtasks that can be worked on " +
|
||||
"in parallel. Return a JSON array of up to the requested count of objects, each with a " +
|
||||
"'directive' (a concrete, self-contained action) and an 'access' tier of exactly one of " +
|
||||
"read | write | full. read = inspect/search only; write = edit files; full = edit + run shell. " +
|
||||
"Prefer fewer, genuinely parallel directives over many that touch the same files. " +
|
||||
"Return ONLY the JSON array, no prose.";
|
||||
|
||||
const { message } = await svc.chat(
|
||||
[
|
||||
{ role: "system", content: systemPrompt },
|
||||
{ role: "user", content: `Task: ${task}\nMax subtasks: ${maxParallel}\n` },
|
||||
],
|
||||
undefined,
|
||||
1024,
|
||||
0.3,
|
||||
);
|
||||
|
||||
const parsed = parseDirectivesJson(message.content ?? "");
|
||||
if (parsed.length > 0) return parsed.slice(0, maxParallel);
|
||||
} catch {
|
||||
/* fall through to the single-directive fallback */
|
||||
}
|
||||
if (task.includes("frontend") || task.includes("ui") || task.includes("client")) {
|
||||
directives.push({ directive: `Implement the frontend/UI components for: ${task}`, access: "write" });
|
||||
// Honest fallback: run the whole task as one agent rather than guessing a split.
|
||||
return [{ directive: task, access: "full" }];
|
||||
}
|
||||
if (task.includes("test") || task.includes("unit")) {
|
||||
directives.push({ directive: `Write unit tests for: ${task}`, access: "read" });
|
||||
}
|
||||
if (directives.length === 0) {
|
||||
for (let i = 0; i < maxParallel; i++) {
|
||||
directives.push({ directive: `Part ${i + 1} of parallel task: ${task}`, access: "write" });
|
||||
|
||||
/** Parse an LLM JSON array of directives, tolerant of stray prose/markdown fences. */
|
||||
export function parseDirectivesJson(raw: string): Array<{ directive: string; access: string }> {
|
||||
const fenced = raw.match(/```(?:json)?\s*([\s\S]*?)```/i);
|
||||
const body = fenced ? fenced[1]! : raw;
|
||||
const start = body.indexOf("[");
|
||||
const end = body.lastIndexOf("]");
|
||||
if (start < 0 || end <= start) return [];
|
||||
try {
|
||||
const arr = JSON.parse(body.slice(start, end + 1)) as unknown;
|
||||
if (!Array.isArray(arr)) return [];
|
||||
return arr
|
||||
.filter((d): d is Record<string, unknown> => typeof d === "object" && d !== null)
|
||||
.map((d) => ({
|
||||
directive: typeof d.directive === "string" ? d.directive.trim() : "",
|
||||
access: ["read", "write", "full"].includes(String(d.access)) ? String(d.access) : "write",
|
||||
}))
|
||||
.filter((d) => d.directive !== "");
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
}
|
||||
return directives.slice(0, maxParallel);
|
||||
}
|
||||
|
||||
@@ -37,12 +37,12 @@ describe("allTools registry", () => {
|
||||
|
||||
describe("read-only tools are parallel-safe", () => {
|
||||
test("parallel-safe list", () => {
|
||||
for (const name of ["read", "grep", "glob", "semantic_search", "list_symbols", "web_search", "recall", "dir_list", "pong", "seq_think", "dir_cache_update"]) {
|
||||
for (const name of ["read", "grep", "glob", "semantic_search", "list_symbols", "web_search", "recall", "dir_list", "pong", "seq_think"]) {
|
||||
expect(toolIsParallelSafe(name)).toBe(true);
|
||||
}
|
||||
});
|
||||
test("mutating tools are not parallel-safe", () => {
|
||||
for (const name of ["edit", "write", "delete", "bash", "git_operator", "git_worktree", "remember", "forget", "todowrite", "plan_enter", "spawn_agents"]) {
|
||||
test("mutating tools (incl. dir_cache_update, which writes the cache) are not parallel-safe", () => {
|
||||
for (const name of ["edit", "write", "delete", "bash", "git_operator", "git_worktree", "remember", "forget", "todowrite", "plan_enter", "spawn_agents", "dir_cache_update"]) {
|
||||
expect(toolIsParallelSafe(name)).toBe(false);
|
||||
}
|
||||
});
|
||||
|
||||
@@ -88,7 +88,6 @@ export function toolIsParallelSafe(name: string): boolean {
|
||||
"dir_list",
|
||||
"pong",
|
||||
"seq_think",
|
||||
"dir_cache_update",
|
||||
].includes(name);
|
||||
}
|
||||
|
||||
|
||||
@@ -1,22 +0,0 @@
|
||||
/**
|
||||
* Complexity heuristics — determine whether a request is complex enough to
|
||||
* warrant hive-mind orchestration.
|
||||
* Mirrors `apps/infrastructure/src/workflow/hive_mind/complexity.rs`.
|
||||
*/
|
||||
|
||||
/** Heuristics to determine if a request is complex enough for hive-mind. */
|
||||
export function isComplexRequest(task: string): boolean {
|
||||
const complexityIndicators = [
|
||||
"refactor",
|
||||
"redesign",
|
||||
"multiple files",
|
||||
"architecture",
|
||||
"migration",
|
||||
"comprehensive",
|
||||
"end-to-end",
|
||||
"full-stack",
|
||||
];
|
||||
|
||||
const taskLower = task.toLowerCase();
|
||||
return complexityIndicators.some((indicator) => taskLower.includes(indicator));
|
||||
}
|
||||
@@ -7,5 +7,4 @@ export { executeWorkflowScript, executePrimitive, formatWorkflowResult, MAX_NODE
|
||||
export { executeWorkflow } from "./orchestrate.ts";
|
||||
export { executeCycle } from "./cycle.ts";
|
||||
export { synthesizeConsensus } from "./synthesis.ts";
|
||||
export { isComplexRequest } from "./complexity.ts";
|
||||
export { writeHiveMindConvergence } from "./docs.ts";
|
||||
|
||||
@@ -0,0 +1,61 @@
|
||||
/**
|
||||
* Tests for the workflow YAML-subset parser, including the folded multi-line
|
||||
* directive support.
|
||||
*/
|
||||
import { describe, expect, test } from "bun:test";
|
||||
import { parseWorkflowScript } from "./script.ts";
|
||||
|
||||
describe("parseWorkflowScript", () => {
|
||||
test("parses a flat single-line workflow", () => {
|
||||
const script = parseWorkflowScript(`
|
||||
name: my-workflow
|
||||
phases:
|
||||
- name: research
|
||||
directive: "Explore the codebase for the auth module."
|
||||
- name: implement
|
||||
directive: "Implement the change."
|
||||
`);
|
||||
expect(script.name).toBe("my-workflow");
|
||||
expect(script.phases.length).toBe(2);
|
||||
expect(script.phases[0]!.name).toBe("research");
|
||||
expect(script.phases[0]!.directive).toBe("Explore the codebase for the auth module.");
|
||||
});
|
||||
|
||||
test("folds an indented multi-line directive into one string", () => {
|
||||
const script = parseWorkflowScript(`
|
||||
name: audit
|
||||
phases:
|
||||
- name: analyze
|
||||
directive: |
|
||||
Look at two things:
|
||||
first, how auth is wired;
|
||||
second, where secrets are stored.
|
||||
`);
|
||||
expect(script.phases.length).toBe(1);
|
||||
const d = script.phases[0]!.directive;
|
||||
expect(d).toContain("Look at two things:");
|
||||
expect(d).toContain("second, where secrets are stored.");
|
||||
// Line breaks are preserved as folded content, not flattened.
|
||||
expect(d).toContain("\n");
|
||||
});
|
||||
|
||||
test("an indented continuation does not swallow the next phase", () => {
|
||||
const script = parseWorkflowScript(`
|
||||
phases:
|
||||
- name: a
|
||||
directive: first
|
||||
and more
|
||||
- name: b
|
||||
directive: second
|
||||
`);
|
||||
expect(script.phases.length).toBe(2);
|
||||
expect(script.phases[0]!.directive).toContain("and more");
|
||||
expect(script.phases[1]!.name).toBe("b");
|
||||
expect(script.phases[1]!.directive).toBe("second");
|
||||
});
|
||||
|
||||
test("empty or unparseable yaml yields no phases", () => {
|
||||
expect(parseWorkflowScript("").phases.length).toBe(0);
|
||||
expect(parseWorkflowScript("name: x\n").phases.length).toBe(0);
|
||||
});
|
||||
});
|
||||
@@ -31,21 +31,62 @@ export function parseWorkflowScript(yaml: string): WorkflowScript {
|
||||
|
||||
let inPhases = false;
|
||||
let currentPhase: Partial<WorkflowPhase> | null = null;
|
||||
/** Set when `directive:` was opened, to fold more-indented lines into it. */
|
||||
let directiveIndent: number | null = null;
|
||||
|
||||
const flushPhase = (): void => {
|
||||
if (currentPhase && (currentPhase.name || currentPhase.directive)) {
|
||||
phases.push({
|
||||
name: currentPhase.name ?? "phase",
|
||||
directive: currentPhase.directive ?? "",
|
||||
});
|
||||
}
|
||||
currentPhase = null;
|
||||
directiveIndent = null;
|
||||
};
|
||||
|
||||
const indentOf = (s: string): number => {
|
||||
const m = s.match(/^(\s*)/);
|
||||
return m ? m[1]!.length : 0;
|
||||
};
|
||||
|
||||
/** Strip one surrounding pair of single/double quotes, YAML-scalar style. */
|
||||
const unquote = (v: string): string => {
|
||||
const t = v.trim();
|
||||
if (t.length >= 2 && ((t.startsWith('"') && t.endsWith('"')) || (t.startsWith("'") && t.endsWith("'")))) {
|
||||
return t.slice(1, -1);
|
||||
}
|
||||
return t;
|
||||
};
|
||||
|
||||
for (const raw of lines) {
|
||||
const trimmed = raw.trim();
|
||||
if (trimmed === "" || trimmed.startsWith("#")) continue;
|
||||
|
||||
// Fold continuation lines into the open directive (indented block text).
|
||||
// A continuation is a non-blank line indented deeper than the directive
|
||||
// key, that is not a new list item and not a `key:` assignment.
|
||||
if (currentPhase && directiveIndent !== null) {
|
||||
const indent = indentOf(raw);
|
||||
const isListItem = trimmed.startsWith("- ");
|
||||
const isKv = /^[\w-]+:\s/.test(trimmed);
|
||||
if (indent > directiveIndent && !isListItem && !isKv) {
|
||||
currentPhase.directive = `${currentPhase.directive ?? ""}\n${trimmed}`;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
// Top-level key: value
|
||||
const topMatch = trimmed.match(/^(\w+):\s*(.*)$/);
|
||||
if (topMatch && !raw.startsWith(" ") && !raw.startsWith("-")) {
|
||||
const key = topMatch[1]!;
|
||||
const val = topMatch[2]!.trim();
|
||||
const val = unquote(topMatch[2]!);
|
||||
if (key === "name" && val !== "") {
|
||||
name = val;
|
||||
} else if (key === "phases") {
|
||||
inPhases = true;
|
||||
}
|
||||
directiveIndent = null;
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -54,18 +95,17 @@ export function parseWorkflowScript(yaml: string): WorkflowScript {
|
||||
// List item: - name: foo or - directive: bar
|
||||
const listMatch = trimmed.match(/^-\s+(\w+):\s*(.*)$/);
|
||||
if (listMatch) {
|
||||
// If there was a previous phase, push it
|
||||
if (currentPhase && (currentPhase.name || currentPhase.directive)) {
|
||||
phases.push({
|
||||
name: currentPhase.name ?? "phase",
|
||||
directive: currentPhase.directive ?? "",
|
||||
});
|
||||
}
|
||||
flushPhase();
|
||||
const key = listMatch[1]!;
|
||||
const val = listMatch[2]!.trim();
|
||||
const val = unquote(listMatch[2]!);
|
||||
currentPhase = { name: undefined, directive: undefined };
|
||||
if (key === "name") currentPhase.name = val;
|
||||
if (key === "directive") currentPhase.directive = val;
|
||||
if (key === "directive") {
|
||||
currentPhase.directive = val;
|
||||
directiveIndent = indentOf(raw);
|
||||
} else {
|
||||
directiveIndent = null;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
@@ -74,20 +114,19 @@ export function parseWorkflowScript(yaml: string): WorkflowScript {
|
||||
const kvMatch = trimmed.match(/^(\w+):\s*(.*)$/);
|
||||
if (kvMatch) {
|
||||
const key = kvMatch[1]!;
|
||||
const val = kvMatch[2]!.trim();
|
||||
const val = unquote(kvMatch[2]!);
|
||||
if (key === "name") currentPhase.name = val;
|
||||
if (key === "directive") currentPhase.directive = val;
|
||||
if (key === "directive") {
|
||||
currentPhase.directive = val;
|
||||
directiveIndent = indentOf(raw);
|
||||
} else {
|
||||
directiveIndent = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Push the last phase
|
||||
if (currentPhase && (currentPhase.name || currentPhase.directive)) {
|
||||
phases.push({
|
||||
name: currentPhase.name ?? "phase",
|
||||
directive: currentPhase.directive ?? "",
|
||||
});
|
||||
}
|
||||
flushPhase();
|
||||
|
||||
return { name, phases };
|
||||
}
|
||||
|
||||
@@ -42,6 +42,10 @@ export interface WiredRuntime {
|
||||
apiBase?: string;
|
||||
/** Fetch the provider's available model ids (via GET /models). */
|
||||
listModels?: () => Promise<string[]>;
|
||||
/** Optional LLM max_tokens override from settings. */
|
||||
maxTokens?: number;
|
||||
/** Optional LLM temperature override from settings. */
|
||||
temperature?: number;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -86,6 +90,8 @@ export async function runSingleProcess(): Promise<WiredRuntime> {
|
||||
model,
|
||||
apiBase: baseUrl,
|
||||
listModels: () => llmClient.listModels(),
|
||||
maxTokens: settings.max_tokens,
|
||||
temperature: settings.temperature,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -121,6 +127,8 @@ export function buildTurnParams(
|
||||
api_key: runtime.apiKey,
|
||||
model: runtime.model,
|
||||
api_base: runtime.apiBase,
|
||||
max_tokens: runtime.maxTokens,
|
||||
temperature: runtime.temperature,
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -54,6 +54,8 @@ function makeTurnRunner(runtime: WiredRuntime, state: AppStateRest): TurnRunner
|
||||
api_key: runtime.apiKey,
|
||||
model: runtime.model,
|
||||
api_base: runtime.apiBase,
|
||||
max_tokens: runtime.maxTokens,
|
||||
temperature: runtime.temperature,
|
||||
});
|
||||
};
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user