Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions command.md
Original file line number Diff line number Diff line change
Expand Up @@ -427,7 +427,7 @@ add a custom evaluator to the current project
agentcore add evaluator llm-as-a-judge [options]
```

add an LLM-as-a-Judge evaluator: another LLM prompted with instructions on how to score a session
add an LLM-as-a-Judge evaluator to the current project

**Options**

Expand All @@ -447,7 +447,7 @@ add an LLM-as-a-Judge evaluator: another LLM prompted with instructions on how t
agentcore add evaluator code-based [options]
```

add a code-based evaluator: scaffold a Python Lambda with custom evaluation logic, or reference an existing Lambda with --lambda-arn
add a code-based evaluator to the current project

**Options**

Expand Down
10 changes: 10 additions & 0 deletions src/components/Root.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -140,6 +140,8 @@ import { AddHarnessScreen } from "../handlers/project/add/harness/screen.tsx";
import { AddConfigBundleScreen } from "../handlers/project/add/config-bundle/screen.tsx";
import { AddPaymentManagerScreen } from "../handlers/project/add/payment-manager/screen.tsx";
import { AddPaymentConnectorScreen } from "../handlers/project/add/payment-connector/screen.tsx";
import { AddLlmAsAJudgeEvaluatorScreen } from "../handlers/project/add/evaluator/llm-as-a-judge/screen.tsx";
import { AddCodeBasedEvaluatorScreen } from "../handlers/project/add/evaluator/code-based/screen.tsx";
import { ProjectStatusScreen } from "../handlers/project/status/screen.tsx";
import { ProjectRemoveScreen } from "../handlers/project/remove/screen.tsx";
import { HelpScreen, RootScreen } from "../handlers/screen.tsx";
Expand Down Expand Up @@ -943,6 +945,14 @@ function RouteTable({ ctx, core }: ScreenProps) {
path="agentcore/add/payment-connector"
element={<AddPaymentConnectorScreen ctx={ctx} core={core} />}
/>
<Route
path="agentcore/add/evaluator/llm-as-a-judge"
element={<AddLlmAsAJudgeEvaluatorScreen ctx={ctx} core={core} />}
/>
<Route
path="agentcore/add/evaluator/code-based"
element={<AddCodeBasedEvaluatorScreen ctx={ctx} core={core} />}
/>
<Route path="agentcore/remove" element={<ProjectRemoveScreen ctx={ctx} core={core} />} />
<Route
path="agentcore/remove/:resourceType"
Expand Down
18 changes: 18 additions & 0 deletions src/components/RouterScreen.test.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -173,6 +173,24 @@ describe("menu rendering", () => {
});
});

describe("narrow terminals", () => {
test("a description too long for the row leaves every command name in the same column", async () => {
const r = renderScreen("/agentcore/add");
await waitForText(r.lastFrame, "❯ ");
await r.resize(60);

await waitForText(r.lastFrame, "── cli");
const lines = r.lastFrame()!.split("\n");
const nameColumns = lines
.map((line) => /^\s{1,3}(?:❯ )?\s*[a-z][a-z0-9-]*\s{2,}\S/.exec(line)?.[0])
.filter((row) => row !== undefined)
.map((row) => row.replace("❯", " ").search(/[a-z]/));
expect(nameColumns.length).toBeGreaterThan(1);
expect(new Set(nameColumns).size).toBe(1);
r.unmount();
});
});

describe("filtering", () => {
test("typing narrows the options to matches", async () => {
const r = renderScreen("/agentcore/harness");
Expand Down
33 changes: 18 additions & 15 deletions src/components/RouterScreen.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -314,21 +314,24 @@ function CommandMenuBody({
overflow="hidden"
flexShrink={0}
>
<Text color={theme.colors.focus}>
{isHighlighted ? `${glyphs.pointer} ` : " "}
</Text>
<Text
bold={isHighlighted}
color={
isHighlighted
? theme.colors.focus
: option.cliOnly
? theme.colors.muted
: theme.colors.text
}
>
{option.name.padEnd(nameWidth)}
</Text>
{/** A description too long for the row is cut and leaves the name in line. **/}
<Box flexShrink={0}>
<Text color={theme.colors.focus}>
{isHighlighted ? `${glyphs.pointer} ` : " "}
</Text>
<Text
bold={isHighlighted}
color={
isHighlighted
? theme.colors.focus
: option.cliOnly
? theme.colors.muted
: theme.colors.text
}
>
{option.name.padEnd(nameWidth)}
</Text>
</Box>
<Text color={theme.colors.muted}>{option.description}</Text>
</Box>
);
Expand Down
7 changes: 7 additions & 0 deletions src/components/wizard/fields.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -611,6 +611,13 @@ export function MultiChoiceField<T>({
);
}

export function promptPreview(prompt: string): string {

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

We should add back the length cap to this

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Fixed in 4656785. The first line is capped at 60 characters again, followed by the line count. Tests cover a long single line and a long first line.

const lines = prompt.split("\n");
const [first = ""] = lines;
const shown = first.length > 60 ? `${first.slice(0, 59)}…` : first;
return lines.length === 1 ? shown : `${shown} · ${lines.length} lines`;
}

export interface SummaryProps {
items: Record<string, string>;
}
Expand Down
1 change: 1 addition & 0 deletions src/components/wizard/index.ts
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,7 @@ export {
MultiTextField,
Summary,
firstIssue,
promptPreview,
type Choice,
type TextFieldProps,
type TextAreaFieldProps,
Expand Down
18 changes: 18 additions & 0 deletions src/components/wizard/wizard.test.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@ import {
ChoiceField,
MultiChoiceField,
MultiTextField,
promptPreview,
ResourceChoiceField,
RevealChoiceField,
Summary,
Expand Down Expand Up @@ -1152,3 +1153,20 @@ describe("Wizard authoring guards", () => {
expect((error as Error).message).toBe('duplicate <Step stepKey="name">');
});
});

describe("promptPreview", () => {
test.each([
["a one-line prompt", "You are a pirate.", "You are a pirate."],
["counts every line", "You are a pirate.\nAnswer in rhyme.", "You are a pirate. · 2 lines"],
["keeps a leading blank line", "\nYou are a pirate.", " · 2 lines"],
[
"counts a trailing newline left by enter",
"You are a pirate.\n",
"You are a pirate. · 2 lines",
],
["cuts a long first line short", `${"x".repeat(70)}\ny`, `${"x".repeat(59)}… · 2 lines`],
["cuts a long single line short", "x".repeat(70), `${"x".repeat(59)}…`],
])("%s", (_label, prompt, preview) => {
expect(promptPreview(prompt)).toBe(preview);
});
});
8 changes: 5 additions & 3 deletions src/core/project/templates/evaluator.ts
Original file line number Diff line number Diff line change
@@ -1,11 +1,13 @@
import { FsTreeNode } from "./fsTree";
import type { AssetSource } from "../source";
import type { Evaluator } from "../../../projectSchemas/evaluator";
import {
DEFAULT_CODE_BASED_TIMEOUT_SECONDS,
type Evaluator,
} from "../../../projectSchemas/evaluator";
import type { TemplateRenderer, TemplateResolver } from "./types";
import { toPythonPackageName } from "../fsUtils";
import type { ManagedEvaluatorScaffoldInput } from "../../../handlers/project/types";

const DEFAULT_TIMEOUT = 60;
const ASSET_DIR = "evaluators/python-lambda";

function buildManagedEvaluatorSpec(input: ManagedEvaluatorScaffoldInput): Evaluator {
Expand All @@ -18,7 +20,7 @@ function buildManagedEvaluatorSpec(input: ManagedEvaluatorScaffoldInput): Evalua
managed: {
codeLocation: `app/${input.name}`,
entrypoint: "lambda_function.handler",
timeoutSeconds: input.timeoutSeconds ?? DEFAULT_TIMEOUT,
timeoutSeconds: input.timeoutSeconds ?? DEFAULT_CODE_BASED_TIMEOUT_SECONDS,
additionalPolicies: ["execution-role-policy.json"],
},
},
Expand Down
1 change: 1 addition & 0 deletions src/handlers/project/add/add.screen.test.tsx
Original file line number Diff line number Diff line change
Expand Up @@ -32,6 +32,7 @@ const WITH_SCREENS = [
"config-bundle",
"payment-manager",
"payment-connector",
"evaluator",
];

describe("project add menu", () => {
Expand Down
Original file line number Diff line number Diff line change
@@ -0,0 +1,166 @@
import { afterEach, describe, expect, test } from "bun:test";
import { join } from "node:path";
import {
cleanupScreens,
flatFrame,
renderScreen,
waitForFlatText,
waitForText,
type RenderScreenResult,
} from "../../../../../testing";
import { createGatewayProjectTestHarness } from "../../gateway-test-support";

const { cleanup, inProject, projectSpec, run } =
createGatewayProjectTestHarness("add-code-based-wizard");

afterEach(cleanup);
afterEach(cleanupScreens);

const LAMBDA_ARN = "arn:aws:lambda:us-west-2:123456789012:function:refund-policy";

async function evaluatorOf(projectRoot: string, name: string) {
return ((await projectSpec(projectRoot)).evaluators ?? []).find(
(evaluator: { name: string }) => evaluator.name === name,
);
}

async function reachLambdaStep(screen: RenderScreenResult, name: string): Promise<void> {
await waitForText(screen.lastFrame, "what should this evaluator be called?");
await screen.write(name);
await screen.press("return");
await waitForText(screen.lastFrame, "what should it score?");
await screen.press("return");
await waitForText(screen.lastFrame, "which Lambda should score it?");
}

describe("project add evaluator code-based wizard", () => {
test("scaffolds the same managed evaluator as the flags", async () => {
const projectRoot = await inProject();
const screen = renderScreen("/agentcore/add/evaluator/code-based");
await reachLambdaStep(screen, "refund_policy");

expect(screen.lastFrame()).toContain("❯ ● scaffold a new Lambda");
expect(screen.lastFrame()).toContain("Python code in app/refund_policy");
expect(screen.lastFrame()).toContain("○ timeout");
await screen.press("return");

await waitForText(screen.lastFrame, "how long may it run?");
expect(screen.lastFrame()).toContain("60");
await screen.press("backspace");
await screen.press("backspace");
await screen.write("120");
await screen.press("return");

await waitForText(screen.lastFrame, "this evaluator will be added to agentcore.json");
const review = flatFrame(screen.lastFrame);
expect(review).toContain("evaluator refund_policy");
expect(review).toContain("level SESSION");
expect(review).toContain("lambda scaffolded in app/refund_policy");
expect(review).toContain("timeout 120 seconds");
await screen.press("return");

await waitForText(screen.lastFrame, "added evaluator 'refund_policy' to 'TestProject'");
await waitForFlatText(
screen.lastFrame,
"note: this evaluator returns Pass for every session until you implement app/refund_policy/lambda_function.py",
);
expect(
await Bun.file(join(projectRoot, "app", "refund_policy", "lambda_function.py")).exists(),
).toBe(true);

await run([
"add",
"evaluator",
"code-based",
"--name",
"flags",
"--level",
"SESSION",
"--timeout-seconds",
"120",
]);
const flags = await evaluatorOf(projectRoot, "flags");
expect(await evaluatorOf(projectRoot, "refund_policy")).toEqual({
...flags,
name: "refund_policy",
config: {
codeBased: {
managed: { ...flags.config.codeBased.managed, codeLocation: "app/refund_policy" },
},
},
});

await screen.press("return");
await waitForText(screen.lastFrame, "add a custom evaluator to the current project");
screen.unmount();
}, 20000);

test("an existing Lambda skips the timeout and writes an external config", async () => {
const projectRoot = await inProject();
const screen = renderScreen("/agentcore/add/evaluator/code-based");
await reachLambdaStep(screen, "refund_policy");

await screen.press("down");
await waitForText(screen.lastFrame, "❯ ● use an existing Lambda");
expect(screen.lastFrame()).not.toContain("timeout");
await screen.press("return");
await waitForText(screen.lastFrame, "Lambda ARN");
await screen.write(LAMBDA_ARN);
await screen.press("return");

await waitForText(screen.lastFrame, "this evaluator will be added to agentcore.json");
const review = flatFrame(screen.lastFrame);
expect(review.replace(/\s/g, "")).toContain(`lambda${LAMBDA_ARN}`);
expect(review).not.toContain("timeout");
await screen.press("return");

await waitForText(screen.lastFrame, "added evaluator 'refund_policy'");
expect(screen.lastFrame()).not.toContain("returns Pass");
expect((await evaluatorOf(projectRoot, "refund_policy")).config).toEqual({
codeBased: { external: { lambdaArn: LAMBDA_ARN } },
});
expect(await Bun.file(join(projectRoot, "app", "refund_policy")).exists()).toBe(false);
screen.unmount();
}, 20000);

test.each([
[
"a malformed Lambda ARN",
["down", "return"],
"refund-policy",
"Must be a valid Lambda function ARN",
],
["a timeout past 300 seconds", ["return"], "0", "Too big: expected number to be <=300"],
] as const)("%s keeps the wizard on its step", async (_label, moves, typed, message) => {
await inProject();
const screen = renderScreen("/agentcore/add/evaluator/code-based");
await reachLambdaStep(screen, "refund_policy");
for (const move of moves) await screen.press(move);
await screen.write(typed);

await screen.press("return");

await waitForText(screen.lastFrame, message);
expect(screen.lastFrame()).not.toContain("this evaluator will be added");
screen.unmount();
});

test("a rejected add reports itself and hands the form back", async () => {
const projectRoot = await inProject();
await run(["add", "evaluator", "code-based", "--name", "refund_policy", "--level", "SESSION"]);
const screen = renderScreen("/agentcore/add/evaluator/code-based");
await reachLambdaStep(screen, "refund_policy");
await screen.press("down");
await screen.press("return");
await screen.write(LAMBDA_ARN);
await screen.press("return");
await waitForText(screen.lastFrame, "this evaluator will be added to agentcore.json");
await screen.press("return");

await waitForFlatText(screen.lastFrame, "a evaluator with name 'refund_policy' already exists");
await screen.press("escape");
await waitForText(screen.lastFrame, "this evaluator will be added to agentcore.json");
expect((await projectSpec(projectRoot)).evaluators).toHaveLength(1);
screen.unmount();
}, 20000);
});
26 changes: 26 additions & 0 deletions src/handlers/project/add/evaluator/code-based/index.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -171,6 +171,32 @@ describe("project add evaluator code-based", () => {
await expectError(promise, requiredMessage ?? /./, InputValidationError);
});

test.each<[string, string[], string]>([
["a path in --name", ["--name", "../outside"], "Must begin with a letter"],
[
"an invalid --kms-key-arn",
["--name", "custom_eval", "--kms-key-arn", "not-a-key"],
"Must be a valid KMS key ARN",
],
])("rejects %s before scaffolding anything", async (_label, flags, message) => {
const { projectRoot, cleanup } = await initProject();
cleanups.push(cleanup);

await expectError(
run(["add", "evaluator", "code-based", "--level", "SESSION", ...flags]),
message,
InputValidationError,
);

for (const path of [
join(projectRoot, "..", "outside"),
join(projectRoot, "outside"),
join(projectRoot, "app", "custom_eval"),
]) {
expect(await Bun.file(join(path, "lambda_function.py")).exists()).toBe(false);
}
});

test.each([
["--metric", "deepeval.FaithfulnessMetric"],
["--model", "bedrock/anthropic.claude-3-5-sonnet-20240620-v1:0"],
Expand Down
Loading
Loading