187 lines
7.7 KiB
TypeScript
187 lines
7.7 KiB
TypeScript
import { describe, expect, test } from "bun:test";
|
|
import { parseConfig, SAFE_DEFAULT_TOOLS } from "./auth";
|
|
import { installCommunityApp, runReadOnlyCommand, sanitizeLogOutput } from "./helpers";
|
|
import { getToolRisk, toolByName } from "./tools";
|
|
import { ToolCallGate, withHardTimeout } from "./concurrency";
|
|
|
|
describe("secure tool configuration", () => {
|
|
test("new and incomplete configs use the read-only baseline", () => {
|
|
const cfg = parseConfig("MUA_API_KEY=test\n");
|
|
expect(cfg.allToolsEnabled).toBe(false);
|
|
expect(cfg.enabledTools).toEqual(SAFE_DEFAULT_TOOLS);
|
|
expect(cfg.enabledTools).not.toContain("unraid_system_shell");
|
|
});
|
|
|
|
test("none means no tools instead of all tools", () => {
|
|
const cfg = parseConfig("MUA_API_KEY=test\nMUA_ENABLED_TOOLS=none\n");
|
|
expect(cfg.allToolsEnabled).toBe(false);
|
|
expect(cfg.enabledTools).toEqual([]);
|
|
});
|
|
|
|
test("legacy all remains backwards compatible and explicit", () => {
|
|
const cfg = parseConfig("MUA_API_KEY=test\nMUA_ENABLED_TOOLS=all\n");
|
|
expect(cfg.allToolsEnabled).toBe(true);
|
|
expect(cfg.enabledTools).toEqual([]);
|
|
});
|
|
|
|
test("toolbox integration is opt-in and parsed independently", () => {
|
|
const off = parseConfig("MUA_API_KEY=test\nMUA_ENABLED_TOOLS=none\n");
|
|
const on = parseConfig(
|
|
"MUA_API_KEY=test\nMUA_ENABLED_TOOLS=none\nMUA_TOOLBOX_ENABLED=true\n",
|
|
);
|
|
expect(off.toolboxEnabled).toBe(false);
|
|
expect(on.toolboxEnabled).toBe(true);
|
|
expect(on.enabledTools).toEqual([]);
|
|
});
|
|
|
|
test("an enabled legacy update also exposes the safer batch update", () => {
|
|
const cfg = parseConfig(
|
|
"MUA_API_KEY=test\nMUA_ENABLED_TOOLS=unraid_docker_update\n",
|
|
);
|
|
expect(cfg.enabledTools).toContain("unraid_docker_update");
|
|
expect(cfg.enabledTools).toContain("unraid_docker_update_verified_batch");
|
|
});
|
|
|
|
test("an enabled read-only shell also exposes bounded file inventory", () => {
|
|
const cfg = parseConfig(
|
|
"MUA_API_KEY=test\nMUA_ENABLED_TOOLS=unraid_system_shell_readonly\n",
|
|
);
|
|
expect(cfg.enabledTools).toContain("unraid_files_inventory");
|
|
expect(cfg.enabledTools).not.toContain("unraid_system_shell");
|
|
});
|
|
|
|
test("an enabled critical shell also exposes bounded asynchronous jobs", () => {
|
|
const cfg = parseConfig(
|
|
"MUA_API_KEY=test\nMUA_ENABLED_TOOLS=unraid_system_shell\n",
|
|
);
|
|
expect(cfg.enabledTools).toContain("unraid_system_job_start");
|
|
expect(cfg.enabledTools).toContain("unraid_system_job_status");
|
|
expect(cfg.enabledTools).toContain("unraid_system_job_cleanup");
|
|
});
|
|
});
|
|
|
|
describe("secret handling", () => {
|
|
test("redacts common key-value and JSON secrets", () => {
|
|
const output = sanitizeLogOutput(
|
|
'API_KEY=very-secret password:also-secret {"token":"third-secret"}',
|
|
);
|
|
expect(output).not.toContain("very-secret");
|
|
expect(output).not.toContain("also-secret");
|
|
expect(output).not.toContain("third-secret");
|
|
expect(output).toContain("[REDACTED]");
|
|
});
|
|
});
|
|
|
|
describe("risk classification", () => {
|
|
test("classifies root shell and container changes as critical", () => {
|
|
expect(getToolRisk("unraid_system_shell")).toBe("critical");
|
|
expect(getToolRisk("unraid_toolbox_exec")).toBe("critical");
|
|
expect(getToolRisk("unraid_toolbox_status")).toBe("read");
|
|
expect(getToolRisk("unraid_system_job_start")).toBe("critical");
|
|
expect(getToolRisk("unraid_system_job_cleanup")).toBe("critical");
|
|
expect(getToolRisk("unraid_system_job_status")).toBe("read");
|
|
expect(getToolRisk("unraid_docker_modify")).toBe("critical");
|
|
expect(getToolRisk("unraid_docker_update_verified_batch")).toBe("critical");
|
|
expect(getToolRisk("unraid_docker_restart")).toBe("write");
|
|
expect(getToolRisk("unraid_network_lan_probe")).toBe("active");
|
|
expect(getToolRisk("unraid_docker_list")).toBe("read");
|
|
expect(getToolRisk("unraid_system_shell_readonly")).toBe("read");
|
|
expect(getToolRisk("unraid_files_inventory")).toBe("read");
|
|
expect(getToolRisk("unraid_system_health")).toBe("read");
|
|
expect(getToolRisk("unraid_ca_search")).toBe("active");
|
|
expect(getToolRisk("unraid_ca_install_preview")).toBe("active");
|
|
expect(getToolRisk("unraid_ca_install")).toBe("critical");
|
|
});
|
|
});
|
|
|
|
describe("agent toolbox tools", () => {
|
|
test("publishes explicit status and execution schemas", () => {
|
|
expect(toolByName("unraid_toolbox_status")).toBeDefined();
|
|
expect(toolByName("unraid_toolbox_exec")).toBeDefined();
|
|
});
|
|
});
|
|
|
|
describe("direct Unraid terminal", () => {
|
|
test("accepts a normal command with the extended timeout", async () => {
|
|
const tool = toolByName("unraid_system_shell");
|
|
expect(tool).toBeDefined();
|
|
const result = JSON.parse(await tool!.handler({ command: "printf ready", timeout_seconds: 301 }));
|
|
expect(result.exit_code).toBe(0);
|
|
expect(result.stdout).toBe("ready");
|
|
});
|
|
|
|
test("rejects excessive timeouts and oversized scripts", async () => {
|
|
const tool = toolByName("unraid_system_shell");
|
|
expect(tool).toBeDefined();
|
|
expect(() => tool!.handler({ command: "true", timeout_seconds: 1801 })).toThrow();
|
|
expect(() => tool!.handler({ command: "x".repeat(262_145) })).toThrow();
|
|
});
|
|
});
|
|
|
|
describe("tool-call concurrency recovery", () => {
|
|
test("stale leases heal without corrupting newer leases", () => {
|
|
const gate = new ToolCallGate(1);
|
|
const first = gate.acquire("first", 10, 100);
|
|
expect(first).not.toBeNull();
|
|
expect(gate.acquire("blocked", 10, 105)).toBeNull();
|
|
|
|
const second = gate.acquire("second", 100, 111);
|
|
expect(second).not.toBeNull();
|
|
gate.release(first!.id); // late release must not affect the newer lease
|
|
expect(gate.snapshot(112).active).toBe(1);
|
|
gate.release(second!.id);
|
|
expect(gate.snapshot(113).active).toBe(0);
|
|
});
|
|
|
|
test("hard timeout rejects work that never settles", async () => {
|
|
const never = new Promise<string>(() => {});
|
|
await expect(withHardTimeout(never, 20, "stuck test")).rejects.toThrow(
|
|
"exceeded hard timeout",
|
|
);
|
|
});
|
|
|
|
test("shell timeout rejects even when a descendant keeps pipes open", async () => {
|
|
const tool = toolByName("unraid_system_shell");
|
|
expect(tool).toBeDefined();
|
|
await expect(
|
|
tool!.handler({ command: "sleep 2 &", timeout_seconds: 1 }),
|
|
).rejects.toThrow("timed out after 1 seconds");
|
|
});
|
|
});
|
|
|
|
describe("read-only shell", () => {
|
|
test("executes an allowlisted program without a shell", async () => {
|
|
const result = JSON.parse(await runReadOnlyCommand("ls", ["-ld", "/"], 5));
|
|
expect(result.exit_code).toBe(0);
|
|
expect(result.mode).toBe("read-only");
|
|
});
|
|
|
|
test("rejects arbitrary programs and mutating subcommands", async () => {
|
|
await expect(runReadOnlyCommand("sh", ["-c", "id"], 5)).rejects.toThrow();
|
|
await expect(runReadOnlyCommand("ip", ["link", "set", "lo", "down"], 5)).rejects.toThrow();
|
|
await expect(runReadOnlyCommand("find", ["/tmp", "-delete"], 5)).rejects.toThrow();
|
|
await expect(runReadOnlyCommand("ss", ["-K", "dst", "127.0.0.1"], 5)).rejects.toThrow();
|
|
});
|
|
|
|
test("bounds read-only output server-side and reports truncation", async () => {
|
|
const result = JSON.parse(
|
|
await runReadOnlyCommand("cat", ["/dev/zero"], 1, 1000),
|
|
);
|
|
expect(result.truncated).toBe(true);
|
|
expect(result.stdout.length).toBeLessThanOrEqual(1000);
|
|
});
|
|
|
|
test("rejects excessive read-only output budgets", async () => {
|
|
await expect(runReadOnlyCommand("ls", ["/"], 5, 999)).rejects.toThrow();
|
|
await expect(runReadOnlyCommand("ls", ["/"], 5, 30001)).rejects.toThrow();
|
|
});
|
|
});
|
|
|
|
describe("Community Applications approval", () => {
|
|
test("rejects an install without a matching preview ticket before any write", async () => {
|
|
await expect(
|
|
installCommunityApp("aaaaaaaaaaaaaaaa", "mua-test", {}, false, true, "missing"),
|
|
).rejects.toThrow("Approval ticket missing");
|
|
});
|
|
});
|