From a1cae3820c5d82d29eb8169f8c4033f364985ae1 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:35:39 +0300 Subject: [PATCH 01/17] chore(vendor): add tinytools dependency Vendors the tinytools package so the build no longer depends on fetching it at build time. Auto-committed-on: dragonfly Co-authored-by: Medulla --- vendor/tinytools | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/vendor/tinytools b/vendor/tinytools index e2bf1be8..59e1cd3b 160000 --- a/vendor/tinytools +++ b/vendor/tinytools @@ -1 +1 @@ -Subproject commit e2bf1be8cff188cfaadcf1b3a3339da8b3e746c4 +Subproject commit 59e1cd3bbdc96ad54c8647537601b7eb6ed5a6f4 From f464990b7e905abea0d02d78a5c82f4eda005118 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:37:05 +0300 Subject: [PATCH 02/17] feat(tinyagents-definition): add definition crate Introduce a new crate for shared agent definition types so downstream crates can depend on a single source of truth. Auto-committed-on: dragonfly Co-authored-by: Medulla --- crates/tinyagents-definition/Cargo.toml | 2 ++ crates/tinyagents-definition/src/lib.rs | 13 +++++++++++++ 2 files changed, 15 insertions(+) diff --git a/crates/tinyagents-definition/Cargo.toml b/crates/tinyagents-definition/Cargo.toml index 000c439b..453f85e2 100644 --- a/crates/tinyagents-definition/Cargo.toml +++ b/crates/tinyagents-definition/Cargo.toml @@ -11,6 +11,8 @@ description = "Host-owned agent definition contract." [dependencies] async-trait = { workspace = true } serde = { workspace = true } +# `AgentDefinition::tool_rules` carries the shared tool-rule vocabulary. +tinytools = { path = "../../vendor/tinytools/crates/tinytools", version = "0.5.0" } [dev-dependencies] serde_json = { workspace = true } diff --git a/crates/tinyagents-definition/src/lib.rs b/crates/tinyagents-definition/src/lib.rs index 7bd7acb7..1c628ebc 100644 --- a/crates/tinyagents-definition/src/lib.rs +++ b/crates/tinyagents-definition/src/lib.rs @@ -63,6 +63,11 @@ pub struct AgentDefinition { /// Canonical tool names this agent may use. #[serde(default)] pub tools: Vec, + /// Pattern rules narrowing which tools this agent may see and call, on + /// top of [`Self::tools`]. Evaluated by the harness on the catalogue, + /// tool search and every call; see [`tinytools::ToolRules`]. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub tool_rules: Option, } impl AgentDefinition { @@ -81,6 +86,7 @@ impl AgentDefinition { model: None, subagents: Vec::new(), tools: Vec::new(), + tool_rules: None, } } @@ -113,6 +119,13 @@ impl AgentDefinition { self } + /// Sets the pattern rules narrowing this agent's tools. + #[must_use] + pub fn with_tool_rules(mut self, rules: tinytools::ToolRules) -> Self { + self.tool_rules = Some(rules); + self + } + /// Sets the host-defined routing role. #[must_use] pub fn with_role(mut self, role: impl Into) -> Self { From e5055888890fdfd5a17e2a4c52e3a87e0b944885 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:37:20 +0300 Subject: [PATCH 03/17] refactor(tool): split rule matching into a dedicated module Moved the rule matching logic out of the tool module into its own rules submodule so the matching behaviour can be tested and reused on its own. No behaviour change. Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../tinyagents-harness/src/tool/rules/mod.rs | 38 +++++++++++++++++++ 1 file changed, 38 insertions(+) create mode 100644 crates/tinyagents-harness/src/tool/rules/mod.rs diff --git a/crates/tinyagents-harness/src/tool/rules/mod.rs b/crates/tinyagents-harness/src/tool/rules/mod.rs new file mode 100644 index 00000000..4ea7b2cb --- /dev/null +++ b/crates/tinyagents-harness/src/tool/rules/mod.rs @@ -0,0 +1,38 @@ +//! Tool rules in the agent loop: one gate for the catalogue, tool search and +//! every call. +//! +//! [`tinytools::ToolRules`] is the vocabulary; this module is where the loop +//! applies it. A run's rules come from two places and stack as layers of one +//! [`tinytools::ToolRuleSet`], so neither can widen the other: +//! +//! - [`RunPolicy::tool_rules`](crate::runtime::RunPolicy::tool_rules) — the +//! harness-wide rules and the [`tinytools::RuleContext`] they are evaluated +//! in (the channel, the agent, the origin of the turn); +//! - the hosted agent definition's +//! [`tool_rules`](tinyagents_definition::AgentDefinition::tool_rules). +//! +//! The definition's exact `tools` allowlist still applies first, unchanged. +//! +//! The [`ToolGate`] answers three questions with the same rules, so a tool a +//! rule removes from the catalogue cannot be found through `tool_search` or +//! called by a name the model guessed: +//! +//! | Site | Surface | +//! |---|---| +//! | direct schemas on the request (and mid-run toolset changes) | `catalog` | +//! | deferred catalogue, `tool_search` answers, replayed promotions | `search` | +//! | admission of a model call, and of a nested call | `call` | +//! +//! At call time the rules also yield an [`tinytools::ApprovalDirective`]: a +//! `require_approval` rule defers the call for approval exactly like a tool +//! that declares `approval_required`, and an `auto_approve` rule waives that +//! declaration. + +mod types; + +pub use types::ToolRulePolicy; +pub(crate) use types::{CallGate, ToolGate}; + +#[cfg(test)] +#[path = "mod_tests.rs"] +mod tests; From eee1de6c1ce3d88271646a0708a56d4c744e8a28 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:37:48 +0300 Subject: [PATCH 04/17] refactor(tool): rename rule types for clarity Renamed the rule type definitions to better reflect their purpose and improve readability across the harness. No behaviour change. Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../src/tool/rules/types.rs | 152 ++++++++++++++++++ 1 file changed, 152 insertions(+) create mode 100644 crates/tinyagents-harness/src/tool/rules/types.rs diff --git a/crates/tinyagents-harness/src/tool/rules/types.rs b/crates/tinyagents-harness/src/tool/rules/types.rs new file mode 100644 index 00000000..1a27400b --- /dev/null +++ b/crates/tinyagents-harness/src/tool/rules/types.rs @@ -0,0 +1,152 @@ +//! The run-level rule policy and the gate the loop consults. + +use std::collections::HashSet; +use std::sync::Arc; + +use serde_json::Value; +use tinytools::{ + ApprovalDirective, RuleContext, Surface, Tool, ToolExposure, ToolRuleSet, ToolRules, + ToolSubject, +}; + +/// Harness-wide tool rules and the context they are evaluated in. +/// +/// Set on [`RunPolicy::tool_rules`](crate::runtime::RunPolicy::tool_rules). +/// The default holds no rules and admits every tool, so a harness that never +/// sets it behaves exactly as before. +#[derive(Clone, Debug, Default, PartialEq)] +pub struct ToolRulePolicy { + /// The rule layers every tool must pass. + pub rules: Arc, + /// Attributes `when` conditions match against: `channel`, `agent`, …. + pub context: RuleContext, +} + +impl ToolRulePolicy { + /// A policy evaluating `rules` in an empty context. + #[must_use] + pub fn new(rules: impl Into) -> Self { + Self { + rules: Arc::new(rules.into()), + context: RuleContext::new(), + } + } + + /// Replaces the evaluation context. + #[must_use] + pub fn with_context(mut self, context: RuleContext) -> Self { + self.context = context; + self + } + + /// Whether the policy admits every tool, so the loop can skip it. + #[must_use] + pub fn is_permissive(&self) -> bool { + self.rules.is_permissive() + } +} + +/// What the rules say about one model call. +#[derive(Debug, Clone, PartialEq, Eq)] +pub(crate) enum CallGate { + /// The call may proceed, under this approval directive. + Admit(ApprovalDirective), + /// The call is refused; the message names the rule. + Refuse(String), +} + +/// The run's resolved tool gate: the definition's exact allowlist plus every +/// rule layer that applies, evaluated in one context. +#[derive(Debug, Clone, Default)] +pub(crate) struct ToolGate { + allowed: Option>, + rules: Option>, + context: RuleContext, +} + +impl ToolGate { + /// Builds the gate from the resolved allowlist, the harness policy and a + /// hosted definition's rules. A permissive rule set is dropped so an + /// unconfigured run pays nothing per tool. + pub(crate) fn new( + allowed: Option>, + policy: &ToolRulePolicy, + definition: Option<&ToolRules>, + ) -> Self { + let rules = match definition.filter(|layer| !layer.is_permissive()) { + Some(layer) => { + let mut merged = (*policy.rules).clone(); + merged.push(layer.clone()); + Some(Arc::new(merged)) + } + None => (!policy.is_permissive()).then(|| policy.rules.clone()), + }; + Self { + allowed, + rules, + context: policy.context.clone(), + } + } + + /// Whether the exact allowlist admits `name`. Rules are not consulted. + pub(crate) fn allows_name(&self, name: &str) -> bool { + self.allowed + .as_ref() + .is_none_or(|allowed| allowed.contains(name)) + } + + /// Whether `name` may be listed on `surface` (catalogue or search). + /// Evaluated against `tool` when the caller has it, otherwise against the + /// bare name. + pub(crate) fn lists(&self, name: &str, tool: Option<&dyn Tool>, surface: Surface) -> bool { + if !self.allows_name(name) { + return false; + } + let Some(rules) = &self.rules else { + return true; + }; + let subject = tool.map_or_else(|| ToolSubject::named(name), ToolSubject::of); + let decision = rules.evaluate(&subject, &self.context, surface, None); + if !decision.visible { + tracing::debug!( + target: "tinyagents::tool_rules", + tool = %name, + surface = ?surface, + rule = ?decision.blocked_by, + "[tool_rules] tool withheld from listing" + ); + } + decision.visible + } + + /// [`Self::lists`] on the surface a tool's exposure puts it on: search + /// for a deferred tool, the catalogue otherwise. + pub(crate) fn lists_tool(&self, tool: &dyn Tool) -> bool { + let surface = if tool.exposure() == ToolExposure::Deferred { + Surface::Search + } else { + Surface::Catalog + }; + self.lists(tool.name(), Some(tool), surface) + } + + /// Decides a concrete call. The allowlist is checked by the caller before + /// lookup; this applies the rules, including any indirect target. + pub(crate) fn admit_call(&self, tool: &dyn Tool, args: &Value) -> CallGate { + let Some(rules) = &self.rules else { + return CallGate::Admit(ApprovalDirective::Default); + }; + let decision = rules.evaluate_call(tool, &self.context, args); + if decision.callable { + CallGate::Admit(decision.approval) + } else { + tracing::debug!( + target: "tinyagents::tool_rules", + tool = %tool.name(), + rule = ?decision.blocked_by, + "[tool_rules] call refused" + ); + CallGate::Refuse(decision.refusal(tool.name())) + } + } +} From 6a5a3c1c2a713f5976b24da7942939e14dfd5dd1 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:38:01 +0300 Subject: [PATCH 05/17] feat(harness): add tool call support to agent runtime The agent runtime now handles tool calls emitted by the model, dispatching them through the tool registry and feeding results back into the conversation. This lets agents invoke registered tools during a run instead of only producing text responses. Auto-committed-on: dragonfly Co-authored-by: Medulla --- crates/tinyagents-harness/src/runtime/agent.rs | 2 ++ crates/tinyagents-harness/src/runtime/types.rs | 12 ++++++++++++ crates/tinyagents-harness/src/tool/mod.rs | 3 +++ 3 files changed, 17 insertions(+) diff --git a/crates/tinyagents-harness/src/runtime/agent.rs b/crates/tinyagents-harness/src/runtime/agent.rs index 2b296ac9..0731bf7e 100644 --- a/crates/tinyagents-harness/src/runtime/agent.rs +++ b/crates/tinyagents-harness/src/runtime/agent.rs @@ -1074,6 +1074,7 @@ impl AgentHarness = definition.tools.into_iter().collect(); let allowed_tools = (!declared_tools.is_empty()).then_some(declared_tools); + let tool_rules = definition.tool_rules; Ok(PreparedAgentTurn { binding: HostInvocationBinding { host: host.clone(), @@ -1081,6 +1082,7 @@ impl AgentHarness { /// "declared empty" share one fail-closed code path instead of an empty /// set silently meaning "unrestricted", as it used to (I-9)). pub(crate) allowed_tools: Option>, + /// The resolved definition's pattern rules, stacked on the harness + /// policy's by the loop's tool gate. + pub(crate) tool_rules: Option, /// Per-turn ordered, nonblocking projection to the optional progress sink. pub(crate) progress: Option, /// The exact invocation-local runtime inherited by authorized children. @@ -73,6 +76,7 @@ impl Clone for HostInvocationBinding, + /// Pattern rules deciding which tools the model may see (catalogue and + /// tool search) and call, evaluated in the policy's context. Stacks with + /// a hosted definition's own rules and its exact `tools` allowlist; see + /// [`crate::tool::ToolRulePolicy`]. + /// + /// The default holds no rules and admits every tool. + pub tool_rules: crate::tool::ToolRulePolicy, /// Whether the loop parses ``-style text-dialect markup out of /// an assistant's visible text under a native tool dialect (see /// [`RunPolicy::tool_dialect`]). A forced text dialect @@ -732,6 +743,7 @@ impl Default for RunPolicy { text_dialect_recovery: TextDialectRecovery::default(), discovery: crate::tool::discover::ToolDiscoveryPolicy::default(), tool_schemas: None, + tool_rules: crate::tool::ToolRulePolicy::default(), output_retry: OutputRetryPolicy::default(), end_strategy: EndStrategy::default(), structured_strategy_override: None, diff --git a/crates/tinyagents-harness/src/tool/mod.rs b/crates/tinyagents-harness/src/tool/mod.rs index 84135f76..b4e8def3 100644 --- a/crates/tinyagents-harness/src/tool/mod.rs +++ b/crates/tinyagents-harness/src/tool/mod.rs @@ -9,6 +9,7 @@ pub mod discover; pub mod effects; pub mod nested; pub mod packs; +mod rules; mod progress; mod prompt; mod schema; @@ -34,6 +35,8 @@ pub use effects::{ ToolEffectStatus, }; pub use nested::NestedToolRunner; +pub(crate) use rules::{CallGate, ToolGate}; +pub use rules::ToolRulePolicy; pub(crate) use progress::{ToolProgressGate, ToolProgressLimits}; pub use prompt::*; pub use schema::*; From c8275a6fb05f4eba3e4f7feff9b88d281b205571 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:38:08 +0300 Subject: [PATCH 06/17] test(harness): cover runtime module wiring Add tests for the runtime module to verify its wiring and behaviour. Auto-committed-on: dragonfly Co-authored-by: Medulla --- crates/tinyagents-harness/src/runtime/mod_tests.rs | 2 ++ 1 file changed, 2 insertions(+) diff --git a/crates/tinyagents-harness/src/runtime/mod_tests.rs b/crates/tinyagents-harness/src/runtime/mod_tests.rs index f609f075..4b4d55fd 100644 --- a/crates/tinyagents-harness/src/runtime/mod_tests.rs +++ b/crates/tinyagents-harness/src/runtime/mod_tests.rs @@ -3493,6 +3493,7 @@ fn host_invocation_binding_fails_closed_on_a_state_mismatch() { model_pin: None, role: None, allowed_tools: None, + tool_rules: None, progress: None, runtime: None, }), @@ -3545,6 +3546,7 @@ fn child_with_data_never_propagates_host_authority() { model_pin: None, role: None, allowed_tools: None, + tool_rules: None, progress: None, runtime: None, }), From f30bd4d5b0961c732d4b958255ba4f238aa5e1be Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:38:30 +0300 Subject: [PATCH 07/17] =?UTF-8?q?chore:=20I=20don't=20see=20a=20diff=20in?= =?UTF-8?q?=20your=20message=20=E2=80=94=20the=20Files,=20Stat,=20and=20Di?= =?UTF-8?q?ff=20sections=20are=20all=20empty.=20Coul?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../src/agent_loop/tools.rs | 100 ++++++++++++------ 1 file changed, 70 insertions(+), 30 deletions(-) diff --git a/crates/tinyagents-harness/src/agent_loop/tools.rs b/crates/tinyagents-harness/src/agent_loop/tools.rs index 961264f3..8c177976 100644 --- a/crates/tinyagents-harness/src/agent_loop/tools.rs +++ b/crates/tinyagents-harness/src/agent_loop/tools.rs @@ -283,13 +283,23 @@ impl AgentHarness { }) } + /// Resolves the run's [`ToolGate`]: the exact allowlist from + /// [`Self::resolve_tool_allowlist`], the harness policy's + /// [`tool_rules`](crate::runtime::RunPolicy::tool_rules) and, on a hosted + /// run, the resolved definition's own rules. Every listing and admission + /// site asks this one gate, so the catalogue, `tool_search` and dispatch + /// cannot disagree about a tool. + pub(super) fn resolve_tool_gate(&self, ctx: &RunContext) -> Result { + let allowed = self.resolve_tool_allowlist(ctx)?; + let binding = crate::runtime::host_invocation_binding::(ctx)?; + let definition_rules = binding.as_ref().and_then(|b| b.tool_rules.as_ref()); + Ok(ToolGate::new(allowed, &self.policy.tool_rules, definition_rules)) + } + /// Builds the run's deferred-tool catalogue: every - /// [`tinytools::ToolExposure::Deferred`] registration the host allow-list - /// admits, or an empty catalogue when discovery is disabled. - pub(super) fn deferred_catalog( - &self, - host_allows: &dyn Fn(&str) -> bool, - ) -> crate::tool::discover::DeferredCatalog { + /// [`tinytools::ToolExposure::Deferred`] registration the gate lets the + /// model search for, or an empty catalogue when discovery is disabled. + pub(super) fn deferred_catalog(&self, gate: &ToolGate) -> crate::tool::discover::DeferredCatalog { if !self.policy.discovery.enabled { return crate::tool::discover::DeferredCatalog::default(); } @@ -297,7 +307,13 @@ impl AgentHarness { .tools .deferred_schemas_with_families() .into_iter() - .filter(|(schema, _)| host_allows(&schema.name)) + .filter(|(schema, _)| { + gate.lists( + &schema.name, + self.tools.get(&schema.name).as_deref(), + tinytools::Surface::Search, + ) + }) .collect::>(); if let Some(preparation) = &self.policy.tool_schemas { let families: Vec> = @@ -334,13 +350,8 @@ impl AgentHarness { // `fail_closed_tool_allowlist` policy as the direct tool set built in // `run_loop_body` — an empty declared list never falls back to // "unrestricted" here either. - let allowed_tools = self.resolve_tool_allowlist(ctx)?; - let host_allows = |name: &str| { - allowed_tools - .as_ref() - .is_none_or(|allowed| allowed.contains(name)) - }; - let catalog = self.deferred_catalog(&host_allows); + let gate = self.resolve_tool_gate(ctx)?; + let catalog = self.deferred_catalog(&gate); if catalog.is_empty() { // Nothing was deferred, so the bridge was never advertised; let // the call fall through to the unknown-tool policy. @@ -576,6 +587,29 @@ impl AgentHarness { // execution and must not overwrite what the gate evaluates. let model_arguments = call.arguments.clone(); + // Tool rules, before any hook runs, so an approval middleware never + // asks a human about a call the rules refuse. Evaluated against the + // registered tool (an unregistered name falls through to the + // unknown-tool policy below) and any target it dispatches to, on the + // raw provider arguments the host gate also sees. + let gate = self.resolve_tool_gate(ctx)?; + let mut rule_approval = tinytools::ApprovalDirective::Default; + if gate.allows_name(&call.name) + && let Some(dispatch) = self.tools.model_dispatch(&call.name) + { + match gate.admit_call(dispatch.tool().as_ref(), &model_arguments) { + CallGate::Admit(approval) => rule_approval = approval, + CallGate::Refuse(message) => { + ctx.limits.rollback_tool_calls(1); + return Ok(ResolvedToolCall::Answered(with_refusal_metadata( + ctx, + &call.id, + tinytools::ToolResult::error(message), + ))); + } + } + } + // The slot is *reserved* above (cap-first, so a middleware hook never // runs for a call the budget has already refused) and *released* here // when `before_tool` refuses the call — an approval denial or an @@ -670,10 +704,7 @@ impl AgentHarness { // Hosted turns carry an explicit definition allowlist. Do not merely // hide disallowed schemas: a model can still fabricate a name, so the // dispatch boundary must reject it too. - let allowed_tools = self.resolve_tool_allowlist(ctx)?; - let is_allowed = allowed_tools - .as_ref() - .is_none_or(|allowed| allowed.contains(&call.name)); + let is_allowed = gate.allows_name(&call.name); let (dispatch, tool) = match is_allowed .then(|| self.tools.model_dispatch(&call.name)) .flatten() @@ -708,10 +739,12 @@ impl AgentHarness { UnknownToolPolicy::Rewrite { tool_name } => self .tools .dispatch(tool_name) - .filter(|_| { - allowed_tools - .as_ref() - .is_none_or(|allowed| allowed.contains(tool_name)) + .filter(|dispatch| { + gate.allows_name(tool_name) + && matches!( + gate.admit_call(dispatch.tool().as_ref(), &arguments), + CallGate::Admit(_) + ) }) .map(|dispatch| (tool_name.clone(), dispatch)), _ => None, @@ -740,16 +773,15 @@ impl AgentHarness { // `unknown_tool`). The attempted arguments are echoed in the // message and kept on the `UnknownToolCall` event. This consumed one tool-call // budget slot above, bounding the loop. - let host_allows = |name: &str| { - allowed_tools - .as_ref() - .is_none_or(|allowed| allowed.contains(name)) - }; let available = self .tools .model_callable_names() .into_iter() - .filter(|name| host_allows(name)) + .filter(|name| { + self.tools + .get(name) + .is_some_and(|tool| gate.lists_tool(tool.as_ref())) + }) .collect::>(); // A host-registered `tool_search` takes precedence over the // intrinsic bridge (see `admit_tool_call`), so only advertise @@ -758,7 +790,7 @@ impl AgentHarness { .tools .dispatch(crate::tool::discover::TOOL_SEARCH_NAME) .is_none() - && !self.deferred_catalog(&host_allows).is_empty(); + && !self.deferred_catalog(&gate).is_empty(); let message = super::unknown_tool::unknown_tool_message( &requested, &arguments, @@ -886,7 +918,15 @@ impl AgentHarness { ))); } let policy = tool.policy(); - if policy.access.approval_required { + // A `require_approval` rule defers like a declared + // `approval_required`; an `auto_approve` rule waives the + // declaration (a stricter rule elsewhere already won). + let needs_approval = match rule_approval { + tinytools::ApprovalDirective::Required => true, + tinytools::ApprovalDirective::Waived => false, + tinytools::ApprovalDirective::Default => policy.access.approval_required, + }; + if needs_approval { ctx.limits.rollback_tool_calls(1); let metadata = serde_json::to_value(&policy.display) .ok() From a3893b9363fbb1baeacf83f6241cbf875245044d Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:38:44 +0300 Subject: [PATCH 08/17] refactor(agent_loop): split tool surface out of run loop Move the tool surface and tool definitions into their own modules so the run loop only handles orchestration. This keeps the loop easier to read and gives the tool wiring a place to grow without bloating the loop. Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../src/agent_loop/run_loop.rs | 11 +++------- .../src/agent_loop/tool_surface.rs | 22 ++++++++++++------- .../src/agent_loop/tools.rs | 5 +++-- 3 files changed, 20 insertions(+), 18 deletions(-) diff --git a/crates/tinyagents-harness/src/agent_loop/run_loop.rs b/crates/tinyagents-harness/src/agent_loop/run_loop.rs index 43b590bf..8dffc483 100644 --- a/crates/tinyagents-harness/src/agent_loop/run_loop.rs +++ b/crates/tinyagents-harness/src/agent_loop/run_loop.rs @@ -216,13 +216,8 @@ impl AgentHarness { // host allow-list gates both halves; `resolve_tool_allowlist` (not a raw // read of `binding.allowed_tools`) is what applies I-9's fail-closed // default, so an empty declared list denies every tool. - let allowed_tools = self.resolve_tool_allowlist(ctx)?; - let host_allows = |name: &str| { - allowed_tools - .as_ref() - .is_none_or(|allowed| allowed.contains(name)) - }; - let mut surface = self.build_tool_surface(ctx, messages, &host_allows).await?; + let gate = self.resolve_tool_gate(ctx)?; + let mut surface = self.build_tool_surface(ctx, messages, &gate).await?; self.check_structured_schema_name(&surface.tool_schemas)?; status.mark_running(HarnessPhase::Middleware); @@ -409,7 +404,7 @@ impl AgentHarness { // the transcript it actually rewrote. ctx.flush_transcript(self.policy.capture, messages); let rewrote = surface - .declare_toolset_changes(self, ctx, messages, &host_allows, patch_profile.as_ref()) + .declare_toolset_changes(self, ctx, messages, &gate, patch_profile.as_ref()) .await?; let rewrote = surface.promote_discovered(messages, patch_profile.as_ref()) || rewrote; if rewrote { diff --git a/crates/tinyagents-harness/src/agent_loop/tool_surface.rs b/crates/tinyagents-harness/src/agent_loop/tool_surface.rs index 482fb6e8..40f86287 100644 --- a/crates/tinyagents-harness/src/agent_loop/tool_surface.rs +++ b/crates/tinyagents-harness/src/agent_loop/tool_surface.rs @@ -26,13 +26,19 @@ impl AgentHarness { async fn direct_tool_schemas( &self, ctx: &RunContext, - host_allows: &(dyn Fn(&str) -> bool + Sync), + gate: &ToolGate, ) -> Result> { let mut schemas = self .tools .schemas() .into_iter() - .filter(|schema| host_allows(&schema.name)) + .filter(|schema| { + gate.lists( + &schema.name, + self.tools.get(&schema.name).as_deref(), + tinytools::Surface::Catalog, + ) + }) .collect::>(); if let Some(toolset) = &self.toolset { let existing: HashSet<&str> = @@ -42,7 +48,7 @@ impl AgentHarness { .await? .into_iter() .filter(|tool| tool.exposure() == tinytools::ToolExposure::Direct) - .filter(|tool| host_allows(tool.name())) + .filter(|tool| gate.lists(tool.name(), Some(tool.as_ref()), tinytools::Surface::Catalog)) .filter(|tool| !existing.contains(tool.name())) .map(|tool| crate::tool::provider_schema(tool.as_ref())) .collect(); @@ -69,15 +75,15 @@ impl AgentHarness { &self, ctx: &RunContext, messages: &[Message], - host_allows: &(dyn Fn(&str) -> bool + Sync), + gate: &ToolGate, ) -> Result { - let mut tool_schemas = self.direct_tool_schemas(ctx, host_allows).await?; + let mut tool_schemas = self.direct_tool_schemas(ctx, gate).await?; // Captured before the bridge schemas are appended below, so // `ToolsAdvertised.direct` reports the actual `Direct`-exposure count. let direct_schema_count = tool_schemas.len(); let direct_tool_schemas = tool_schemas.clone(); let mut bridge_schemas: Vec = Vec::new(); - let deferred_catalog = self.deferred_catalog(host_allows); + let deferred_catalog = self.deferred_catalog(gate); // A resumed transcript carries promoted declarations in SystemMessage // patches. Only restore names still admitted into this run's catalogue. let promoted_schemas: BTreeMap = @@ -181,13 +187,13 @@ impl ToolSurface { harness: &AgentHarness, ctx: &RunContext, messages: &mut Vec, - host_allows: &(dyn Fn(&str) -> bool + Sync), + gate: &ToolGate, patch_profile: Option<&tinyinference_llm::model::ModelProfile>, ) -> Result { if harness.toolset.is_none() { return Ok(false); } - let live_schemas = harness.direct_tool_schemas(ctx, host_allows).await?; + let live_schemas = harness.direct_tool_schemas(ctx, gate).await?; let mut rewrote = false; if let Some(patch) = tool_changes::diff_tool_set(&self.declared_tool_schemas, &live_schemas) { diff --git a/crates/tinyagents-harness/src/agent_loop/tools.rs b/crates/tinyagents-harness/src/agent_loop/tools.rs index 8c177976..33554cf9 100644 --- a/crates/tinyagents-harness/src/agent_loop/tools.rs +++ b/crates/tinyagents-harness/src/agent_loop/tools.rs @@ -99,8 +99,9 @@ use super::model_call::ToolCallBase; use super::*; use crate::tool::{ - DeferredToolRequests, LedgerFailure, ToolDispatch, ToolEffectSettle, ToolEffectStart, - ToolEffectStatus, ToolProgressGate, ToolProgressLimits, provider_schema, + CallGate, DeferredToolRequests, LedgerFailure, ToolDispatch, ToolEffectSettle, + ToolEffectStart, ToolEffectStatus, ToolGate, ToolProgressGate, ToolProgressLimits, + provider_schema, }; use sha2::{Digest, Sha256}; use tinyinference_llm::message::ContentBlock; From 87be68bac8e0a74b985ddcb1251896545d550461 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:38:54 +0300 Subject: [PATCH 09/17] refactor(agent_loop): split nested agent loop into its own module Moved the nested agent loop logic out of run_loop into a dedicated nested module and extracted tool surface handling into tool_surface. This separates the nested execution path from the top-level loop, making both easier to follow without changing behaviour. Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../tinyagents-harness/src/agent_loop/nested.rs | 16 +++++++++++----- .../src/agent_loop/run_loop.rs | 7 ++++--- .../src/agent_loop/tool_surface.rs | 1 + 3 files changed, 16 insertions(+), 8 deletions(-) diff --git a/crates/tinyagents-harness/src/agent_loop/nested.rs b/crates/tinyagents-harness/src/agent_loop/nested.rs index 5946b91d..13568a19 100644 --- a/crates/tinyagents-harness/src/agent_loop/nested.rs +++ b/crates/tinyagents-harness/src/agent_loop/nested.rs @@ -856,17 +856,23 @@ impl AgentHarness { mut call: ToolCall, ) -> Result<(Arc>, ToolCall)> { let name = call.name.clone(); - let allowed_tools = self.resolve_tool_allowlist(ctx)?; - let is_allowed = allowed_tools - .as_ref() - .is_none_or(|allowed| allowed.contains(&name)); - let Some(dispatch) = is_allowed + let gate = self.resolve_tool_gate(ctx)?; + let Some(dispatch) = gate + .allows_name(&name) .then(|| self.tools.model_dispatch(&name)) .flatten() else { return Err(TinyAgentsError::ToolNotFound(name)); }; let tool = dispatch.tool(); + // Tool rules apply to a nested call exactly as to a model call; a + // refusal reads like any other nested-call failure. + let rule_approval = match gate.admit_call(tool.as_ref(), &call.arguments) { + crate::tool::CallGate::Admit(approval) => approval, + crate::tool::CallGate::Refuse(message) => { + return Err(TinyAgentsError::ToolFailed(message)); + } + }; // Same ordering rule as `admit_tool_call`: strip host-injected keys, // inject the authoritative values, then validate the model-facing diff --git a/crates/tinyagents-harness/src/agent_loop/run_loop.rs b/crates/tinyagents-harness/src/agent_loop/run_loop.rs index 8dffc483..99a50ec1 100644 --- a/crates/tinyagents-harness/src/agent_loop/run_loop.rs +++ b/crates/tinyagents-harness/src/agent_loop/run_loop.rs @@ -213,9 +213,10 @@ impl AgentHarness { // Build the tool surface once (see `tool_surface.rs`): the direct tool // set plus the deferred catalogue behind the `tool_search` bridge. The - // host allow-list gates both halves; `resolve_tool_allowlist` (not a raw - // read of `binding.allowed_tools`) is what applies I-9's fail-closed - // default, so an empty declared list denies every tool. + // tool gate (the host allow-list plus the run's tool rules) gates both + // halves; `resolve_tool_allowlist` underneath it (not a raw read of + // `binding.allowed_tools`) is what applies I-9's fail-closed default, + // so an empty declared list denies every tool. let gate = self.resolve_tool_gate(ctx)?; let mut surface = self.build_tool_surface(ctx, messages, &gate).await?; self.check_structured_schema_name(&surface.tool_schemas)?; diff --git a/crates/tinyagents-harness/src/agent_loop/tool_surface.rs b/crates/tinyagents-harness/src/agent_loop/tool_surface.rs index 40f86287..9140a88f 100644 --- a/crates/tinyagents-harness/src/agent_loop/tool_surface.rs +++ b/crates/tinyagents-harness/src/agent_loop/tool_surface.rs @@ -12,6 +12,7 @@ use std::collections::{BTreeMap, BTreeSet, HashSet}; use super::tool_changes; +use crate::tool::ToolGate; use super::*; impl AgentHarness { From bab764e5d0fd013573c22fdf3ebf69e50e9ba363 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:39:07 +0300 Subject: [PATCH 10/17] fix(agent_loop): honor rule approval directives in nested tool checks Nested tool calls now consult the rule's approval directive instead of relying solely on the tool policy, so rules that waive approval no longer trigger an approval error and rules that require it always do. Auto-committed-on: dragonfly Co-authored-by: Medulla --- crates/tinyagents-harness/src/agent_loop/nested.rs | 7 ++++++- 1 file changed, 6 insertions(+), 1 deletion(-) diff --git a/crates/tinyagents-harness/src/agent_loop/nested.rs b/crates/tinyagents-harness/src/agent_loop/nested.rs index 13568a19..f03f9916 100644 --- a/crates/tinyagents-harness/src/agent_loop/nested.rs +++ b/crates/tinyagents-harness/src/agent_loop/nested.rs @@ -917,7 +917,12 @@ impl AgentHarness { )) })?; - if crate::tool::is_external_tool(tool.as_ref()) || tool.policy().access.approval_required { + let needs_approval = match rule_approval { + tinytools::ApprovalDirective::Required => true, + tinytools::ApprovalDirective::Waived => false, + tinytools::ApprovalDirective::Default => tool.policy().access.approval_required, + }; + if crate::tool::is_external_tool(tool.as_ref()) || needs_approval { return Err(approval_error(&name)); } From 8953d97ff45f181dd18dad791e0664a94543c93d Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:39:46 +0300 Subject: [PATCH 11/17] chore(deps): add tinytools to lockfile Record tinytools as a dependency in Cargo.lock so the resolved dependency graph stays in sync. Auto-committed-on: dragonfly Co-authored-by: Medulla --- Cargo.lock | 1 + 1 file changed, 1 insertion(+) diff --git a/Cargo.lock b/Cargo.lock index 0c1461dc..a2da44ec 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1704,6 +1704,7 @@ dependencies = [ "async-trait", "serde", "serde_json", + "tinytools", "tokio", ] From cb4ffc642e65ce81218a45d9d110d2553c423752 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:41:10 +0300 Subject: [PATCH 12/17] test(harness): cover tool rule evaluation in agent loop Adds unit tests for the tool rules used by the agent loop, exercising the allow and deny paths so future changes to rule evaluation are caught by the suite. Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../src/agent_loop/tool_rules_tests.rs | 356 ++++++++++++++++++ 1 file changed, 356 insertions(+) create mode 100644 crates/tinyagents-harness/src/agent_loop/tool_rules_tests.rs diff --git a/crates/tinyagents-harness/src/agent_loop/tool_rules_tests.rs b/crates/tinyagents-harness/src/agent_loop/tool_rules_tests.rs new file mode 100644 index 00000000..6a074e39 --- /dev/null +++ b/crates/tinyagents-harness/src/agent_loop/tool_rules_tests.rs @@ -0,0 +1,356 @@ +//! Tool rules in the agent loop: one rule set decides the catalogue, the +//! `tool_search` results and every call, including approval, nested calls, +//! indirect targets and a hosted definition's own rules. + +use std::sync::{Arc, Mutex}; + +use async_trait::async_trait; +use serde_json::{Value, json}; + +use crate::context::{RunConfig, RunContext}; +use crate::host::{ + AllowAllSecurityGate, FixedModelResolver, HostCapabilities, StaticContextComposer, +}; +use crate::runtime::{AgentHarness, AgentInvocation, AgentTurnRequest, RunPolicy}; +use crate::testkit::ScriptedModel; +use crate::tool::ToolRulePolicy; +use tinyagents_definition::{AgentDefinition, InMemoryDefinitionRegistry}; +use tinyinference_llm::message::Message; +use tinyinference_llm::model::ModelResponse; +use tinyinference_llm::tool::ToolCall; +use tinytools::{ + RuleContext, Tool, ToolExposure, ToolPolicy, ToolResult, ToolRuleSet, ToolRules, ToolSubject, +}; + +// ── Helpers ───────────────────────────────────────────────────────────────── + +struct RuleTool { + name: &'static str, + exposure: ToolExposure, + policy: ToolPolicy, + /// For a dispatcher: the tool its `action` argument names. + dispatches: bool, + seen: Mutex>, +} + +impl RuleTool { + fn new(name: &'static str) -> Arc { + Self::build(name, ToolExposure::Direct, ToolPolicy::read_only(), false) + } + + fn deferred(name: &'static str) -> Arc { + Self::build(name, ToolExposure::Deferred, ToolPolicy::read_only(), false) + } + + fn approval_gated(name: &'static str) -> Arc { + Self::build( + name, + ToolExposure::Direct, + ToolPolicy::classified().requiring_approval(), + false, + ) + } + + fn dispatcher(name: &'static str) -> Arc { + Self::build(name, ToolExposure::Direct, ToolPolicy::read_only(), true) + } + + fn build( + name: &'static str, + exposure: ToolExposure, + policy: ToolPolicy, + dispatches: bool, + ) -> Arc { + Arc::new(Self { + name, + exposure, + policy, + dispatches, + seen: Mutex::new(Vec::new()), + }) + } + + fn calls(&self) -> usize { + self.seen.lock().unwrap().len() + } +} + +#[async_trait] +impl Tool for RuleTool { + fn name(&self) -> &str { + self.name + } + fn description(&self) -> &str { + "rule test tool" + } + fn parameters_schema(&self) -> Value { + json!({"type": "object"}) + } + fn exposure(&self) -> ToolExposure { + self.exposure + } + fn policy(&self) -> ToolPolicy { + self.policy.clone() + } + fn indirect_target(&self, args: &Value) -> Option { + if !self.dispatches { + return None; + } + args.get("action")?.as_str().map(ToolSubject::named) + } + async fn execute(&self, arguments: Value) -> anyhow::Result { + self.seen.lock().unwrap().push(arguments); + Ok(ToolResult::success(format!("{} ran", self.name))) + } +} + +fn calls(calls: Vec<(&str, &str, Value)>) -> ModelResponse { + let mut response = ModelResponse::assistant(""); + for (id, name, args) in calls { + response + .message + .tool_calls + .push(ToolCall::new(id, name, args)); + } + response +} + +fn rules(value: Value) -> ToolRules { + serde_json::from_value(value).expect("rules parse") +} + +fn harness_with(model: Arc, policy: ToolRulePolicy) -> AgentHarness<()> { + let mut harness: AgentHarness<()> = AgentHarness::new(); + harness.register_model("scripted", model as _); + harness.with_policy(RunPolicy { + tool_rules: policy, + ..RunPolicy::default() + }); + harness +} + +fn tool_text(messages: &[Message], call_id: &str) -> String { + messages + .iter() + .find_map(|message| match message { + Message::Tool(tool) if tool.tool_call_id == call_id => Some(message.text()), + _ => None, + }) + .unwrap_or_else(|| panic!("no result for {call_id}")) +} + +fn tool_names(request: &tinyinference_llm::model::ModelRequest) -> Vec { + request.tools.iter().map(|tool| tool.name.clone()).collect() +} + +async fn run(harness: &AgentHarness<()>, name: &str) -> crate::middleware::AgentRun { + harness + .invoke(&(), (), RunConfig::new(name), vec![Message::user("go")]) + .await + .expect("run completes") +} + +// ── Catalogue and search ──────────────────────────────────────────────────── + +#[tokio::test] +async fn denied_and_hidden_tools_leave_the_catalogue_and_search() { + let model = Arc::new(ScriptedModel::new(vec![ + calls(vec![("search", "tool_search", json!({"query": "deferred"}))]), + ModelResponse::assistant("done"), + ])); + let policy = ToolRulePolicy::new(rules(json!({ + "rules": [ + { "id": "no-secrets", "effect": "deny", "match": { "name": "secret_*" } }, + { "effect": "hide", "match": { "name": "quiet" } }, + ], + }))); + let mut harness = harness_with(model.clone(), policy); + for tool in [ + RuleTool::new("open"), + RuleTool::new("secret_direct"), + RuleTool::new("quiet"), + RuleTool::deferred("deferred_open"), + RuleTool::deferred("secret_deferred"), + ] { + harness.register_tool(tool); + } + + let run = run(&harness, "catalogue").await; + + let first = &model.requests()[0]; + let names = tool_names(first); + assert!(names.contains(&"open".to_string()), "{names:?}"); + assert!(!names.contains(&"secret_direct".to_string()), "{names:?}"); + assert!(!names.contains(&"quiet".to_string()), "{names:?}"); + let search_schema = first + .tools + .iter() + .find(|tool| tool.name == "tool_search") + .expect("deferred tools advertise the bridge"); + assert!(!search_schema.description.contains("secret_deferred")); + + let answer = tool_text(&run.messages, "search"); + assert!(answer.contains("deferred_open"), "{answer}"); + assert!(!answer.contains("secret_deferred"), "{answer}"); +} + +#[tokio::test] +async fn an_unconfigured_policy_changes_nothing() { + let model = Arc::new(ScriptedModel::new(vec![ModelResponse::assistant("done")])); + let mut harness = harness_with(model.clone(), ToolRulePolicy::default()); + harness.register_tool(RuleTool::new("a")); + harness.register_tool(RuleTool::new("b")); + run(&harness, "plain").await; + assert_eq!(tool_names(&model.requests()[0]), ["a", "b"]); +} + +// ── Calls ─────────────────────────────────────────────────────────────────── + +#[tokio::test] +async fn a_denied_call_is_refused_with_the_rule_and_a_hidden_one_runs() { + let model = Arc::new(ScriptedModel::new(vec![ + calls(vec![ + ("c1", "secret_direct", json!({})), + ("c2", "quiet", json!({})), + ]), + ModelResponse::assistant("done"), + ])); + let policy = ToolRulePolicy::new(rules(json!({ + "name": "config", + "rules": [ + { "id": "no-secrets", "effect": "deny", "match": { "name": "secret_*" }, "reason": "secrets stay put" }, + { "effect": "hide", "match": { "name": "quiet" } }, + ], + }))); + let secret = RuleTool::new("secret_direct"); + let quiet = RuleTool::new("quiet"); + let mut harness = harness_with(model, policy); + harness.register_tool(secret.clone()); + harness.register_tool(quiet.clone()); + + let run = run(&harness, "calls").await; + + assert_eq!(secret.calls(), 0, "a denied tool never runs"); + assert_eq!(quiet.calls(), 1, "a hidden tool stays callable"); + let refusal = tool_text(&run.messages, "c1"); + assert!(refusal.contains("rule 'no-secrets'"), "{refusal}"); + assert!(refusal.contains("secrets stay put"), "{refusal}"); + assert_eq!(run.text().as_deref(), Some("done")); +} + +#[tokio::test] +async fn context_conditions_select_the_rules_that_apply() { + let model = Arc::new(ScriptedModel::new(vec![ + calls(vec![("c1", "shell", json!({}))]), + ModelResponse::assistant("done"), + ])); + let policy = ToolRulePolicy::new(rules(json!({ "rules": [ + { "effect": "deny", "match": { "name": "shell" }, "when": { "channel": "telegram" } }, + ] }))) + .with_context(RuleContext::new().with("channel", "telegram")); + let shell = RuleTool::new("shell"); + let mut harness = harness_with(model, policy); + harness.register_tool(shell.clone()); + run(&harness, "context").await; + assert_eq!(shell.calls(), 0); +} + +#[tokio::test] +async fn an_indirect_target_is_checked_against_the_rules() { + let model = Arc::new(ScriptedModel::new(vec![ + calls(vec![ + ("c1", "execute", json!({"action": "GMAIL_DELETE_EMAIL"})), + ("c2", "execute", json!({"action": "GMAIL_SEND_EMAIL"})), + ]), + ModelResponse::assistant("done"), + ])); + let policy = ToolRulePolicy::new(rules(json!({ "rules": [ + { "id": "no-delete", "effect": "deny", "match": { "name": "*_delete_*" } }, + ] }))); + let execute = RuleTool::dispatcher("execute"); + let mut harness = harness_with(model, policy); + harness.register_tool(execute.clone()); + + let run = run(&harness, "indirect").await; + + assert_eq!(execute.calls(), 1, "only the allowed action ran"); + assert!(tool_text(&run.messages, "c1").contains("rule 'no-delete'")); + assert_eq!(tool_text(&run.messages, "c2"), "execute ran"); +} + +// ── Approval ──────────────────────────────────────────────────────────────── + +#[tokio::test] +async fn require_approval_defers_and_auto_approve_waives_a_declaration() { + let model = Arc::new(ScriptedModel::new(vec![ + calls(vec![ + ("c1", "gated", json!({})), + ("c2", "send", json!({})), + ]), + ModelResponse::assistant("never reached"), + ])); + let policy = ToolRulePolicy::new(rules(json!({ "rules": [ + { "effect": "auto_approve", "match": { "name": "gated" } }, + { "effect": "require_approval", "match": { "name": "send" } }, + ] }))); + let gated = RuleTool::approval_gated("gated"); + let send = RuleTool::new("send"); + let mut harness = harness_with(model, policy); + harness.register_tool(gated.clone()); + harness.register_tool(send.clone()); + + let run = run(&harness, "approval").await; + + assert_eq!(gated.calls(), 1, "auto_approve waived the declared approval"); + assert_eq!(send.calls(), 0, "require_approval deferred the call"); + let deferred = run.deferred.expect("the run waits for approval"); + assert_eq!(deferred.approvals.len(), 1); + assert_eq!(deferred.approvals[0].id, "c2"); +} + +// ── Hosted definitions ────────────────────────────────────────────────────── + +#[tokio::test] +async fn a_hosted_definition_stacks_its_rules_on_the_policy() { + let model = Arc::new(ScriptedModel::new(vec![ + calls(vec![("c1", "web_fetch", json!({}))]), + ModelResponse::assistant("done"), + ])); + let definition = AgentDefinition::new("helper", "Helper", "test helper") + .with_tools(["file_read", "web_fetch", "shell"]) + .with_tool_rules(ToolRules::from_allow_deny(["file_*", "web_*"], Vec::::new())); + let host = HostCapabilities::new( + Arc::new(StaticContextComposer::empty()), + Arc::new(InMemoryDefinitionRegistry::new(vec![definition])), + Arc::new(AllowAllSecurityGate), + Arc::new(FixedModelResolver::new(model.clone())), + ); + // The harness policy denies `web_*`; the definition allows only + // `file_*`/`web_*`; the allowlist names three tools. Only `file_read` + // passes all three. + let policy = ToolRulePolicy::new(ToolRuleSet::single(ToolRules::from_allow_deny( + Vec::::new(), + ["web_*"], + ))); + let mut harness = harness_with(model.clone(), policy); + let web = RuleTool::new("web_fetch"); + for tool in [RuleTool::new("file_read"), web.clone(), RuleTool::new("shell")] { + harness.register_tool(tool); + } + + let run = harness + .invoke_agent( + AgentInvocation::new( + host, + AgentTurnRequest::new("helper", vec![Message::user("go")]), + RunContext::new(RunConfig::new("hosted"), ()), + ), + &(), + ) + .await + .expect("run completes"); + + assert_eq!(tool_names(&model.requests()[0]), ["file_read"]); + assert_eq!(web.calls(), 0); + assert!(tool_text(&run.messages, "c1").contains("not permitted by tool rules")); +} From c81eba341abee1573343364ae3d1a5fbeaeae127 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:41:38 +0300 Subject: [PATCH 13/17] feat(harness): add tool rule tests for agent loop Adds test coverage for the tool rule handling in the agent loop, verifying that rules are applied correctly during tool execution. Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../tinyagents-harness/src/agent_loop/mod.rs | 3 + .../src/tool/rules/mod_tests.rs | 120 ++++++++++++++++++ 2 files changed, 123 insertions(+) create mode 100644 crates/tinyagents-harness/src/tool/rules/mod_tests.rs diff --git a/crates/tinyagents-harness/src/agent_loop/mod.rs b/crates/tinyagents-harness/src/agent_loop/mod.rs index e5ac8df5..662a7cc3 100644 --- a/crates/tinyagents-harness/src/agent_loop/mod.rs +++ b/crates/tinyagents-harness/src/agent_loop/mod.rs @@ -179,6 +179,9 @@ mod terminal_outcome_test; #[path = "mod_tests.rs"] mod test; #[cfg(test)] +#[path = "tool_rules_tests.rs"] +mod tool_rules_test; +#[cfg(test)] #[path = "unknown_tool_tests.rs"] mod unknown_tool_test; diff --git a/crates/tinyagents-harness/src/tool/rules/mod_tests.rs b/crates/tinyagents-harness/src/tool/rules/mod_tests.rs new file mode 100644 index 00000000..10cc4d3e --- /dev/null +++ b/crates/tinyagents-harness/src/tool/rules/mod_tests.rs @@ -0,0 +1,120 @@ +//! Unit tests for the run's tool gate: how the allowlist, the harness policy +//! and a definition's rules combine. + +use super::*; + +use std::collections::HashSet; + +use serde_json::json; +use tinytools::{ApprovalDirective, RuleContext, Surface, ToolRules}; + +fn deny(pattern: &str) -> ToolRules { + ToolRules::from_allow_deny(Vec::::new(), [pattern]) +} + +#[test] +fn a_default_gate_admits_everything() { + let gate = ToolGate::default(); + assert!(gate.allows_name("x")); + assert!(gate.lists("x", None, Surface::Catalog)); + assert!(ToolRulePolicy::default().is_permissive()); +} + +#[test] +fn the_allowlist_still_applies_first() { + let allowed: HashSet = ["a".to_string()].into(); + let gate = ToolGate::new(Some(allowed), &ToolRulePolicy::default(), None); + assert!(gate.lists("a", None, Surface::Search)); + assert!(!gate.lists("b", None, Surface::Search)); + assert!(!gate.allows_name("b")); +} + +#[test] +fn policy_and_definition_rules_both_apply() { + let policy = ToolRulePolicy::new(deny("x_*")); + let definition = deny("y_*"); + let gate = ToolGate::new(None, &policy, Some(&definition)); + assert!(!gate.lists("x_1", None, Surface::Catalog)); + assert!(!gate.lists("y_1", None, Surface::Catalog)); + assert!(gate.lists("z_1", None, Surface::Catalog)); +} + +#[test] +fn a_permissive_definition_layer_is_ignored() { + let policy = ToolRulePolicy::new(deny("x_*")); + let gate = ToolGate::new(None, &policy, Some(&ToolRules::allow_all())); + assert!(!gate.lists("x_1", None, Surface::Catalog)); + let empty = ToolGate::new(None, &ToolRulePolicy::default(), Some(&ToolRules::allow_all())); + assert!(empty.rules.is_none()); +} + +#[test] +fn the_policy_context_reaches_when_conditions() { + let rules: ToolRules = serde_json::from_value(json!({ "rules": [ + { "effect": "deny", "match": { "name": "shell" }, "when": { "channel": "web" } }, + ] })) + .expect("rules"); + let web = ToolRulePolicy::new(rules.clone()).with_context(RuleContext::new().with("channel", "web")); + let cli = ToolRulePolicy::new(rules).with_context(RuleContext::new().with("channel", "cli")); + assert!(!ToolGate::new(None, &web, None).lists("shell", None, Surface::Catalog)); + assert!(ToolGate::new(None, &cli, None).lists("shell", None, Surface::Catalog)); +} + +struct Named(&'static str, tinytools::ToolExposure); + +#[async_trait::async_trait] +impl Tool for Named { + fn name(&self) -> &str { + self.0 + } + fn description(&self) -> &str { + "named" + } + fn parameters_schema(&self) -> serde_json::Value { + json!({"type": "object"}) + } + fn exposure(&self) -> tinytools::ToolExposure { + self.1 + } + async fn execute(&self, _args: serde_json::Value) -> anyhow::Result { + Ok(tinytools::ToolResult::success("ok")) + } +} + +#[test] +fn lists_tool_picks_the_surface_from_exposure() { + let rules: ToolRules = serde_json::from_value(json!({ "rules": [ + { "effect": "deny", "on": ["search"], "match": { "name": "*" } }, + ] })) + .expect("rules"); + let gate = ToolGate::new(None, &ToolRulePolicy::new(rules), None); + assert!(gate.lists_tool(&Named("direct", tinytools::ToolExposure::Direct))); + assert!(!gate.lists_tool(&Named("deferred", tinytools::ToolExposure::Deferred))); +} + +#[test] +fn admit_call_reports_approval_and_refusals() { + let rules: ToolRules = serde_json::from_value(json!({ "rules": [ + { "effect": "require_approval", "match": { "name": "send" } }, + { "id": "nope", "effect": "deny", "match": { "name": "drop" } }, + ] })) + .expect("rules"); + let gate = ToolGate::new(None, &ToolRulePolicy::new(rules), None); + let direct = tinytools::ToolExposure::Direct; + assert_eq!( + gate.admit_call(&Named("send", direct), &json!({})), + CallGate::Admit(ApprovalDirective::Required) + ); + assert_eq!( + gate.admit_call(&Named("read", direct), &json!({})), + CallGate::Admit(ApprovalDirective::Default) + ); + match gate.admit_call(&Named("drop", direct), &json!({})) { + CallGate::Refuse(message) => assert!(message.contains("rule 'nope'"), "{message}"), + other => panic!("expected a refusal, got {other:?}"), + } + assert_eq!( + ToolGate::default().admit_call(&Named("drop", direct), &json!({})), + CallGate::Admit(ApprovalDirective::Default) + ); +} From 9de6f517a83586de52badf56baf2b9911991d991 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:41:49 +0300 Subject: [PATCH 14/17] style: format tool rule code and reorder imports Reformat long expressions and import lists across the tool rule modules to satisfy rustfmt, and sort the rules and progress module declarations and re-exports alphabetically. No behaviour changes. Auto-committed-on: dragonfly Co-authored-by: Medulla --- .../src/agent_loop/tool_rules_tests.rs | 28 +++++++++++++------ .../src/agent_loop/tool_surface.rs | 10 +++++-- .../src/agent_loop/tools.rs | 16 +++++++---- crates/tinyagents-harness/src/tool/mod.rs | 6 ++-- .../src/tool/rules/mod_tests.rs | 9 ++++-- 5 files changed, 49 insertions(+), 20 deletions(-) diff --git a/crates/tinyagents-harness/src/agent_loop/tool_rules_tests.rs b/crates/tinyagents-harness/src/agent_loop/tool_rules_tests.rs index 6a074e39..6c33beac 100644 --- a/crates/tinyagents-harness/src/agent_loop/tool_rules_tests.rs +++ b/crates/tinyagents-harness/src/agent_loop/tool_rules_tests.rs @@ -155,7 +155,11 @@ async fn run(harness: &AgentHarness<()>, name: &str) -> crate::middleware::Agent #[tokio::test] async fn denied_and_hidden_tools_leave_the_catalogue_and_search() { let model = Arc::new(ScriptedModel::new(vec![ - calls(vec![("search", "tool_search", json!({"query": "deferred"}))]), + calls(vec![( + "search", + "tool_search", + json!({"query": "deferred"}), + )]), ModelResponse::assistant("done"), ])); let policy = ToolRulePolicy::new(rules(json!({ @@ -283,10 +287,7 @@ async fn an_indirect_target_is_checked_against_the_rules() { #[tokio::test] async fn require_approval_defers_and_auto_approve_waives_a_declaration() { let model = Arc::new(ScriptedModel::new(vec![ - calls(vec![ - ("c1", "gated", json!({})), - ("c2", "send", json!({})), - ]), + calls(vec![("c1", "gated", json!({})), ("c2", "send", json!({}))]), ModelResponse::assistant("never reached"), ])); let policy = ToolRulePolicy::new(rules(json!({ "rules": [ @@ -301,7 +302,11 @@ async fn require_approval_defers_and_auto_approve_waives_a_declaration() { let run = run(&harness, "approval").await; - assert_eq!(gated.calls(), 1, "auto_approve waived the declared approval"); + assert_eq!( + gated.calls(), + 1, + "auto_approve waived the declared approval" + ); assert_eq!(send.calls(), 0, "require_approval deferred the call"); let deferred = run.deferred.expect("the run waits for approval"); assert_eq!(deferred.approvals.len(), 1); @@ -318,7 +323,10 @@ async fn a_hosted_definition_stacks_its_rules_on_the_policy() { ])); let definition = AgentDefinition::new("helper", "Helper", "test helper") .with_tools(["file_read", "web_fetch", "shell"]) - .with_tool_rules(ToolRules::from_allow_deny(["file_*", "web_*"], Vec::::new())); + .with_tool_rules(ToolRules::from_allow_deny( + ["file_*", "web_*"], + Vec::::new(), + )); let host = HostCapabilities::new( Arc::new(StaticContextComposer::empty()), Arc::new(InMemoryDefinitionRegistry::new(vec![definition])), @@ -334,7 +342,11 @@ async fn a_hosted_definition_stacks_its_rules_on_the_policy() { ))); let mut harness = harness_with(model.clone(), policy); let web = RuleTool::new("web_fetch"); - for tool in [RuleTool::new("file_read"), web.clone(), RuleTool::new("shell")] { + for tool in [ + RuleTool::new("file_read"), + web.clone(), + RuleTool::new("shell"), + ] { harness.register_tool(tool); } diff --git a/crates/tinyagents-harness/src/agent_loop/tool_surface.rs b/crates/tinyagents-harness/src/agent_loop/tool_surface.rs index 9140a88f..9527ef2e 100644 --- a/crates/tinyagents-harness/src/agent_loop/tool_surface.rs +++ b/crates/tinyagents-harness/src/agent_loop/tool_surface.rs @@ -12,8 +12,8 @@ use std::collections::{BTreeMap, BTreeSet, HashSet}; use super::tool_changes; -use crate::tool::ToolGate; use super::*; +use crate::tool::ToolGate; impl AgentHarness { /// The direct tool schemas: the registry's `Direct` schemas filtered by the @@ -49,7 +49,13 @@ impl AgentHarness { .await? .into_iter() .filter(|tool| tool.exposure() == tinytools::ToolExposure::Direct) - .filter(|tool| gate.lists(tool.name(), Some(tool.as_ref()), tinytools::Surface::Catalog)) + .filter(|tool| { + gate.lists( + tool.name(), + Some(tool.as_ref()), + tinytools::Surface::Catalog, + ) + }) .filter(|tool| !existing.contains(tool.name())) .map(|tool| crate::tool::provider_schema(tool.as_ref())) .collect(); diff --git a/crates/tinyagents-harness/src/agent_loop/tools.rs b/crates/tinyagents-harness/src/agent_loop/tools.rs index 33554cf9..69917d35 100644 --- a/crates/tinyagents-harness/src/agent_loop/tools.rs +++ b/crates/tinyagents-harness/src/agent_loop/tools.rs @@ -99,9 +99,8 @@ use super::model_call::ToolCallBase; use super::*; use crate::tool::{ - CallGate, DeferredToolRequests, LedgerFailure, ToolDispatch, ToolEffectSettle, - ToolEffectStart, ToolEffectStatus, ToolGate, ToolProgressGate, ToolProgressLimits, - provider_schema, + CallGate, DeferredToolRequests, LedgerFailure, ToolDispatch, ToolEffectSettle, ToolEffectStart, + ToolEffectStatus, ToolGate, ToolProgressGate, ToolProgressLimits, provider_schema, }; use sha2::{Digest, Sha256}; use tinyinference_llm::message::ContentBlock; @@ -294,13 +293,20 @@ impl AgentHarness { let allowed = self.resolve_tool_allowlist(ctx)?; let binding = crate::runtime::host_invocation_binding::(ctx)?; let definition_rules = binding.as_ref().and_then(|b| b.tool_rules.as_ref()); - Ok(ToolGate::new(allowed, &self.policy.tool_rules, definition_rules)) + Ok(ToolGate::new( + allowed, + &self.policy.tool_rules, + definition_rules, + )) } /// Builds the run's deferred-tool catalogue: every /// [`tinytools::ToolExposure::Deferred`] registration the gate lets the /// model search for, or an empty catalogue when discovery is disabled. - pub(super) fn deferred_catalog(&self, gate: &ToolGate) -> crate::tool::discover::DeferredCatalog { + pub(super) fn deferred_catalog( + &self, + gate: &ToolGate, + ) -> crate::tool::discover::DeferredCatalog { if !self.policy.discovery.enabled { return crate::tool::discover::DeferredCatalog::default(); } diff --git a/crates/tinyagents-harness/src/tool/mod.rs b/crates/tinyagents-harness/src/tool/mod.rs index b4e8def3..6065368c 100644 --- a/crates/tinyagents-harness/src/tool/mod.rs +++ b/crates/tinyagents-harness/src/tool/mod.rs @@ -9,9 +9,9 @@ pub mod discover; pub mod effects; pub mod nested; pub mod packs; -mod rules; mod progress; mod prompt; +mod rules; mod schema; mod schema_compact; mod schema_prepare; @@ -35,10 +35,10 @@ pub use effects::{ ToolEffectStatus, }; pub use nested::NestedToolRunner; -pub(crate) use rules::{CallGate, ToolGate}; -pub use rules::ToolRulePolicy; pub(crate) use progress::{ToolProgressGate, ToolProgressLimits}; pub use prompt::*; +pub use rules::ToolRulePolicy; +pub(crate) use rules::{CallGate, ToolGate}; pub use schema::*; pub use schema_compact::*; pub use schema_prepare::*; diff --git a/crates/tinyagents-harness/src/tool/rules/mod_tests.rs b/crates/tinyagents-harness/src/tool/rules/mod_tests.rs index 10cc4d3e..1576523b 100644 --- a/crates/tinyagents-harness/src/tool/rules/mod_tests.rs +++ b/crates/tinyagents-harness/src/tool/rules/mod_tests.rs @@ -44,7 +44,11 @@ fn a_permissive_definition_layer_is_ignored() { let policy = ToolRulePolicy::new(deny("x_*")); let gate = ToolGate::new(None, &policy, Some(&ToolRules::allow_all())); assert!(!gate.lists("x_1", None, Surface::Catalog)); - let empty = ToolGate::new(None, &ToolRulePolicy::default(), Some(&ToolRules::allow_all())); + let empty = ToolGate::new( + None, + &ToolRulePolicy::default(), + Some(&ToolRules::allow_all()), + ); assert!(empty.rules.is_none()); } @@ -54,7 +58,8 @@ fn the_policy_context_reaches_when_conditions() { { "effect": "deny", "match": { "name": "shell" }, "when": { "channel": "web" } }, ] })) .expect("rules"); - let web = ToolRulePolicy::new(rules.clone()).with_context(RuleContext::new().with("channel", "web")); + let web = + ToolRulePolicy::new(rules.clone()).with_context(RuleContext::new().with("channel", "web")); let cli = ToolRulePolicy::new(rules).with_context(RuleContext::new().with("channel", "cli")); assert!(!ToolGate::new(None, &web, None).lists("shell", None, Surface::Catalog)); assert!(ToolGate::new(None, &cli, None).lists("shell", None, Surface::Catalog)); From a8e9a53b8c485a80dc5cd0a94966f7f04e87e42f Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:42:07 +0300 Subject: [PATCH 15/17] test(tool): import Tool trait in rules tests Add the Tool trait to the tinytools imports in the rules test module so the tests can reference it. Auto-committed-on: dragonfly Co-authored-by: Medulla --- crates/tinyagents-harness/src/tool/rules/mod_tests.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/crates/tinyagents-harness/src/tool/rules/mod_tests.rs b/crates/tinyagents-harness/src/tool/rules/mod_tests.rs index 1576523b..2c4f1e82 100644 --- a/crates/tinyagents-harness/src/tool/rules/mod_tests.rs +++ b/crates/tinyagents-harness/src/tool/rules/mod_tests.rs @@ -6,7 +6,7 @@ use super::*; use std::collections::HashSet; use serde_json::json; -use tinytools::{ApprovalDirective, RuleContext, Surface, ToolRules}; +use tinytools::{ApprovalDirective, RuleContext, Surface, Tool, ToolRules}; fn deny(pattern: &str) -> ToolRules { ToolRules::from_allow_deny(Vec::::new(), [pattern]) From ca248628d790590de6f25b180d57821be08fc4c5 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:42:50 +0300 Subject: [PATCH 16/17] test(tool): assert catalog lists on permissive rule layer The permissive definition layer test now checks that listing a catalog surface returns an empty result instead of asserting the internal rules field is unset, so the test exercises observable behaviour rather than implementation detail. Auto-committed-on: dragonfly Co-authored-by: Medulla --- crates/tinyagents-harness/src/tool/rules/mod_tests.rs | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/crates/tinyagents-harness/src/tool/rules/mod_tests.rs b/crates/tinyagents-harness/src/tool/rules/mod_tests.rs index 2c4f1e82..d1e83c0a 100644 --- a/crates/tinyagents-harness/src/tool/rules/mod_tests.rs +++ b/crates/tinyagents-harness/src/tool/rules/mod_tests.rs @@ -49,7 +49,7 @@ fn a_permissive_definition_layer_is_ignored() { &ToolRulePolicy::default(), Some(&ToolRules::allow_all()), ); - assert!(empty.rules.is_none()); + assert!(empty.lists("anything", None, Surface::Catalog)); } #[test] From 4eb24b3b13611d378bf029addccb97f8191b2a62 Mon Sep 17 00:00:00 2001 From: Steven Enamakel Date: Fri, 9 Oct 2026 09:46:26 +0300 Subject: [PATCH 17/17] docs(harness): link tool rules page from module index Add a reference to the new tool-rules document in the harness README so the allow, deny, hide, and approval pattern documentation is discoverable from the module index. Auto-committed-on: dragonfly Co-authored-by: Medulla --- docs/modules/harness/README.md | 1 + docs/modules/harness/tool-rules.md | 73 ++++++++++++++++++++++++++++++ 2 files changed, 74 insertions(+) create mode 100644 docs/modules/harness/tool-rules.md diff --git a/docs/modules/harness/README.md b/docs/modules/harness/README.md index 7b2eeb48..bdd9d2c8 100644 --- a/docs/modules/harness/README.md +++ b/docs/modules/harness/README.md @@ -242,6 +242,7 @@ Feature details: - [Tool execution context and rich returns (B1/B2)](tool-context.md) - [Nested tool calls (C9)](nested-tool-calls.md) - [Tool exposure, discovery, and schema budgets](tool-discovery.md) +- [Tool rules: allow / deny / hide / approval patterns](tool-rules.md) - [Tool dialects](tool-dialect.md) - [Middleware feature](middleware.md) - [Repeat-progress guard](repeat-progress.md) diff --git a/docs/modules/harness/tool-rules.md b/docs/modules/harness/tool-rules.md new file mode 100644 index 00000000..4ee4ea95 --- /dev/null +++ b/docs/modules/harness/tool-rules.md @@ -0,0 +1,73 @@ +# Tool rules + +Pattern rules decide which tools a run's model may **see** and **call**. The +vocabulary is [`tinytools::ToolRules`](../../../vendor/tinytools/docs/specs/tool-rules.md); +this page is how the agent loop applies it. + +## Where rules come from + +A run stacks up to three restrictions. Every one must admit a tool: + +1. The hosted definition's exact `tools` allowlist (`AgentDefinition::tools`), + with its fail-closed default (I-9). Unchanged. +2. `RunPolicy::tool_rules`: a `ToolRulePolicy { rules, context }`. Use this + for harness-wide rules, and for the `RuleContext` (`channel`, `agent`, + `origin`, …) that `when` conditions match. +3. `AgentDefinition::tool_rules` on a hosted run, added as another layer. + +A layer can only narrow: two allowlists intersect, and a `deny` anywhere wins. + +## One gate, every surface + +`ToolGate` (crate-private, `tool/rules/`) answers every question the loop asks +about a tool: + +| Site | Surface | +|---|---| +| Direct schemas on the request, and mid-run toolset changes | `catalog` | +| The deferred catalogue, the `tool_search` manifest and answers, replayed promotions | `search` | +| Admission of a model call; the unknown-tool "closest available" list | `call` / listing | +| Admission of a nested call | `call` | + +A tool a rule removes from the catalogue therefore cannot be found through +`tool_search`, or called under a name the model guessed. + +## Calls + +- The rules run **before** `before_tool` middleware. An approval middleware + never asks a human about a call the rules refuse. +- They see the raw provider arguments, which is what the host security gate + also sees. +- A refused call is answered with a tool error naming the rule, its layer and + its reason. It frees its tool-call budget slot, as a middleware refusal + does. +- When a tool reports `Tool::indirect_target(args)` (a connector's execute + tool, a skill runner), the target is checked as well. +- `require_approval` defers the call exactly as a declared + `approval_required` would. `auto_approve` waives that declaration. In a + nested call, both map to the existing nested approval refusal. + +## Example + +```rust +use tinyagents_harness::runtime::RunPolicy; +use tinyagents_harness::tool::ToolRulePolicy; +use tinytools::{RuleContext, ToolRules}; + +let rules: ToolRules = serde_json::from_value(serde_json::json!({ + "rules": [ + { "id": "no-mcp", "effect": "deny", "match": { "name": "mcp_*" }, + "except": { "family": "github" } }, + { "effect": "require_approval", "match": { "side_effects": ["payment"] } }, + { "effect": "deny", "match": { "name": "shell" }, "when": { "channel": "telegram" } }, + ], +}))?; +let policy = RunPolicy { + tool_rules: ToolRulePolicy::new(rules) + .with_context(RuleContext::new().with("channel", "telegram")), + ..RunPolicy::default() +}; +``` + +The default `ToolRulePolicy` holds no rules. A harness that never sets it +pays nothing per tool and behaves exactly as before.