diff --git a/.verify/BoundaryML_baml.json b/.verify/BoundaryML_baml.json
new file mode 100644
index 0000000..a0240db
--- /dev/null
+++ b/.verify/BoundaryML_baml.json
@@ -0,0 +1 @@
+{"id":701494311,"node_id":"R_kgDOKc_0Jw","name":"baml","full_name":"BoundaryML/baml","private":false,"owner":{"login":"BoundaryML","id":124114301,"node_id":"O_kgDOB2XVfQ","avatar_url":"https://avatars.githubusercontent.com/u/124114301?v=4","gravatar_id":"","url":"https://api.github.com/users/BoundaryML","html_url":"https://github.com/BoundaryML","followers_url":"https://api.github.com/users/BoundaryML/followers","following_url":"https://api.github.com/users/BoundaryML/following{/other_user}","gists_url":"https://api.github.com/users/BoundaryML/gists{/gist_id}","starred_url":"https://api.github.com/users/BoundaryML/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/BoundaryML/subscriptions","organizations_url":"https://api.github.com/users/BoundaryML/orgs","repos_url":"https://api.github.com/users/BoundaryML/repos","events_url":"https://api.github.com/users/BoundaryML/events{/privacy}","received_events_url":"https://api.github.com/users/BoundaryML/received_events","type":"Organization","user_view_type":"public","site_admin":false},"html_url":"https://github.com/BoundaryML/baml","description":"The programming language for agents","fork":false,"url":"https://api.github.com/repos/BoundaryML/baml","forks_url":"https://api.github.com/repos/BoundaryML/baml/forks","keys_url":"https://api.github.com/repos/BoundaryML/baml/keys{/key_id}","collaborators_url":"https://api.github.com/repos/BoundaryML/baml/collaborators{/collaborator}","teams_url":"https://api.github.com/repos/BoundaryML/baml/teams","hooks_url":"https://api.github.com/repos/BoundaryML/baml/hooks","issue_events_url":"https://api.github.com/repos/BoundaryML/baml/issues/events{/number}","events_url":"https://api.github.com/repos/BoundaryML/baml/events","assignees_url":"https://api.github.com/repos/BoundaryML/baml/assignees{/user}","branches_url":"https://api.github.com/repos/BoundaryML/baml/branches{/branch}","tags_url":"https://api.github.com/repos/BoundaryML/baml/tags","blobs_url":"https://api.github.com/repos/BoundaryML/baml/git/blobs{/sha}","git_tags_url":"https://api.github.com/repos/BoundaryML/baml/git/tags{/sha}","git_refs_url":"https://api.github.com/repos/BoundaryML/baml/git/refs{/sha}","trees_url":"https://api.github.com/repos/BoundaryML/baml/git/trees{/sha}","statuses_url":"https://api.github.com/repos/BoundaryML/baml/statuses/{sha}","languages_url":"https://api.github.com/repos/BoundaryML/baml/languages","stargazers_url":"https://api.github.com/repos/BoundaryML/baml/stargazers","contributors_url":"https://api.github.com/repos/BoundaryML/baml/contributors","subscribers_url":"https://api.github.com/repos/BoundaryML/baml/subscribers","subscription_url":"https://api.github.com/repos/BoundaryML/baml/subscription","commits_url":"https://api.github.com/repos/BoundaryML/baml/commits{/sha}","git_commits_url":"https://api.github.com/repos/BoundaryML/baml/git/commits{/sha}","comments_url":"https://api.github.com/repos/BoundaryML/baml/comments{/number}","issue_comment_url":"https://api.github.com/repos/BoundaryML/baml/issues/comments{/number}","contents_url":"https://api.github.com/repos/BoundaryML/baml/contents/{+path}","compare_url":"https://api.github.com/repos/BoundaryML/baml/compare/{base}...{head}","merges_url":"https://api.github.com/repos/BoundaryML/baml/merges","archive_url":"https://api.github.com/repos/BoundaryML/baml/{archive_format}{/ref}","downloads_url":"https://api.github.com/repos/BoundaryML/baml/downloads","issues_url":"https://api.github.com/repos/BoundaryML/baml/issues{/number}","pulls_url":"https://api.github.com/repos/BoundaryML/baml/pulls{/number}","milestones_url":"https://api.github.com/repos/BoundaryML/baml/milestones{/number}","notifications_url":"https://api.github.com/repos/BoundaryML/baml/notifications{?since,all,participating}","labels_url":"https://api.github.com/repos/BoundaryML/baml/labels{/name}","releases_url":"https://api.github.com/repos/BoundaryML/baml/releases{/id}","deployments_url":"https://api.github.com/repos/BoundaryML/baml/deployments","created_at":"2023-10-06T18:57:41Z","updated_at":"2026-08-12T14:13:46Z","pushed_at":"2026-08-12T11:15:37Z","git_url":"git://github.com/BoundaryML/baml.git","ssh_url":"git@github.com:BoundaryML/baml.git","clone_url":"https://github.com/BoundaryML/baml.git","svn_url":"https://github.com/BoundaryML/baml","homepage":"https://boundaryml.com/explore","size":655500,"stargazers_count":8927,"watchers_count":8927,"language":"Rust","has_issues":true,"has_projects":false,"has_downloads":false,"has_wiki":false,"has_pages":true,"has_discussions":true,"forks_count":472,"mirror_url":null,"archived":false,"disabled":false,"open_issues_count":309,"license":{"key":"apache-2.0","name":"Apache License 2.0","spdx_id":"Apache-2.0","url":"https://api.github.com/licenses/apache-2.0","node_id":"MDc6TGljZW5zZTI="},"allow_forking":true,"is_template":false,"web_commit_signoff_required":true,"has_pull_requests":true,"pull_request_creation_policy":"all","topics":["boundaryml","guardrails","llm","programming-language","structured-data"],"visibility":"public","forks":472,"open_issues":309,"watchers":8927,"default_branch":"canary","temp_clone_token":null,"custom_properties":{},"organization":{"login":"BoundaryML","id":124114301,"node_id":"O_kgDOB2XVfQ","avatar_url":"https://avatars.githubusercontent.com/u/124114301?v=4","gravatar_id":"","url":"https://api.github.com/users/BoundaryML","html_url":"https://github.com/BoundaryML","followers_url":"https://api.github.com/users/BoundaryML/followers","following_url":"https://api.github.com/users/BoundaryML/following{/other_user}","gists_url":"https://api.github.com/users/BoundaryML/gists{/gist_id}","starred_url":"https://api.github.com/users/BoundaryML/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/BoundaryML/subscriptions","organizations_url":"https://api.github.com/users/BoundaryML/orgs","repos_url":"https://api.github.com/users/BoundaryML/repos","events_url":"https://api.github.com/users/BoundaryML/events{/privacy}","received_events_url":"https://api.github.com/users/BoundaryML/received_events","type":"Organization","user_view_type":"public","site_admin":false},"network_count":472,"subscribers_count":33}
\ No newline at end of file
diff --git a/.verify/BoundaryML_baml_README.md b/.verify/BoundaryML_baml_README.md
new file mode 100644
index 0000000..1becba2
--- /dev/null
+++ b/.verify/BoundaryML_baml_README.md
@@ -0,0 +1 @@
+404: Not Found
\ No newline at end of file
diff --git a/.verify/BoundaryML_baml_README_alt.md b/.verify/BoundaryML_baml_README_alt.md
new file mode 100644
index 0000000..2c17988
--- /dev/null
+++ b/.verify/BoundaryML_baml_README_alt.md
@@ -0,0 +1,52 @@
+
+
+
+
+
+
+
+
+# BAML: Basically A Made-up Language
+
+BAML is the programming language for agents.
+
+[](https://pypi.org/project/baml-py/)
+
+[Homepage](https://www.boundaryml.com/) | [Explore BAML](https://www.boundaryml.com/explore) | [Discord](https://www.boundaryml.com/discord)
+
+
+
+BAML looks like TypeScript, but every feature is built so agents make fewer mistakes:
+
+- It has a type system like Rust, but compiles even faster than Go.
+- Types persist at runtime. There is no `any` nor casting dangerously to any type.
+- Errors are typed and statically analyzed.
+- The filesystem describes the modules/namespaces.
+- Has green threads, and colorless concurrency like Go
+- Built-in tests / eval framework
+- Built-in stdlib for agents
+- Every baml tool is natively designed for agents, with no garbage outputs, etc.
+- Can be run standalone or adopt incrementally (you can call a BAML function from TS, Py, Go, C#, Java, etc).
+
+[Explore the website and examples](https://www.boundaryml.com/explore).
+
+## Try it out
+
+```bash
+brew install baml
+baml agent install
+baml init
+baml ide install --code
+```
+
+Or read the [quickstart](https://boundaryml.com/quickstart).
+
+## Contributing
+
+See our [guide on getting started](/CONTRIBUTING.md).
+
+---
+
+Made with ❤️ by Boundary. HQ in Seattle, WA.
+
+We're hiring software engineers who love Rust. [Email us](mailto:founders@boundaryml.com) or reach out on [Discord](https://www.boundaryml.com/discord).
diff --git a/.verify/NVIDIA-NeMo_Switchyard.json b/.verify/NVIDIA-NeMo_Switchyard.json
new file mode 100644
index 0000000..e90b296
--- /dev/null
+++ b/.verify/NVIDIA-NeMo_Switchyard.json
@@ -0,0 +1 @@
+{"id":1243889840,"node_id":"R_kgDOSiRAsA","name":"Switchyard","full_name":"NVIDIA-NeMo/Switchyard","private":false,"owner":{"login":"NVIDIA-NeMo","id":213689629,"node_id":"O_kgDODLylHQ","avatar_url":"https://avatars.githubusercontent.com/u/213689629?v=4","gravatar_id":"","url":"https://api.github.com/users/NVIDIA-NeMo","html_url":"https://github.com/NVIDIA-NeMo","followers_url":"https://api.github.com/users/NVIDIA-NeMo/followers","following_url":"https://api.github.com/users/NVIDIA-NeMo/following{/other_user}","gists_url":"https://api.github.com/users/NVIDIA-NeMo/gists{/gist_id}","starred_url":"https://api.github.com/users/NVIDIA-NeMo/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/NVIDIA-NeMo/subscriptions","organizations_url":"https://api.github.com/users/NVIDIA-NeMo/orgs","repos_url":"https://api.github.com/users/NVIDIA-NeMo/repos","events_url":"https://api.github.com/users/NVIDIA-NeMo/events{/privacy}","received_events_url":"https://api.github.com/users/NVIDIA-NeMo/received_events","type":"Organization","user_view_type":"public","site_admin":false},"html_url":"https://github.com/NVIDIA-NeMo/Switchyard","description":null,"fork":false,"url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard","forks_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/forks","keys_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/keys{/key_id}","collaborators_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/collaborators{/collaborator}","teams_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/teams","hooks_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/hooks","issue_events_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/issues/events{/number}","events_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/events","assignees_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/assignees{/user}","branches_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/branches{/branch}","tags_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/tags","blobs_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/git/blobs{/sha}","git_tags_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/git/tags{/sha}","git_refs_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/git/refs{/sha}","trees_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/git/trees{/sha}","statuses_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/statuses/{sha}","languages_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/languages","stargazers_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/stargazers","contributors_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/contributors","subscribers_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/subscribers","subscription_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/subscription","commits_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/commits{/sha}","git_commits_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/git/commits{/sha}","comments_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/comments{/number}","issue_comment_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/issues/comments{/number}","contents_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/contents/{+path}","compare_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/compare/{base}...{head}","merges_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/merges","archive_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/{archive_format}{/ref}","downloads_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/downloads","issues_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/issues{/number}","pulls_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/pulls{/number}","milestones_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/milestones{/number}","notifications_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/notifications{?since,all,participating}","labels_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/labels{/name}","releases_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/releases{/id}","deployments_url":"https://api.github.com/repos/NVIDIA-NeMo/Switchyard/deployments","created_at":"2026-05-19T19:06:33Z","updated_at":"2026-08-12T14:34:33Z","pushed_at":"2026-08-12T13:32:56Z","git_url":"git://github.com/NVIDIA-NeMo/Switchyard.git","ssh_url":"git@github.com:NVIDIA-NeMo/Switchyard.git","clone_url":"https://github.com/NVIDIA-NeMo/Switchyard.git","svn_url":"https://github.com/NVIDIA-NeMo/Switchyard","homepage":null,"size":18862,"stargazers_count":656,"watchers_count":656,"language":"Rust","has_issues":true,"has_projects":true,"has_downloads":false,"has_wiki":true,"has_pages":true,"has_discussions":true,"forks_count":75,"mirror_url":null,"archived":false,"disabled":false,"open_issues_count":82,"license":{"key":"apache-2.0","name":"Apache License 2.0","spdx_id":"Apache-2.0","url":"https://api.github.com/licenses/apache-2.0","node_id":"MDc6TGljZW5zZTI="},"allow_forking":true,"is_template":false,"web_commit_signoff_required":false,"has_pull_requests":true,"pull_request_creation_policy":"all","topics":[],"visibility":"public","forks":75,"open_issues":82,"watchers":656,"default_branch":"main","temp_clone_token":null,"custom_properties":{"approval-to-make-public":"jocohen@nvidia.com|38d5c0a6","nspect-id":"NSPECT-O5T7-NSWU"},"organization":{"login":"NVIDIA-NeMo","id":213689629,"node_id":"O_kgDODLylHQ","avatar_url":"https://avatars.githubusercontent.com/u/213689629?v=4","gravatar_id":"","url":"https://api.github.com/users/NVIDIA-NeMo","html_url":"https://github.com/NVIDIA-NeMo","followers_url":"https://api.github.com/users/NVIDIA-NeMo/followers","following_url":"https://api.github.com/users/NVIDIA-NeMo/following{/other_user}","gists_url":"https://api.github.com/users/NVIDIA-NeMo/gists{/gist_id}","starred_url":"https://api.github.com/users/NVIDIA-NeMo/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/NVIDIA-NeMo/subscriptions","organizations_url":"https://api.github.com/users/NVIDIA-NeMo/orgs","repos_url":"https://api.github.com/users/NVIDIA-NeMo/repos","events_url":"https://api.github.com/users/NVIDIA-NeMo/events{/privacy}","received_events_url":"https://api.github.com/users/NVIDIA-NeMo/received_events","type":"Organization","user_view_type":"public","site_admin":false},"network_count":75,"subscribers_count":1}
\ No newline at end of file
diff --git a/.verify/NVIDIA-NeMo_Switchyard_README.md b/.verify/NVIDIA-NeMo_Switchyard_README.md
new file mode 100644
index 0000000..cdc2b51
--- /dev/null
+++ b/.verify/NVIDIA-NeMo_Switchyard_README.md
@@ -0,0 +1,165 @@
+
+
+
+
+# Switchyard
+
+Switchyard is a Rust proxy and library for LLM traffic. It routes requests
+across providers, translates between OpenAI and Anthropic APIs, records
+operational metrics, and provides typed, composable routing algorithms.
+
+**Why Switchyard?** Point a coding agent such as Claude Code or Codex at an
+open-source model. Switchyard translates between the OpenAI Chat, Anthropic
+Messages, and OpenAI Responses formats, so the agent keeps speaking its native
+API while the request is served by vLLM, NVIDIA NIM, Ollama, or any
+OpenAI-compatible endpoint. The same proxy can spread traffic across several
+models for A/B benchmarking, apply signal-driven stage routing, or run a custom
+algorithm you write yourself.
+
+## Features
+
+- **Protocol Translation**: convert between OpenAI Chat, Anthropic Messages, and OpenAI Responses formats
+- **Multi-Backend Routing**: random routing, LLM-as-classifier routing, signal-driven stage-router, or your own algorithm
+- **Operational Metrics**: Prometheus metrics cover requests, errors, latency, tokens, and routing overhead
+
+## Maturity
+
+Switchyard is pre-alpha software that is evolving rapidly. The API and algorithms are expected to change significantly before we reach v1.0.
+
+> [!WARNING]
+> Experimental software. Not for production use.
+
+## Quick Start
+
+Choose the launcher path to run Claude Code, Codex CLI, or OpenClaw through
+Switchyard. Choose the server path to run Switchyard as a standalone proxy.
+Choose the library path to embed routing in your own Rust application.
+
+### Launcher Path
+
+Install [`uv`](https://docs.astral.sh/uv/getting-started/installation/) if it is
+not already available, then install the published Switchyard tool:
+
+```bash
+curl -LsSf https://astral.sh/uv/install.sh | sh
+source "$HOME/.local/bin/env"
+uv tool install --python 3.12 "nemo-switchyard[cli]"
+```
+
+The coding agent you launch must also be installed and on your `PATH`. This does
+not install the standalone `switchyard-server` binary; use the Server Path for
+that.
+
+Set an OpenRouter key and launch against the packaged deployment:
+
+```bash
+export OPENROUTER_API_KEY="your-openrouter-key" # pragma: allowlist secret
+switchyard launch claude --model switchyard
+switchyard launch codex --model switchyard
+switchyard launch openclaw --model switchyard
+```
+
+To use your own native TOML deployment, pass its route ID and configuration:
+
+```bash
+switchyard launch claude --model my-route --config routes.toml
+```
+
+### Server Path
+
+Use this path to install and run the standalone Rust proxy. Install
+[Rust with Cargo](https://rust-lang.org/tools/install/), then install the
+published binary:
+
+```bash
+cargo install --locked switchyard-server
+switchyard-server --help
+```
+
+Cargo builds the release binary and installs it into `~/.cargo/bin` by default.
+
+Create `routes.toml` using the
+[Getting Started guide](docs/getting_started.md#server-path), then validate it
+and start the server:
+
+```bash
+export OPENROUTER_API_KEY="your-openrouter-key" # pragma: allowlist secret
+switchyard-server --config routes.toml --dry-run
+switchyard-server --config routes.toml --host 127.0.0.1 --port 4000
+```
+
+Verify the proxy in another terminal:
+
+```bash
+curl http://localhost:4000/health
+```
+
+For a complete configuration and a test request, follow
+[Getting Started](docs/getting_started.md).
+
+### Library Path
+
+`switchyard-libsy` embeds the routing algorithms in your own Rust application.
+It never calls a model itself: an algorithm decides which target to use and
+hands every model call back to you, so it drops into an existing proxy, gateway,
+or agent runtime without owning an HTTP stack. Pair it with
+`switchyard-llm-client` when you want the calls made for you.
+
+```toml
+[dependencies]
+switchyard-libsy = { git = "https://github.com/NVIDIA-NeMo/Switchyard.git" }
+switchyard-protocol = { git = "https://github.com/NVIDIA-NeMo/Switchyard.git" }
+```
+
+See [Getting Started](docs/getting_started.md#library-path) for setup and the
+algorithm list, or the [`switchyard-libsy`](crates/libsy/README.md) crate docs.
+
+## Routing Strategies
+
+| Strategy | Use it when | Route `type` |
+|---|---|---|
+| [LLM Classifier](docs/routing_algorithms/llm_classifier_routing.md) | Request content should decide whether a turn needs the weak or strong tier. | `llm_classifier` |
+| [Stage Router](docs/routing_algorithms/stage_router_routing.md) | Signals already in the conversation, such as tool results and errors, should route most turns without an extra model call. | `stage_router` |
+| [Escalation Router](docs/routing_algorithms/escalation_router_routing.md) | Every turn runs on the weak tier first, and a judge reads that answer to decide whether to send the same request to the strong tier. | `llm_classifier` with `mode = "escalation"` |
+| [Random](docs/routing_algorithms/random_routing.md) | You need a fixed traffic split for A/B tests, baselines, or cost experiments. | `random` |
+
+A `passthrough` route registers one target under one model ID with no routing
+decision. See the [Routing Overview](docs/routing_algorithms/overview.md) for
+the common route shape and self-hosted targets.
+
+## Architecture
+
+```mermaid
+flowchart LR
+ clients["Clients"]
+ switchyard["Switchyard routing · translation · fallback"]
+ backends["Model backends"]
+
+ clients -->|"OpenAI / Anthropic API"| switchyard
+ switchyard -->|"provider-native format"| backends
+```
+
+Clients keep their native OpenAI or Anthropic API format. Switchyard picks a
+configured backend, forwards the request in that backend's own format, and
+translates the response back into the shape the client expects. The server
+accepts OpenAI Chat Completions, OpenAI Responses, and Anthropic Messages. Each
+configured LLM client selects one upstream format.
+
+## Documentation
+
+- **[Getting Started](docs/getting_started.md)**: complete launcher and standalone server walkthroughs
+- **[Core Concepts](docs/core_concepts.md)**: LLM clients, targets, routes, model IDs, and routing algorithms
+- **[Routing Overview](docs/routing_algorithms/overview.md)**: choose and configure a routing algorithm
+- **[`switchyard-server`](crates/switchyard-server/README.md)**: server configuration, routing algorithms, and metrics
+- **[`switchyard-libsy`](crates/libsy/README.md)**: embed routing algorithms in a Rust application
+- **[`switchyard-protocol`](crates/protocol/README.md)**: provider-neutral request, response, and streaming types
+- **[`switchyard-translation`](crates/switchyard-translation/README.md)**: request, response, and stream translation
+
+## Community
+
+- **Issues**: [GitHub Issues](https://github.com/NVIDIA-NeMo/Switchyard/issues)
+- **Code of Conduct**: [Code of Conduct](CODE_OF_CONDUCT.md)
+
+## License
+
+[Apache 2.0 License](LICENSE). Copyright NVIDIA Corporation.
diff --git a/.verify/NVIDIA-NeMo_Switchyard_README_alt.md b/.verify/NVIDIA-NeMo_Switchyard_README_alt.md
new file mode 100644
index 0000000..cdc2b51
--- /dev/null
+++ b/.verify/NVIDIA-NeMo_Switchyard_README_alt.md
@@ -0,0 +1,165 @@
+
+
+
+
+# Switchyard
+
+Switchyard is a Rust proxy and library for LLM traffic. It routes requests
+across providers, translates between OpenAI and Anthropic APIs, records
+operational metrics, and provides typed, composable routing algorithms.
+
+**Why Switchyard?** Point a coding agent such as Claude Code or Codex at an
+open-source model. Switchyard translates between the OpenAI Chat, Anthropic
+Messages, and OpenAI Responses formats, so the agent keeps speaking its native
+API while the request is served by vLLM, NVIDIA NIM, Ollama, or any
+OpenAI-compatible endpoint. The same proxy can spread traffic across several
+models for A/B benchmarking, apply signal-driven stage routing, or run a custom
+algorithm you write yourself.
+
+## Features
+
+- **Protocol Translation**: convert between OpenAI Chat, Anthropic Messages, and OpenAI Responses formats
+- **Multi-Backend Routing**: random routing, LLM-as-classifier routing, signal-driven stage-router, or your own algorithm
+- **Operational Metrics**: Prometheus metrics cover requests, errors, latency, tokens, and routing overhead
+
+## Maturity
+
+Switchyard is pre-alpha software that is evolving rapidly. The API and algorithms are expected to change significantly before we reach v1.0.
+
+> [!WARNING]
+> Experimental software. Not for production use.
+
+## Quick Start
+
+Choose the launcher path to run Claude Code, Codex CLI, or OpenClaw through
+Switchyard. Choose the server path to run Switchyard as a standalone proxy.
+Choose the library path to embed routing in your own Rust application.
+
+### Launcher Path
+
+Install [`uv`](https://docs.astral.sh/uv/getting-started/installation/) if it is
+not already available, then install the published Switchyard tool:
+
+```bash
+curl -LsSf https://astral.sh/uv/install.sh | sh
+source "$HOME/.local/bin/env"
+uv tool install --python 3.12 "nemo-switchyard[cli]"
+```
+
+The coding agent you launch must also be installed and on your `PATH`. This does
+not install the standalone `switchyard-server` binary; use the Server Path for
+that.
+
+Set an OpenRouter key and launch against the packaged deployment:
+
+```bash
+export OPENROUTER_API_KEY="your-openrouter-key" # pragma: allowlist secret
+switchyard launch claude --model switchyard
+switchyard launch codex --model switchyard
+switchyard launch openclaw --model switchyard
+```
+
+To use your own native TOML deployment, pass its route ID and configuration:
+
+```bash
+switchyard launch claude --model my-route --config routes.toml
+```
+
+### Server Path
+
+Use this path to install and run the standalone Rust proxy. Install
+[Rust with Cargo](https://rust-lang.org/tools/install/), then install the
+published binary:
+
+```bash
+cargo install --locked switchyard-server
+switchyard-server --help
+```
+
+Cargo builds the release binary and installs it into `~/.cargo/bin` by default.
+
+Create `routes.toml` using the
+[Getting Started guide](docs/getting_started.md#server-path), then validate it
+and start the server:
+
+```bash
+export OPENROUTER_API_KEY="your-openrouter-key" # pragma: allowlist secret
+switchyard-server --config routes.toml --dry-run
+switchyard-server --config routes.toml --host 127.0.0.1 --port 4000
+```
+
+Verify the proxy in another terminal:
+
+```bash
+curl http://localhost:4000/health
+```
+
+For a complete configuration and a test request, follow
+[Getting Started](docs/getting_started.md).
+
+### Library Path
+
+`switchyard-libsy` embeds the routing algorithms in your own Rust application.
+It never calls a model itself: an algorithm decides which target to use and
+hands every model call back to you, so it drops into an existing proxy, gateway,
+or agent runtime without owning an HTTP stack. Pair it with
+`switchyard-llm-client` when you want the calls made for you.
+
+```toml
+[dependencies]
+switchyard-libsy = { git = "https://github.com/NVIDIA-NeMo/Switchyard.git" }
+switchyard-protocol = { git = "https://github.com/NVIDIA-NeMo/Switchyard.git" }
+```
+
+See [Getting Started](docs/getting_started.md#library-path) for setup and the
+algorithm list, or the [`switchyard-libsy`](crates/libsy/README.md) crate docs.
+
+## Routing Strategies
+
+| Strategy | Use it when | Route `type` |
+|---|---|---|
+| [LLM Classifier](docs/routing_algorithms/llm_classifier_routing.md) | Request content should decide whether a turn needs the weak or strong tier. | `llm_classifier` |
+| [Stage Router](docs/routing_algorithms/stage_router_routing.md) | Signals already in the conversation, such as tool results and errors, should route most turns without an extra model call. | `stage_router` |
+| [Escalation Router](docs/routing_algorithms/escalation_router_routing.md) | Every turn runs on the weak tier first, and a judge reads that answer to decide whether to send the same request to the strong tier. | `llm_classifier` with `mode = "escalation"` |
+| [Random](docs/routing_algorithms/random_routing.md) | You need a fixed traffic split for A/B tests, baselines, or cost experiments. | `random` |
+
+A `passthrough` route registers one target under one model ID with no routing
+decision. See the [Routing Overview](docs/routing_algorithms/overview.md) for
+the common route shape and self-hosted targets.
+
+## Architecture
+
+```mermaid
+flowchart LR
+ clients["Clients"]
+ switchyard["Switchyard routing · translation · fallback"]
+ backends["Model backends"]
+
+ clients -->|"OpenAI / Anthropic API"| switchyard
+ switchyard -->|"provider-native format"| backends
+```
+
+Clients keep their native OpenAI or Anthropic API format. Switchyard picks a
+configured backend, forwards the request in that backend's own format, and
+translates the response back into the shape the client expects. The server
+accepts OpenAI Chat Completions, OpenAI Responses, and Anthropic Messages. Each
+configured LLM client selects one upstream format.
+
+## Documentation
+
+- **[Getting Started](docs/getting_started.md)**: complete launcher and standalone server walkthroughs
+- **[Core Concepts](docs/core_concepts.md)**: LLM clients, targets, routes, model IDs, and routing algorithms
+- **[Routing Overview](docs/routing_algorithms/overview.md)**: choose and configure a routing algorithm
+- **[`switchyard-server`](crates/switchyard-server/README.md)**: server configuration, routing algorithms, and metrics
+- **[`switchyard-libsy`](crates/libsy/README.md)**: embed routing algorithms in a Rust application
+- **[`switchyard-protocol`](crates/protocol/README.md)**: provider-neutral request, response, and streaming types
+- **[`switchyard-translation`](crates/switchyard-translation/README.md)**: request, response, and stream translation
+
+## Community
+
+- **Issues**: [GitHub Issues](https://github.com/NVIDIA-NeMo/Switchyard/issues)
+- **Code of Conduct**: [Code of Conduct](CODE_OF_CONDUCT.md)
+
+## License
+
+[Apache 2.0 License](LICENSE). Copyright NVIDIA Corporation.
diff --git a/.verify/cactus-compute_needle.json b/.verify/cactus-compute_needle.json
new file mode 100644
index 0000000..1a51082
--- /dev/null
+++ b/.verify/cactus-compute_needle.json
@@ -0,0 +1 @@
+{"id":1165361576,"node_id":"R_kgDORXYBqA","name":"needle","full_name":"cactus-compute/needle","private":false,"owner":{"login":"cactus-compute","id":196640840,"node_id":"O_kgDOC7iASA","avatar_url":"https://avatars.githubusercontent.com/u/196640840?v=4","gravatar_id":"","url":"https://api.github.com/users/cactus-compute","html_url":"https://github.com/cactus-compute","followers_url":"https://api.github.com/users/cactus-compute/followers","following_url":"https://api.github.com/users/cactus-compute/following{/other_user}","gists_url":"https://api.github.com/users/cactus-compute/gists{/gist_id}","starred_url":"https://api.github.com/users/cactus-compute/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/cactus-compute/subscriptions","organizations_url":"https://api.github.com/users/cactus-compute/orgs","repos_url":"https://api.github.com/users/cactus-compute/repos","events_url":"https://api.github.com/users/cactus-compute/events{/privacy}","received_events_url":"https://api.github.com/users/cactus-compute/received_events","type":"Organization","user_view_type":"public","site_admin":false},"html_url":"https://github.com/cactus-compute/needle","description":"14MB foundation model for tiny devices; phones, wearables, smart home, and robots.","fork":false,"url":"https://api.github.com/repos/cactus-compute/needle","forks_url":"https://api.github.com/repos/cactus-compute/needle/forks","keys_url":"https://api.github.com/repos/cactus-compute/needle/keys{/key_id}","collaborators_url":"https://api.github.com/repos/cactus-compute/needle/collaborators{/collaborator}","teams_url":"https://api.github.com/repos/cactus-compute/needle/teams","hooks_url":"https://api.github.com/repos/cactus-compute/needle/hooks","issue_events_url":"https://api.github.com/repos/cactus-compute/needle/issues/events{/number}","events_url":"https://api.github.com/repos/cactus-compute/needle/events","assignees_url":"https://api.github.com/repos/cactus-compute/needle/assignees{/user}","branches_url":"https://api.github.com/repos/cactus-compute/needle/branches{/branch}","tags_url":"https://api.github.com/repos/cactus-compute/needle/tags","blobs_url":"https://api.github.com/repos/cactus-compute/needle/git/blobs{/sha}","git_tags_url":"https://api.github.com/repos/cactus-compute/needle/git/tags{/sha}","git_refs_url":"https://api.github.com/repos/cactus-compute/needle/git/refs{/sha}","trees_url":"https://api.github.com/repos/cactus-compute/needle/git/trees{/sha}","statuses_url":"https://api.github.com/repos/cactus-compute/needle/statuses/{sha}","languages_url":"https://api.github.com/repos/cactus-compute/needle/languages","stargazers_url":"https://api.github.com/repos/cactus-compute/needle/stargazers","contributors_url":"https://api.github.com/repos/cactus-compute/needle/contributors","subscribers_url":"https://api.github.com/repos/cactus-compute/needle/subscribers","subscription_url":"https://api.github.com/repos/cactus-compute/needle/subscription","commits_url":"https://api.github.com/repos/cactus-compute/needle/commits{/sha}","git_commits_url":"https://api.github.com/repos/cactus-compute/needle/git/commits{/sha}","comments_url":"https://api.github.com/repos/cactus-compute/needle/comments{/number}","issue_comment_url":"https://api.github.com/repos/cactus-compute/needle/issues/comments{/number}","contents_url":"https://api.github.com/repos/cactus-compute/needle/contents/{+path}","compare_url":"https://api.github.com/repos/cactus-compute/needle/compare/{base}...{head}","merges_url":"https://api.github.com/repos/cactus-compute/needle/merges","archive_url":"https://api.github.com/repos/cactus-compute/needle/{archive_format}{/ref}","downloads_url":"https://api.github.com/repos/cactus-compute/needle/downloads","issues_url":"https://api.github.com/repos/cactus-compute/needle/issues{/number}","pulls_url":"https://api.github.com/repos/cactus-compute/needle/pulls{/number}","milestones_url":"https://api.github.com/repos/cactus-compute/needle/milestones{/number}","notifications_url":"https://api.github.com/repos/cactus-compute/needle/notifications{?since,all,participating}","labels_url":"https://api.github.com/repos/cactus-compute/needle/labels{/name}","releases_url":"https://api.github.com/repos/cactus-compute/needle/releases{/id}","deployments_url":"https://api.github.com/repos/cactus-compute/needle/deployments","created_at":"2026-02-24T04:50:47Z","updated_at":"2026-08-12T14:40:10Z","pushed_at":"2026-08-11T15:53:23Z","git_url":"git://github.com/cactus-compute/needle.git","ssh_url":"git@github.com:cactus-compute/needle.git","clone_url":"https://github.com/cactus-compute/needle.git","svn_url":"https://github.com/cactus-compute/needle","homepage":"https://cactuscompute.com","size":4309,"stargazers_count":3963,"watchers_count":3963,"language":"Python","has_issues":true,"has_projects":true,"has_downloads":false,"has_wiki":true,"has_pages":false,"has_discussions":false,"forks_count":295,"mirror_url":null,"archived":false,"disabled":false,"open_issues_count":30,"license":{"key":"mit","name":"MIT License","spdx_id":"MIT","url":"https://api.github.com/licenses/mit","node_id":"MDc6TGljZW5zZTEz"},"allow_forking":true,"is_template":false,"web_commit_signoff_required":false,"has_pull_requests":true,"pull_request_creation_policy":"all","topics":["cactus","gemini","gemma","llm","on-device-ai"],"visibility":"public","forks":295,"open_issues":30,"watchers":3963,"default_branch":"main","temp_clone_token":null,"custom_properties":{},"organization":{"login":"cactus-compute","id":196640840,"node_id":"O_kgDOC7iASA","avatar_url":"https://avatars.githubusercontent.com/u/196640840?v=4","gravatar_id":"","url":"https://api.github.com/users/cactus-compute","html_url":"https://github.com/cactus-compute","followers_url":"https://api.github.com/users/cactus-compute/followers","following_url":"https://api.github.com/users/cactus-compute/following{/other_user}","gists_url":"https://api.github.com/users/cactus-compute/gists{/gist_id}","starred_url":"https://api.github.com/users/cactus-compute/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/cactus-compute/subscriptions","organizations_url":"https://api.github.com/users/cactus-compute/orgs","repos_url":"https://api.github.com/users/cactus-compute/repos","events_url":"https://api.github.com/users/cactus-compute/events{/privacy}","received_events_url":"https://api.github.com/users/cactus-compute/received_events","type":"Organization","user_view_type":"public","site_admin":false},"network_count":295,"subscribers_count":31}
\ No newline at end of file
diff --git a/.verify/cactus-compute_needle_README.md b/.verify/cactus-compute_needle_README.md
new file mode 100644
index 0000000..dee52a5
--- /dev/null
+++ b/.verify/cactus-compute_needle_README.md
@@ -0,0 +1,268 @@
+
+
+# Needle 2
+
+Needle 2 is an open 45M-parameter model for tool calling, device use and structured extraction. The whole model is a single 14MB binary that runs a full session in about 28MB of RAM. It is built on our Simple Attention Network findings, compressed to CQ2-bit with Cactus Quants, and baked into its own engine. On the benchmarks below, Needle 2 trades wins with other small models like FunctionGemma 270M, LFM2.5 230M and Apple FM, at 5x to 70x smaller, and 2 bits against their f16.
+
+This repository is the Python package: inference, LoRA fine-tuning, and export. `pip install cactus-needle`, describe your tools, and call them from Python. The inference engine is fetched once from Hugging Face and cached; there is nothing else to build.
+
+- **Self-contained**: weights baked into a single 14MB engine; no separate model files to manage, and inference does no network.
+- **Simple contract**: tool calls come back as structured data, text in, JSON out; a byte-level grammar compiled from your schemas constrains every token.
+- **Confidence-gated**: every response carries a calibrated confidence score from a learned head; set a threshold, act above it, escalate below it.
+- **Tool retrieval**: declare a large catalogue and a built-in retrieval head renders only the top five tools per turn, with the grammar constrained to that subset.
+- **Bounded memory**: a 256-token sliding window with the tools pinned as KV sinks, so total memory stays near 28MB no matter how long the conversation runs.
+
+Weights: [huggingface.co/Cactus-Compute/needle2](https://huggingface.co/Cactus-Compute/needle2) · source: [github.com/cactus-compute/needle](https://github.com/cactus-compute/needle).
+
+
+
+## Simple Attention Network
+
+Needle 2 is a Simple Attention Network, our dense small-model recipe: a Hadamard MLP in place of the FFN, GQA attention, engram key-value memory, and multi-lane hyper-connections. See the paper for the design and ablations: [arXiv:2607.18363](https://arxiv.org/abs/2607.18363).
+
+
+
+Each block carries its update rule. Here x̂ is the RMS-normalised flattening of the four residual streams, H the orthonormal Walsh-Hadamard transform (a fixed matrix, applied in n log n time with no weights to read), (kₜ, vₜ) rows gathered from hashed n-gram tables, and P the doubly-stochastic normalisation of the routing logits A, computed by Sinkhorn iteration; a, b, g and all σ-gates are learned and input-dependent. Both attention and MLP residuals are sandwich-normed and gated, the engram sites fire at two layers, and decoding is constrained by a byte-level grammar compiled from the declared schemas.
+
+## Quickstart
+
+```sh
+pip install cactus-needle
+```
+
+Needle reads your tool descriptions to decide what to call and how to fill arguments, so describing them well is the whole game. You can do it three ways, from least to most control.
+
+**Simple**: decorate a function. The signature gives the argument types, the docstring is the tool description, and `run()` completes the loop: model picks the call, Needle executes your function, feeds the result back, and returns the final response with the executed tool results attached as `results`.
+
+```python
+import needle
+
+@needle.tool
+def get_weather(city: str):
+ "Get the current weather for a city."
+ return {"city": city, "temp_c": 27, "sky": "clear"}
+
+agent = needle.Needle(tools=[get_weather])
+print(agent.run("what's it like in Lagos right now?")["results"])
+# [{'city': 'Lagos', 'temp_c': 27, 'sky': 'clear'}]
+```
+
+**Medium**: describe each argument and offer choices. Needle reads a Google-style `Args:` block for per-parameter descriptions; a default makes an argument optional; a `Literal` becomes a fixed set the model must choose from (it cannot emit anything else).
+
+```python
+from typing import Literal
+
+@needle.tool
+def set_thermostat(temperature: int, mode: Literal["heat", "cool", "auto"] = "auto"):
+ """Set the thermostat.
+
+ Args:
+ temperature: target temperature in Celsius
+ mode: heating strategy to use
+ """
+ return {"temperature": temperature, "mode": mode}
+
+agent = needle.Needle(tools=[set_thermostat])
+agent.run("make it 21 and cool the room")
+```
+
+**Advanced**: constrain the values with `needle.Field`, attached inline via `Annotated`. Ranges, patterns, lengths, and item counts are compiled into the decode grammar, so the model can only ever emit values that satisfy them.
+
+```python
+from typing import Annotated
+
+@needle.tool
+def send_money(
+ amount: Annotated[float, needle.Field(gt=0, le=10000, description="USD, up to 10,000")],
+ to: Annotated[str, needle.Field(pattern=r"^@[a-z0-9_]+$", description="recipient handle")],
+ memo: Annotated[str, needle.Field(max_length=80)] = "",
+):
+ "Send money to a handle."
+ return {"sent": amount, "to": to}
+```
+
+`Field` supports `description`, `enum`, `const`, `ge`/`le`/`gt`/`lt`, `multiple_of`, `min_length`/`max_length`, `pattern`, `format`, `min_items`/`max_items`, and `unique_items`.
+
+**Extraction**: to pull structured data out of text, declare the shape and call `extract()`. Pass a Pydantic model and you get a typed object back.
+
+```python
+from pydantic import BaseModel
+
+class Invoice(BaseModel):
+ vendor: str
+ total: float
+ due_date: str
+
+invoice = needle.extract("Invoice from Acme Corp, $1,200.00, due 2026-09-01", Invoice)
+print(invoice.vendor, invoice.total) # -> Acme Corp 1200.0
+```
+
+**By hand** - the decorator just builds a JSON schema; you can pass that schema directly, which is exactly what Needle consumes. This is how you set descriptions and constraints without the decorator:
+
+```python
+tools = [{
+ "name": "set_lights",
+ "description": "Turn a room's lights on or off and set brightness",
+ "parameters": {
+ "type": "object",
+ "properties": {
+ "room": {"type": "string", "description": "which room to control"},
+ "on": {"type": "boolean"},
+ "brightness": {"type": "integer", "minimum": 0, "maximum": 100},
+ },
+ "required": ["room", "on"],
+ },
+}]
+agent = needle.Needle(tools=tools)
+```
+
+Prefer to drive the loop yourself instead of `run()`? `complete()` returns the raw call and you execute it:
+
+```python
+import json
+response = agent.complete("dim the living room to 30")
+if response["type"] == "call":
+ result = set_lights(**response["function_calls"][0]["arguments"])
+ response = agent.complete(json.dumps(result)) # feed the result back
+```
+
+With a large catalogue, persist tool embeddings across runs with `needle.Needle(tools=..., tool_index_path="tools.idx")`. Every turn returns one JSON object:
+
+```json
+{
+ "type": "call",
+ "success": true,
+ "error": null,
+ "error_code": null,
+ "function_calls": [ { "name": "set_lights", "arguments": { "room": "living room", "on": true, "brightness": 30 } } ],
+ "reasoning": "'living room' -> room; 'dim' -> on true, brightness 30",
+ "confidence": 0.94,
+ "prefill_tps": 4300.0,
+ "decode_tps": 850.0
+}
+```
+
+## Playground
+
+Try any model in the browser: pick a preset, edit the tools or prompt, and Run. Follow-up queries continue the same conversation.
+
+```sh
+needle playground # base model, http://127.0.0.1:7860
+needle playground --weights my.cact # a tuned model
+```
+
+The server downloads and initializes the model before serving, so the first query is instant. The **Finetune on these tools** button runs the fine-tuning pipeline below from the UI and hands back a downloadable `.cact`.
+
+## Behaviour
+
+Needle solves every problem as a function call. The context declares what may be called; the model answers with calls. Performing an action and extracting structured data are the same operation, the only difference is what you declare.
+
+- A request no declared tool can serve is refused with the empty call `[]`. That is the whole contract for off-topic input; there is no free-text fallback.
+- Arguments contain only values evidenced by the input. An optional field with no evidence is omitted, not guessed; omission is the field-level `[]`.
+- `reasoning` is the model's short derivation of each argument from its source span (`'ten minutes' -> minutes 10`). It is generated unconstrained; only the call itself is grammar-constrained, so the JSON cannot be malformed while the derivation stays legible.
+- After you execute a call, pass the result back as the next `complete()`. The model continues from it, and later arguments may depend on earlier results: `search_for_contact` first, then `send_instant_message` with the returned `contact_id`. A final `"type": "respond"` with empty `function_calls` signals the loop is done; the answer is the tool results themselves, which `run()` collects on the final response as `results`. No free text is generated.
+- A session shares one toolset. Later turns are bare queries against the same tools; `reset()` rewinds the conversation and keeps the tools loaded.
+
+## Extraction
+
+Extraction is not a separate mode - it is tool calling with one tool. Declare the record as the only schema and pass the content where the query goes; the returned call's `arguments` are the extracted fields. With one declared tool the grammar admits exactly one call of that name, so schema conformance is guaranteed rather than requested. Use the `extract()` helper for a typed result (shown in Quickstart), or pass a plain schema and read the call:
+
+```python
+receipt = [{
+ "name": "receipt",
+ "description": "A purchase receipt shared as text",
+ "parameters": {
+ "type": "object",
+ "properties": {
+ "merchant": {"type": "string"},
+ "total": {"type": "number"},
+ "currency": {"type": "string"},
+ "line_items": {"type": "array", "items": {"type": "object"}},
+ },
+ "required": ["merchant", "total"],
+ },
+}]
+agent = needle.Needle(tools=receipt)
+print(agent.complete("GreenMart receipt: oat milk 3.50, total 7.75 paid by visa")["function_calls"])
+# -> [{"name": "receipt", "arguments": {"merchant": "GreenMart", "total": 7.75}}]
+```
+
+Because it is the same operation, everything else applies unchanged: `confidence` gates the extraction, unsupported input returns the empty call `[]`, and fine-tuning uses the same data format (the record as the tool, the passage as the query).
+
+## System facts
+
+An optional system turn carries environment state as facts, never instructions:
+
+```python
+agent = needle.Needle(tools=tools, system="date: 2026-07-21 Tue 14:30; locale: en-US; device: phone; battery: 62%")
+```
+
+Recognized keys are `date`, `locale`, `device`, `battery`, `network`, `location`, `user`, and `assistant`. The model resolves relative language against them: "tomorrow at 7" becomes an absolute time only when a `date:` fact licenses it, otherwise the human phrase passes through verbatim. `assistant:` declares the identity the model binds to. Needle trains with and without the turn, so omitting it is safe; instructions placed there do not steer the model.
+
+## Tool retrieval
+
+Five or fewer declared tools render directly. Above that, retrieval engages: at init every tool schema is embedded once by a built-in contrastive head, each turn embeds the query, and only the five highest-scoring tools enter the context, with the grammar rebuilt over just that subset. An unselected tool is unreachable, not merely unlikely. `tool_index_path` persists the embeddings on disk, keyed by a fingerprint over the schemas and the model; a matching fingerprint loads instantly, a changed schema re-embeds only what changed.
+
+## Confidence
+
+The `confidence` field is the minimum of two signals: a calibrated post-hoc head that scores the full prompt plus the call the model just produced, and the decoding probability of the call tokens. A call is accepted only when both agree, so the failure mode is escalation, not wrong execution. The contract: pick a threshold for your product, act at or above it, re-ask or route to a bigger model below it. Off-topic requests return the empty call `[]`.
+
+## Fine-tuning
+
+Needle fine-tunes with LoRA on the frozen base and merges the adapter at export, so a run is cheap and the tuned model is still a single `.cact` that runs on the same engine. The workflow is: (optionally) synthesize data, LoRA fine-tune, then build a tuned `.cact`.
+
+**Data format.** A JSONL file, one example per line. `reasoning` is optional; an off-topic example has `answers: []`.
+
+```json
+{"query": "dim the kitchen to 10", "tools": [{"name": "set_lights", "parameters": {"type": "object", "properties": {"room": {"type": "string"}, "brightness": {"type": "integer"}}, "required": ["room"]}}], "answers": [{"name": "set_lights", "arguments": {"room": "kitchen", "brightness": 10}}], "reasoning": "'kitchen' -> room; 'dim to 10' -> brightness 10"}
+```
+
+**1. Synthesize data (optional).** Needs `OPENROUTER_API_KEY`. Seed from a tool schema file, or expand an existing set:
+
+```sh
+export OPENROUTER_API_KEY=sk-or-...
+needle generate-data --tools my_tools.json --num-samples 500 --output data.jsonl
+needle generate-data --augment data.jsonl --num-samples 500 # expand an existing JSONL
+```
+
+**2. LoRA fine-tune.** The base checkpoint auto-downloads from Hugging Face if you do not pass `--checkpoint`. `--generate N` first synthesizes N more examples from the tools in your data (also needs `OPENROUTER_API_KEY`).
+
+```sh
+needle finetune data.jsonl --epochs 3
+needle finetune data.jsonl --epochs 3 --generate 300 --lora-rank 16 --lora-alpha 32
+```
+
+Key options: `--lora-rank` (default 16), `--lora-alpha` (32), `--lr` (1e-4), `--batch-size` (16), `--max-len` (1024), `--checkpoint `, `--out `. The adapter is written to `checkpoints/needle_lora.pkl`.
+
+**3. Build a tuned `.cact`.** Merge the adapter into the base and quantize. The base auto-downloads if absent.
+
+```sh
+needle build checkpoints/needle2.pkl --lora checkpoints/needle_lora.pkl --out my_needle.cact
+```
+
+Add `--bits 2` (default 4) for a smaller model, or set `NEEDLE_HF_REPO=/` and pass `--upload` to publish the `.cact`.
+
+**4. Run it.** The engine is weights-agnostic, so a tuned `.cact` runs on it directly - no recompilation:
+
+```python
+import needle
+agent = needle.Needle(weights="my_needle.cact", tools=[...])
+agent.run("...")
+```
+
+## Citation
+
+Needle 2 is built by the Cactus Compute team. If you use it in your work, please cite:
+
+```bibtex
+@misc{needle2_2026,
+ title = {Needle 2: A 45M-Parameter Foundation Tool-Calling Model for Tiny Devices},
+ author = {Ndubuaku, Henry and Mosoyan, Karen and Mroz, Jakub and Cylich, Noah and
+ Kumar, Satyajit and Sandhu, Parkirat and Shemet, Roman and Lee, Justin H.},
+ year = {2026},
+ organization = {Cactus Compute, Inc.},
+ howpublished = {\url{https://github.com/cactus-compute/needle}}
+}
+```
+
+Reach out on founders@cactuscompute.com for partnerships, collaborations, synergies and deploying Needle2 in your product.
diff --git a/.verify/cactus-compute_needle_README_alt.md b/.verify/cactus-compute_needle_README_alt.md
new file mode 100644
index 0000000..dee52a5
--- /dev/null
+++ b/.verify/cactus-compute_needle_README_alt.md
@@ -0,0 +1,268 @@
+
+
+# Needle 2
+
+Needle 2 is an open 45M-parameter model for tool calling, device use and structured extraction. The whole model is a single 14MB binary that runs a full session in about 28MB of RAM. It is built on our Simple Attention Network findings, compressed to CQ2-bit with Cactus Quants, and baked into its own engine. On the benchmarks below, Needle 2 trades wins with other small models like FunctionGemma 270M, LFM2.5 230M and Apple FM, at 5x to 70x smaller, and 2 bits against their f16.
+
+This repository is the Python package: inference, LoRA fine-tuning, and export. `pip install cactus-needle`, describe your tools, and call them from Python. The inference engine is fetched once from Hugging Face and cached; there is nothing else to build.
+
+- **Self-contained**: weights baked into a single 14MB engine; no separate model files to manage, and inference does no network.
+- **Simple contract**: tool calls come back as structured data, text in, JSON out; a byte-level grammar compiled from your schemas constrains every token.
+- **Confidence-gated**: every response carries a calibrated confidence score from a learned head; set a threshold, act above it, escalate below it.
+- **Tool retrieval**: declare a large catalogue and a built-in retrieval head renders only the top five tools per turn, with the grammar constrained to that subset.
+- **Bounded memory**: a 256-token sliding window with the tools pinned as KV sinks, so total memory stays near 28MB no matter how long the conversation runs.
+
+Weights: [huggingface.co/Cactus-Compute/needle2](https://huggingface.co/Cactus-Compute/needle2) · source: [github.com/cactus-compute/needle](https://github.com/cactus-compute/needle).
+
+
+
+## Simple Attention Network
+
+Needle 2 is a Simple Attention Network, our dense small-model recipe: a Hadamard MLP in place of the FFN, GQA attention, engram key-value memory, and multi-lane hyper-connections. See the paper for the design and ablations: [arXiv:2607.18363](https://arxiv.org/abs/2607.18363).
+
+
+
+Each block carries its update rule. Here x̂ is the RMS-normalised flattening of the four residual streams, H the orthonormal Walsh-Hadamard transform (a fixed matrix, applied in n log n time with no weights to read), (kₜ, vₜ) rows gathered from hashed n-gram tables, and P the doubly-stochastic normalisation of the routing logits A, computed by Sinkhorn iteration; a, b, g and all σ-gates are learned and input-dependent. Both attention and MLP residuals are sandwich-normed and gated, the engram sites fire at two layers, and decoding is constrained by a byte-level grammar compiled from the declared schemas.
+
+## Quickstart
+
+```sh
+pip install cactus-needle
+```
+
+Needle reads your tool descriptions to decide what to call and how to fill arguments, so describing them well is the whole game. You can do it three ways, from least to most control.
+
+**Simple**: decorate a function. The signature gives the argument types, the docstring is the tool description, and `run()` completes the loop: model picks the call, Needle executes your function, feeds the result back, and returns the final response with the executed tool results attached as `results`.
+
+```python
+import needle
+
+@needle.tool
+def get_weather(city: str):
+ "Get the current weather for a city."
+ return {"city": city, "temp_c": 27, "sky": "clear"}
+
+agent = needle.Needle(tools=[get_weather])
+print(agent.run("what's it like in Lagos right now?")["results"])
+# [{'city': 'Lagos', 'temp_c': 27, 'sky': 'clear'}]
+```
+
+**Medium**: describe each argument and offer choices. Needle reads a Google-style `Args:` block for per-parameter descriptions; a default makes an argument optional; a `Literal` becomes a fixed set the model must choose from (it cannot emit anything else).
+
+```python
+from typing import Literal
+
+@needle.tool
+def set_thermostat(temperature: int, mode: Literal["heat", "cool", "auto"] = "auto"):
+ """Set the thermostat.
+
+ Args:
+ temperature: target temperature in Celsius
+ mode: heating strategy to use
+ """
+ return {"temperature": temperature, "mode": mode}
+
+agent = needle.Needle(tools=[set_thermostat])
+agent.run("make it 21 and cool the room")
+```
+
+**Advanced**: constrain the values with `needle.Field`, attached inline via `Annotated`. Ranges, patterns, lengths, and item counts are compiled into the decode grammar, so the model can only ever emit values that satisfy them.
+
+```python
+from typing import Annotated
+
+@needle.tool
+def send_money(
+ amount: Annotated[float, needle.Field(gt=0, le=10000, description="USD, up to 10,000")],
+ to: Annotated[str, needle.Field(pattern=r"^@[a-z0-9_]+$", description="recipient handle")],
+ memo: Annotated[str, needle.Field(max_length=80)] = "",
+):
+ "Send money to a handle."
+ return {"sent": amount, "to": to}
+```
+
+`Field` supports `description`, `enum`, `const`, `ge`/`le`/`gt`/`lt`, `multiple_of`, `min_length`/`max_length`, `pattern`, `format`, `min_items`/`max_items`, and `unique_items`.
+
+**Extraction**: to pull structured data out of text, declare the shape and call `extract()`. Pass a Pydantic model and you get a typed object back.
+
+```python
+from pydantic import BaseModel
+
+class Invoice(BaseModel):
+ vendor: str
+ total: float
+ due_date: str
+
+invoice = needle.extract("Invoice from Acme Corp, $1,200.00, due 2026-09-01", Invoice)
+print(invoice.vendor, invoice.total) # -> Acme Corp 1200.0
+```
+
+**By hand** - the decorator just builds a JSON schema; you can pass that schema directly, which is exactly what Needle consumes. This is how you set descriptions and constraints without the decorator:
+
+```python
+tools = [{
+ "name": "set_lights",
+ "description": "Turn a room's lights on or off and set brightness",
+ "parameters": {
+ "type": "object",
+ "properties": {
+ "room": {"type": "string", "description": "which room to control"},
+ "on": {"type": "boolean"},
+ "brightness": {"type": "integer", "minimum": 0, "maximum": 100},
+ },
+ "required": ["room", "on"],
+ },
+}]
+agent = needle.Needle(tools=tools)
+```
+
+Prefer to drive the loop yourself instead of `run()`? `complete()` returns the raw call and you execute it:
+
+```python
+import json
+response = agent.complete("dim the living room to 30")
+if response["type"] == "call":
+ result = set_lights(**response["function_calls"][0]["arguments"])
+ response = agent.complete(json.dumps(result)) # feed the result back
+```
+
+With a large catalogue, persist tool embeddings across runs with `needle.Needle(tools=..., tool_index_path="tools.idx")`. Every turn returns one JSON object:
+
+```json
+{
+ "type": "call",
+ "success": true,
+ "error": null,
+ "error_code": null,
+ "function_calls": [ { "name": "set_lights", "arguments": { "room": "living room", "on": true, "brightness": 30 } } ],
+ "reasoning": "'living room' -> room; 'dim' -> on true, brightness 30",
+ "confidence": 0.94,
+ "prefill_tps": 4300.0,
+ "decode_tps": 850.0
+}
+```
+
+## Playground
+
+Try any model in the browser: pick a preset, edit the tools or prompt, and Run. Follow-up queries continue the same conversation.
+
+```sh
+needle playground # base model, http://127.0.0.1:7860
+needle playground --weights my.cact # a tuned model
+```
+
+The server downloads and initializes the model before serving, so the first query is instant. The **Finetune on these tools** button runs the fine-tuning pipeline below from the UI and hands back a downloadable `.cact`.
+
+## Behaviour
+
+Needle solves every problem as a function call. The context declares what may be called; the model answers with calls. Performing an action and extracting structured data are the same operation, the only difference is what you declare.
+
+- A request no declared tool can serve is refused with the empty call `[]`. That is the whole contract for off-topic input; there is no free-text fallback.
+- Arguments contain only values evidenced by the input. An optional field with no evidence is omitted, not guessed; omission is the field-level `[]`.
+- `reasoning` is the model's short derivation of each argument from its source span (`'ten minutes' -> minutes 10`). It is generated unconstrained; only the call itself is grammar-constrained, so the JSON cannot be malformed while the derivation stays legible.
+- After you execute a call, pass the result back as the next `complete()`. The model continues from it, and later arguments may depend on earlier results: `search_for_contact` first, then `send_instant_message` with the returned `contact_id`. A final `"type": "respond"` with empty `function_calls` signals the loop is done; the answer is the tool results themselves, which `run()` collects on the final response as `results`. No free text is generated.
+- A session shares one toolset. Later turns are bare queries against the same tools; `reset()` rewinds the conversation and keeps the tools loaded.
+
+## Extraction
+
+Extraction is not a separate mode - it is tool calling with one tool. Declare the record as the only schema and pass the content where the query goes; the returned call's `arguments` are the extracted fields. With one declared tool the grammar admits exactly one call of that name, so schema conformance is guaranteed rather than requested. Use the `extract()` helper for a typed result (shown in Quickstart), or pass a plain schema and read the call:
+
+```python
+receipt = [{
+ "name": "receipt",
+ "description": "A purchase receipt shared as text",
+ "parameters": {
+ "type": "object",
+ "properties": {
+ "merchant": {"type": "string"},
+ "total": {"type": "number"},
+ "currency": {"type": "string"},
+ "line_items": {"type": "array", "items": {"type": "object"}},
+ },
+ "required": ["merchant", "total"],
+ },
+}]
+agent = needle.Needle(tools=receipt)
+print(agent.complete("GreenMart receipt: oat milk 3.50, total 7.75 paid by visa")["function_calls"])
+# -> [{"name": "receipt", "arguments": {"merchant": "GreenMart", "total": 7.75}}]
+```
+
+Because it is the same operation, everything else applies unchanged: `confidence` gates the extraction, unsupported input returns the empty call `[]`, and fine-tuning uses the same data format (the record as the tool, the passage as the query).
+
+## System facts
+
+An optional system turn carries environment state as facts, never instructions:
+
+```python
+agent = needle.Needle(tools=tools, system="date: 2026-07-21 Tue 14:30; locale: en-US; device: phone; battery: 62%")
+```
+
+Recognized keys are `date`, `locale`, `device`, `battery`, `network`, `location`, `user`, and `assistant`. The model resolves relative language against them: "tomorrow at 7" becomes an absolute time only when a `date:` fact licenses it, otherwise the human phrase passes through verbatim. `assistant:` declares the identity the model binds to. Needle trains with and without the turn, so omitting it is safe; instructions placed there do not steer the model.
+
+## Tool retrieval
+
+Five or fewer declared tools render directly. Above that, retrieval engages: at init every tool schema is embedded once by a built-in contrastive head, each turn embeds the query, and only the five highest-scoring tools enter the context, with the grammar rebuilt over just that subset. An unselected tool is unreachable, not merely unlikely. `tool_index_path` persists the embeddings on disk, keyed by a fingerprint over the schemas and the model; a matching fingerprint loads instantly, a changed schema re-embeds only what changed.
+
+## Confidence
+
+The `confidence` field is the minimum of two signals: a calibrated post-hoc head that scores the full prompt plus the call the model just produced, and the decoding probability of the call tokens. A call is accepted only when both agree, so the failure mode is escalation, not wrong execution. The contract: pick a threshold for your product, act at or above it, re-ask or route to a bigger model below it. Off-topic requests return the empty call `[]`.
+
+## Fine-tuning
+
+Needle fine-tunes with LoRA on the frozen base and merges the adapter at export, so a run is cheap and the tuned model is still a single `.cact` that runs on the same engine. The workflow is: (optionally) synthesize data, LoRA fine-tune, then build a tuned `.cact`.
+
+**Data format.** A JSONL file, one example per line. `reasoning` is optional; an off-topic example has `answers: []`.
+
+```json
+{"query": "dim the kitchen to 10", "tools": [{"name": "set_lights", "parameters": {"type": "object", "properties": {"room": {"type": "string"}, "brightness": {"type": "integer"}}, "required": ["room"]}}], "answers": [{"name": "set_lights", "arguments": {"room": "kitchen", "brightness": 10}}], "reasoning": "'kitchen' -> room; 'dim to 10' -> brightness 10"}
+```
+
+**1. Synthesize data (optional).** Needs `OPENROUTER_API_KEY`. Seed from a tool schema file, or expand an existing set:
+
+```sh
+export OPENROUTER_API_KEY=sk-or-...
+needle generate-data --tools my_tools.json --num-samples 500 --output data.jsonl
+needle generate-data --augment data.jsonl --num-samples 500 # expand an existing JSONL
+```
+
+**2. LoRA fine-tune.** The base checkpoint auto-downloads from Hugging Face if you do not pass `--checkpoint`. `--generate N` first synthesizes N more examples from the tools in your data (also needs `OPENROUTER_API_KEY`).
+
+```sh
+needle finetune data.jsonl --epochs 3
+needle finetune data.jsonl --epochs 3 --generate 300 --lora-rank 16 --lora-alpha 32
+```
+
+Key options: `--lora-rank` (default 16), `--lora-alpha` (32), `--lr` (1e-4), `--batch-size` (16), `--max-len` (1024), `--checkpoint `, `--out `. The adapter is written to `checkpoints/needle_lora.pkl`.
+
+**3. Build a tuned `.cact`.** Merge the adapter into the base and quantize. The base auto-downloads if absent.
+
+```sh
+needle build checkpoints/needle2.pkl --lora checkpoints/needle_lora.pkl --out my_needle.cact
+```
+
+Add `--bits 2` (default 4) for a smaller model, or set `NEEDLE_HF_REPO=/` and pass `--upload` to publish the `.cact`.
+
+**4. Run it.** The engine is weights-agnostic, so a tuned `.cact` runs on it directly - no recompilation:
+
+```python
+import needle
+agent = needle.Needle(weights="my_needle.cact", tools=[...])
+agent.run("...")
+```
+
+## Citation
+
+Needle 2 is built by the Cactus Compute team. If you use it in your work, please cite:
+
+```bibtex
+@misc{needle2_2026,
+ title = {Needle 2: A 45M-Parameter Foundation Tool-Calling Model for Tiny Devices},
+ author = {Ndubuaku, Henry and Mosoyan, Karen and Mroz, Jakub and Cylich, Noah and
+ Kumar, Satyajit and Sandhu, Parkirat and Shemet, Roman and Lee, Justin H.},
+ year = {2026},
+ organization = {Cactus Compute, Inc.},
+ howpublished = {\url{https://github.com/cactus-compute/needle}}
+}
+```
+
+Reach out on founders@cactuscompute.com for partnerships, collaborations, synergies and deploying Needle2 in your product.
diff --git a/.verify/embabel_embabel-agent.json b/.verify/embabel_embabel-agent.json
new file mode 100644
index 0000000..38bb00b
--- /dev/null
+++ b/.verify/embabel_embabel-agent.json
@@ -0,0 +1 @@
+{"id":963665823,"node_id":"R_kgDOOXBfnw","name":"embabel-agent","full_name":"embabel/embabel-agent","private":false,"owner":{"login":"embabel","id":152664703,"node_id":"O_kgDOCRl6fw","avatar_url":"https://avatars.githubusercontent.com/u/152664703?v=4","gravatar_id":"","url":"https://api.github.com/users/embabel","html_url":"https://github.com/embabel","followers_url":"https://api.github.com/users/embabel/followers","following_url":"https://api.github.com/users/embabel/following{/other_user}","gists_url":"https://api.github.com/users/embabel/gists{/gist_id}","starred_url":"https://api.github.com/users/embabel/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/embabel/subscriptions","organizations_url":"https://api.github.com/users/embabel/orgs","repos_url":"https://api.github.com/users/embabel/repos","events_url":"https://api.github.com/users/embabel/events{/privacy}","received_events_url":"https://api.github.com/users/embabel/received_events","type":"Organization","user_view_type":"public","site_admin":false},"html_url":"https://github.com/embabel/embabel-agent","description":"Agent framework for the JVM. Pronounced Em-BAY-bel /ɛmˈbeɪbəl/","fork":false,"url":"https://api.github.com/repos/embabel/embabel-agent","forks_url":"https://api.github.com/repos/embabel/embabel-agent/forks","keys_url":"https://api.github.com/repos/embabel/embabel-agent/keys{/key_id}","collaborators_url":"https://api.github.com/repos/embabel/embabel-agent/collaborators{/collaborator}","teams_url":"https://api.github.com/repos/embabel/embabel-agent/teams","hooks_url":"https://api.github.com/repos/embabel/embabel-agent/hooks","issue_events_url":"https://api.github.com/repos/embabel/embabel-agent/issues/events{/number}","events_url":"https://api.github.com/repos/embabel/embabel-agent/events","assignees_url":"https://api.github.com/repos/embabel/embabel-agent/assignees{/user}","branches_url":"https://api.github.com/repos/embabel/embabel-agent/branches{/branch}","tags_url":"https://api.github.com/repos/embabel/embabel-agent/tags","blobs_url":"https://api.github.com/repos/embabel/embabel-agent/git/blobs{/sha}","git_tags_url":"https://api.github.com/repos/embabel/embabel-agent/git/tags{/sha}","git_refs_url":"https://api.github.com/repos/embabel/embabel-agent/git/refs{/sha}","trees_url":"https://api.github.com/repos/embabel/embabel-agent/git/trees{/sha}","statuses_url":"https://api.github.com/repos/embabel/embabel-agent/statuses/{sha}","languages_url":"https://api.github.com/repos/embabel/embabel-agent/languages","stargazers_url":"https://api.github.com/repos/embabel/embabel-agent/stargazers","contributors_url":"https://api.github.com/repos/embabel/embabel-agent/contributors","subscribers_url":"https://api.github.com/repos/embabel/embabel-agent/subscribers","subscription_url":"https://api.github.com/repos/embabel/embabel-agent/subscription","commits_url":"https://api.github.com/repos/embabel/embabel-agent/commits{/sha}","git_commits_url":"https://api.github.com/repos/embabel/embabel-agent/git/commits{/sha}","comments_url":"https://api.github.com/repos/embabel/embabel-agent/comments{/number}","issue_comment_url":"https://api.github.com/repos/embabel/embabel-agent/issues/comments{/number}","contents_url":"https://api.github.com/repos/embabel/embabel-agent/contents/{+path}","compare_url":"https://api.github.com/repos/embabel/embabel-agent/compare/{base}...{head}","merges_url":"https://api.github.com/repos/embabel/embabel-agent/merges","archive_url":"https://api.github.com/repos/embabel/embabel-agent/{archive_format}{/ref}","downloads_url":"https://api.github.com/repos/embabel/embabel-agent/downloads","issues_url":"https://api.github.com/repos/embabel/embabel-agent/issues{/number}","pulls_url":"https://api.github.com/repos/embabel/embabel-agent/pulls{/number}","milestones_url":"https://api.github.com/repos/embabel/embabel-agent/milestones{/number}","notifications_url":"https://api.github.com/repos/embabel/embabel-agent/notifications{?since,all,participating}","labels_url":"https://api.github.com/repos/embabel/embabel-agent/labels{/name}","releases_url":"https://api.github.com/repos/embabel/embabel-agent/releases{/id}","deployments_url":"https://api.github.com/repos/embabel/embabel-agent/deployments","created_at":"2025-04-10T03:06:07Z","updated_at":"2026-08-12T14:28:40Z","pushed_at":"2026-08-12T08:02:50Z","git_url":"git://github.com/embabel/embabel-agent.git","ssh_url":"git@github.com:embabel/embabel-agent.git","clone_url":"https://github.com/embabel/embabel-agent.git","svn_url":"https://github.com/embabel/embabel-agent","homepage":"https://hub.embabel.com","size":19954,"stargazers_count":4175,"watchers_count":4175,"language":"Kotlin","has_issues":true,"has_projects":true,"has_downloads":false,"has_wiki":true,"has_pages":false,"has_discussions":true,"forks_count":410,"mirror_url":null,"archived":false,"disabled":false,"open_issues_count":70,"license":{"key":"apache-2.0","name":"Apache License 2.0","spdx_id":"Apache-2.0","url":"https://api.github.com/licenses/apache-2.0","node_id":"MDc6TGljZW5zZTI="},"allow_forking":true,"is_template":false,"web_commit_signoff_required":false,"has_pull_requests":true,"pull_request_creation_policy":"all","topics":["agent","agentic-ai","agents","ai","ai-agents","aiagentframework","genai","generative-ai","java","kotlin","llms","multi-agents","multi-agents-orchestration","multi-agents-system","spring"],"visibility":"public","forks":410,"open_issues":70,"watchers":4175,"default_branch":"main","temp_clone_token":null,"custom_properties":{},"organization":{"login":"embabel","id":152664703,"node_id":"O_kgDOCRl6fw","avatar_url":"https://avatars.githubusercontent.com/u/152664703?v=4","gravatar_id":"","url":"https://api.github.com/users/embabel","html_url":"https://github.com/embabel","followers_url":"https://api.github.com/users/embabel/followers","following_url":"https://api.github.com/users/embabel/following{/other_user}","gists_url":"https://api.github.com/users/embabel/gists{/gist_id}","starred_url":"https://api.github.com/users/embabel/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/embabel/subscriptions","organizations_url":"https://api.github.com/users/embabel/orgs","repos_url":"https://api.github.com/users/embabel/repos","events_url":"https://api.github.com/users/embabel/events{/privacy}","received_events_url":"https://api.github.com/users/embabel/received_events","type":"Organization","user_view_type":"public","site_admin":false},"network_count":410,"subscribers_count":63}
\ No newline at end of file
diff --git a/.verify/embabel_embabel-agent_README.md b/.verify/embabel_embabel-agent_README.md
new file mode 100644
index 0000000..bf6303c
--- /dev/null
+++ b/.verify/embabel_embabel-agent_README.md
@@ -0,0 +1,1437 @@
+# [Embabel Agent Framework](https://hub.embabel.com)
+
+
+
+[](https://docs.embabel.com/embabel-agent/guide/1.5.0-SNAPSHOT/)
+[](https://central.sonatype.com/artifact/com.embabel.agent/embabel-agent-api)
+
+[](https://www.yourkit.com/)
+[](https://www.ej-technologies.com/products/jprofiler/overview.html)
+[](https://sonarcloud.io/summary/new_code?id=embabel_embabel-agent)
+[](https://discord.gg/t6bjkyj93q)
+
+[//]: # ([](https://sonarcloud.io/summary/new_code?id=embabel_embabel-agent))
+
+[//]: # ([](https://sonarcloud.io/summary/new_code?id=embabel_embabel-agent))
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+[](https://www.apache.org/licenses/LICENSE-2.0)
+[](https://github.com/embabel/embabel-agent/pulse)
+
+
+
+Embabel (Em-BAY-bel) is a framework for authoring agentic flows on the JVM that seamlessly mix LLM-prompted interactions
+with code and domain models. Supports
+intelligent path finding towards goals. Written in Kotlin
+but offers a natural usage
+model from Java.
+From the creator of Spring.
+
+
+
+## Talk to the Docs
+
+Have questions? [Talk to the docs via the Embabel-powered hub](https://hub.embabel.com) — an
+Embabel agent that answers your questions about the framework in natural language.
+
+## Key Concepts
+
+Models agentic flows in terms of:
+
+- **Actions**: Steps an agent takes
+- **Goals**: What an agent is trying to achieve
+- **Conditions**: Conditions to assess before executing an action or determining that a goal has been achieved.
+ Conditions are reassessed after each action is executed.
+- **Domain model**: Objects underpinning the flow and informing Actions, Goals and Conditions.
+- **Plan**: A sequence of actions to achieve a goal. Plans are dynamically formulated by the system, not the programmer.
+ The
+ system replans after the completion of each action, allowing it to adapt to new information as well as observe the
+ effects of the previous action.
+ This is effectively an [OODA loop](https://en.wikipedia.org/wiki/OODA_loop).
+
+> Application developers don't usually have to deal with these concepts directly,
+> as most conditions result from data flow defined in code, allowing the system to infer
+> pre and post conditions.
+
+These concepts underpin these differentiators versus other agent frameworks:
+
+- **Sophisticated planning.** Goes beyond a finite state machine or sequential execution
+ with nesting by introducing a true planning step, using a
+ non-LLM AI algorithm. This enables the system to perform tasks it wasn’t programmed to do by combining known
+ steps in
+ a novel order, as well as make decisions about parallelization and other runtime behavior.
+- **Superior extensibility and reuse**: Because of dynamic planning, adding more domain objects, actions, goals and
+ conditions
+ can extend the capability of the system, _without editing FSM definitions_ or existing code.
+- **Strong typing and the benefits of object orientation**: Actions, goals and conditions are informed by a domain
+ model, which can
+ include behavior. Everything is strongly typed and prompts and
+ manually authored code interact cleanly. No more magic maps. Enjoy full refactoring support.
+
+Other benefits:
+
+- **Platform abstraction**: Clean separation between programming model and platform internals allows running locally
+ while
+ potentially offering higher QoS in production without changing application code.
+- **Designed for LLM mixing**: It is easy to build applications that mix LLMs, ensuring the most cost-effective yet
+ capable solution.
+ This enables the system to leverage the strengths of different models for different tasks. In particular, it
+ facilitates
+ the use of local models for point tasks. This can be important for cost and privacy.
+- **Built on Spring and the JVM,** making it easy to access existing enterprise functionality and capabilities.
+ For example:
+ - Spring can inject and manage agents, including using Spring AOP to decorate functions.
+ - Robust persistence and transaction management solutions are available.
+- **Designed for testability** from the ground up. Both unit testing and agent end to end testing are easy.
+
+Flows can be authored in one of two ways:
+
+- An annotation-based model similar to Spring MVC, with types annotated with the Spring stereotype `@Agent`, using
+ `@Goal`, `@Condition` and
+ `@Action` methods.
+- Idiomatic Kotlin DSL with `agent {` and `action {` blocks.
+
+Either way, flows are backed by a domain model of objects that can have rich behavior.
+
+> We are working toward allowing natural language actions and goals to be deployed.
+
+The planning step is pluggable.
+
+The default planning approach is
+[Goal Oriented Action Planning](https://medium.com/@vedantchaudhari/goal-oriented-action-planning-34035ed40d0b).
+GOAP is a popular AI planning algorithm used in gaming. It allows for dynamic decision-making and action selection based
+on the current state of the world and the goals of the agent.
+
+Goals, actions and plans are independent of GOAP. Embabel also
+supports [Utility AI](https://en.wikipedia.org/wiki/Utility_system) out of the box, which can run the same actions but
+chooses actions
+based on (potentially dynamic) utility scores rather than strict preconditions and postconditions. This is valuable for
+exploration and
+open-ended tasks, when we do not need to achieve a specific goal but want to maximize overall utility.
+
+The framework executes via an `AgentPlatform` implementation.
+
+An agent platform supports the following modes of execution:
+
+- **Focused**, where user code requests particular functionality: User code calls a method to run a particular agent,
+ passing in input. This is ideal for code-driven flows such as a flow invoked in response to an incoming event.
+- **Closed**, where user intent (or another incoming event) is classified to choose an agent. The platform tries to
+ find a
+ suitable agent among all the agents it knows about.
+ Agent choice is dynamic, but only actions defined within the particular agent
+ will run.
+- **Open**, where the user's intent is assessed and the platform uses _all_ its resources to try to achieve it. The
+ platform tries to find a
+ suitable goal among all the goals it knows about and builds a custom agent to achieve it from the start state,
+ including relevant actions and conditions. The platform will not proceed if it is unconvinced as to the applicability
+ of any goal. The `GoalChoiceApprover` interface provides developers a way to limit goal choice further.
+
+Open mode is the most powerful, but least deterministic.
+> In open mode, the platform is capable of finding novel paths that were not envisioned by developers, and even
+> combining functionality from multiple providers.
+
+Even in open mode, the platform will only perform individual steps
+that have been specified. (Of course, steps may themselves be LLM
+transforms, in which case the prompts are controlled by user code but the
+results are still non-deterministic.)
+
+Possible future modes:
+
+- **Evolving** mode: Where the platform can work with multiple goals in the same process and modify a running process to
+ add further goals and agents.
+ For example, an action can realize that it has become important to achieve additional goals.
+
+Embabel agent systems will also support federation, both with other Embabel systems (allowing planning to incorporate
+remote actions and goals) and third party agent frameworks.
+
+## Quick Start
+
+Get an agent running in under 5 minutes.
+
+Create your own agent repo from our [Java](https://github.com/embabel/java-agent-template)
+or [Kotlin](https://github.com/embabel/kotlin-agent-template) GitHub template by clicking the "Use this template"
+button.
+
+You'll have an agent running in under a minute
+if you already have an `OPENAI_API_KEY` and have Maven installed.
+
+**📚 For examples and tutorials**, see
+the [Embabel Agent Examples Repository](https://github.com/embabel/embabel-agent-examples)
+
+**🚗 For a sophisticated, realistic example application**, see
+the [Tripper travel planner agent](https://github.com/embabel/tripper)
+
+
+
+*AI-generated travel itinerary with detailed recommendations*
+
+
+
+*Map link included in output*
+
+## Why Is Embabel Needed?
+
+TL;DR Because the evolution of agent frameworks is early and there's a lot of room for improvement; because an agent
+framework on the JVM will deliver great business value.
+
+- _Why do we need an agent framework at all_? We can write code without higher level abstractions, directly invoking
+ LLMs and controlling flow directly in code. However, a higher level agent framework offers compelling benefits. For
+ example:
+ - Breaking up LLM interactions, making them simpler and more focused. This maximizes reuse and minimizes cost and
+ errors. It often allows us to use cheaper models for point interactions.
+ - Facilitating both unit and integration testing, which remain as important with agentic systems as with any other
+ software systems.
+ - Increasing composability where subflows and individual actions can be reused
+ - Making applications more manageable and robust, enabling a workflow manager to control their execution and retry
+ operations while maintaining previous state
+ - Enhancing safety through the ability to apply guardrails in many places
+- _Why do we need an agent framework for the JVM when solutions exist in Python?_: While agent frameworks initially
+ appeared predominantly Python, it's early and there's plenty of room for novel and
+ superior
+ approaches. The key adjacency is not the LLM--which is a simple HTTP call away--but existing code and
+ infrastructure
+ assets that are more valuable on the JVM than in Python.
+- _Why not use just Spring AI?_ Spring AI is great. We build on it, and embrace the Spring component model. However, we
+ believe that most applications should work with higher
+ level APIs. An analogy: Spring AI exists at the level of the Servlet API, while Embabel is more like Spring MVC.
+ Complex requirements are much easier to express and test in Embabel than with direct use of Spring AI.
+- _Why not attempt to contribute this project to Spring?_ This project requires different governance
+ from Spring, where most projects exist in stable environments and dependability and stability outweighs rapid
+ innovation. Second, the
+ concepts are not JVM-specific. We hope that Embabel will become the leading agent framework across platforms. While
+ the Spring brand is valuable in Java, it is not in TypeScript or Python.
+
+## Show Me The Code
+
+In Java or Kotlin, agent implementation code is intuitive and easy to test.
+
+
+Java
+
+```java
+
+@Agent(description = "Find news based on a person's star sign")
+public class StarNewsFinder {
+
+ private final HoroscopeService horoscopeService;
+ private final int storyCount;
+
+ // Services are injected by Spring
+ public StarNewsFinder(
+ HoroscopeService horoscopeService,
+ @Value("${star-news-finder.story.count:5}") int storyCount) {
+ this.horoscopeService = horoscopeService;
+ this.storyCount = storyCount;
+ }
+
+ @Action
+ public StarPerson extractStarPerson(UserInput userInput, Ai ai) {
+ return ai
+ .withLlm(OpenAiModels.GPT_41)
+ .createObjectIfPossible(
+ """
+ Create a person from this user input, extracting their name and star sign:
+ %s""".formatted(userInput.getContent()),
+ StarPerson.class
+ );
+ }
+
+ @Action
+ public Horoscope retrieveHoroscope(StarPerson starPerson) {
+ return new Horoscope(horoscopeService.dailyHoroscope(starPerson.sign()));
+ }
+
+ // toolGroups specifies tools that are required for this action to run
+ @Action(toolGroups = {CoreToolGroups.WEB})
+ public RelevantNewsStories findNewsStories(
+ StarPerson person,
+ Horoscope horoscope,
+ Ai ai) {
+ var prompt = """
+ %s is an astrology believer with the sign %s.
+ Their horoscope for today is:
+ %s
+ Given this, use web tools and generate search queries
+ to find %d relevant news stories summarize them in a few sentences.
+ Include the URL for each story.
+ Do not look for another horoscope reading or return results directly about astrology;
+ find stories relevant to the reading above.
+
+ For example:
+ - If the horoscope says that they may
+ want to work on relationships, you could find news stories about
+ novel gifts
+ - If the horoscope says that they may want to work on their career,
+ find news stories about training courses.""".formatted(
+ person.name(), person.sign(), horoscope.summary(), storyCount);
+ return ai
+ .withDefaultLlm()
+ .createObject(prompt, RelevantNewsStories.class);
+ }
+
+ // The @AchievesGoal annotation indicates that completing this action
+ // achieves the given goal, so the agent can be complete
+ @AchievesGoal(
+ description = "Write an amusing writeup for the target person based on their horoscope and current news stories",
+ export = @Export(
+ remote = true,
+ name = "starNewsWriteupJava",
+ startingInputTypes = {StarPerson.class, UserInput.class})
+ )
+ @Action
+ public Writeup writeup(
+ StarPerson person,
+ RelevantNewsStories relevantNewsStories,
+ Horoscope horoscope,
+ Ai ai) {
+ var llm = LlmOptions
+ .withModel(OpenAiModels.GPT_41_MINI)
+ // High temperature for creativity
+ .withTemperature(0.9);
+
+ var newsItems = relevantNewsStories.getItems().stream()
+ .map(item -> "- " + item.getUrl() + ": " + item.getSummary())
+ .collect(Collectors.joining("\n"));
+
+ var prompt = """
+ Take the following news stories and write up something
+ amusing for the target person.
+
+ Begin by summarizing their horoscope in a concise, amusing way, then
+ talk about the news. End with a surprising signoff.
+
+ %s is an astrology believer with the sign %s.
+ Their horoscope for today is:
+ %s
+ Relevant news stories are:
+ %s
+
+ Format it as Markdown with links.""".formatted(
+ person.name(), person.sign(), horoscope.summary(), newsItems);
+ return ai
+ .withLlm(llm)
+ .createObject(prompt, Writeup.class);
+ }
+}
+```
+
+
+
+
+Kotlin
+
+```kotlin
+@Agent(description = "Find news based on a person's star sign")
+class StarNewsFinder(
+ // Services such as Horoscope are injected by Spring
+ private val horoscopeService: HoroscopeService,
+ // Potentially externalized by Spring
+ @param:Value("\${star-news-finder.story.count:5}")
+ private val storyCount: Int = 5,
+) {
+
+ @Action
+ fun extractPerson(
+ userInput: UserInput,
+ ai: Ai
+ ): StarPerson =
+ // All prompts are typesafe
+ ai.withDefaultLlm()
+ .createObject("Create a person from this user input, extracting their name and star sign: $userInput")
+
+ // This action doesn't use an LLM
+ // Embabel makes it easy to mix LLM use with regular code
+ @Action
+ fun retrieveHoroscope(starPerson: StarPerson) =
+ Horoscope(horoscopeService.dailyHoroscope(starPerson.sign))
+
+ // This action uses tools
+ // "toolGroups" specifies tools that are required for this action to run
+ @Action(toolGroups = [ToolGroup.WEB])
+ fun findNewsStories(
+ person: StarPerson,
+ horoscope: Horoscope,
+ ai: Ai,
+ ): RelevantNewsStories =
+ ai.withDefaultLlm().createObject(
+ """
+ ${person.name} is an astrology believer with the sign ${person.sign}.
+ Their horoscope for today is:
+ ${horoscope.summary}
+ Given this, use web tools and generate search queries
+ to find $storyCount relevant news stories summarize them in a few sentences.
+ Include the URL for each story.
+ Do not look for another horoscope reading or return results directly about astrology;
+ find stories relevant to the reading above.
+
+ For example:
+ - If the horoscope says that they may
+ want to work on relationships, you could find news stories about
+ novel gifts
+ - If the horoscope says that they may want to work on their career,
+ find news stories about training courses.
+ """.trimIndent()
+ )
+
+ // The @AchievesGoal annotation indicates that completing this action
+ // achieves the given goal, so the agent run will be complete
+ @AchievesGoal(
+ description = "Write an amusing writeup for the target person based on their horoscope and current news stories",
+ )
+ @Action
+ fun writeup(
+ person: StarPerson,
+ relevantNewsStories: RelevantNewsStories,
+ horoscope: Horoscope,
+ ai: Ai,
+ ): Writeup =
+ ai
+ .withLlm(
+ LlmOptions
+ .withModel(model)
+ .withTemperature(0.9)
+ )
+ .createObject(
+ """
+ Take the following news stories and write up something
+ amusing for the target person.
+
+ Begin by summarizing their horoscope in a concise, amusing way, then
+ talk about the news. End with a surprising signoff.
+
+ ${person.name} is an astrology believer with the sign ${person.sign}.
+ Their horoscope for today is:
+ ${horoscope.summary}
+ Relevant news stories are:
+ ${relevantNewsStories.items.joinToString("\n") { "- ${it.url}: ${it.summary}" }}
+
+ Format it as Markdown with links.
+ """.trimIndent()
+ )
+
+}
+```
+
+
+
+
+The following domain classes ensure type safety:
+
+
+Java
+
+```java
+
+@JsonClassDescription("Person with astrology details")
+@JsonDeserialize(as = StarPerson.class)
+public record StarPerson(
+ String name,
+ @JsonPropertyDescription("Star sign") String sign
+) implements Person {
+
+ @JsonCreator
+ public StarPerson(
+ @JsonProperty("name") String name,
+ @JsonProperty("sign") String sign
+ ) {
+ this.name = name;
+ this.sign = sign;
+ }
+
+ @Override
+ public String getName() {
+ return name;
+ }
+}
+
+public record Horoscope(String summary) {
+}
+
+@JsonClassDescription("Writeup relating to a person's horoscope and relevant news")
+public record Writeup(String text) implements HasContent {
+
+ @JsonCreator
+ public Writeup(@JsonProperty("text") String text) {
+ this.text = text;
+ }
+
+ @Override
+ public String getContent() {
+ return text;
+ }
+}
+
+```
+
+
+
+
+Kotlin
+
+```kotlin
+data class RelevantNewsStories(
+ val items: List
+)
+
+data class NewsStory(
+ val url: String,
+
+ val summary: String,
+)
+
+data class Subject(
+ val name: String,
+ val sign: String,
+)
+
+data class Horoscope(
+ val summary: String,
+)
+
+data class FunnyWriteup(
+ override val text: String,
+) : HasContent
+```
+
+
+
+It's easy to unit test your agents to ensure that they correctly execute logic
+and pass the correct prompts and hyperparameters to LLMs. For example:
+
+```java
+public class StarNewsFinderTest {
+
+ @Test
+ void writeupPromptMustContainKeyData() {
+ HoroscopeService horoscopeService = mock(HoroscopeService.class);
+ StarNewsFinder starNewsFinder = new StarNewsFinder(horoscopeService, 5);
+ var context = new FakeOperationContext();
+ context.expectResponse(new com.embabel.example.horoscope.Writeup("Gonna be a good day"));
+
+ NewsStory cockatoos = new NewsStory(
+ "https://fake.com.au",
+ "Cockatoo behavior",
+ "Cockatoos are eating cabbages"
+ );
+
+ NewsStory emus = new NewsStory(
+ "https://morefake.com.au",
+ "Emu movements",
+ "Emus are massing"
+ );
+
+ StarPerson starPerson = new StarPerson("Lynda", "Scorpio");
+ RelevantNewsStories relevantNewsStories = new RelevantNewsStories(Arrays.asList(cockatoos, emus));
+ Horoscope horoscope = new Horoscope("This is a good day for you");
+
+ starNewsFinder.writeup(starPerson, relevantNewsStories, horoscope, context);
+
+ var prompt = context.getLlmInvocations().getFirst().getPrompt();
+ var toolGroups = context.getLlmInvocations().getFirst().getInteraction().getToolGroups();
+
+
+ assertTrue(prompt.contains(starPerson.getName()));
+ assertTrue(prompt.contains(starPerson.sign()));
+ assertTrue(prompt.contains(cockatoos.getSummary()));
+ assertTrue(prompt.contains(emus.getSummary()));
+
+ assertTrue(toolGroups.isEmpty(), "The LLM should not have been given any tool groups");
+ }
+}
+```
+
+## Dog Food Policy
+
+We believe that all aspects of software development and business can and should
+be greatly accelerated through the use of AI agents. The ultimate decision
+makers remain human, but they can and should be greatly augmented.
+
+> This project practices extreme dogfooding.
+
+
+
+Our key principles:
+
+1. **We will use AI agents to help every aspect of the project:** coding, documentation, community management, producing
+ marketing copy etc.
+ Any
+ human performing a task should ask why it cannot be automated, and strive toward maximum automation.
+2. **Developers retain ultimate control.** Developers are responsible for guiding agents toward the solution and
+ iterating
+ as necessary. A developer who commits or merges an agent contribution
+ is responsible for ensuring that it meets the project coding standards, which are
+ independent of the use of agents. For example, code must be human-readable.
+3. **We will favour open source agents built on the Embabel platform,** and contribute improvements. While
+ commercial agents
+ may be more advanced in some areas, we believe that our
+ platform is the best general solution for automation and by dogfooding we will improve it fastest.
+ By open sourcing agents used on our open source projects, we will maximize benefit to the community.
+4. **We will prioritize agents that help accelerate our progress.** Per the flight safety advice to fit your own mask
+ before helping others, we will prioritize
+ agents that help us accelerate our own progress. This will not only produce useful examples, but increase overall
+ project velocity.
+
+Developers must carefully read all code they commit and improve generated code if possible.
+
+> Coding agents are a special case. While the `embabel-agent-code` submodule offers support for project modification
+> that is useful for project bootstrapping, coding agents are the most mature of commercial agents, and their vendors
+> are
+> heavily subsidising their users, making it economically irrational to insist on our own platform.
+
+## Getting Started
+
+- Get the bits
+- Set up your environment
+- Run the application
+
+### Getting the bits
+
+Choose one of the following:
+
+- Clone the repository via `git clone https://github.com/embabel/embabel-agent`
+- Create a new Spring Boot project and add the necessary dependencies (see "Using Embabel Agent Framework in Your
+ Project" below)
+
+### Environment variables
+
+> Environment variables are consistent with common usage, rather than Spring AI.
+> For example, we prefer `OPENAI_API_KEY` to `SPRING_AI_OPENAI_API_KEY`.
+
+Required:
+
+- `OPENAI_API_KEY`: For the OpenAI API
+
+Optional:
+
+- `ANTHROPIC_API_KEY`: For the Anthropic API. Necessary for the coding agent.
+- `MINIMAX_API_KEY`: For the [MiniMax](https://www.minimax.io) API. Supports MiniMax-M3, MiniMax-M2.7 and MiniMax-M2.7-highspeed models.
+- `ZAI_API_KEY`: For the [Z.ai](https://z.ai) (Zhipu AI) API. Supports GLM-5.2, GLM-4.7, GLM-4.6, GLM-4.5-Air and GLM-4.7-Flash models.
+- OCI Generative AI uses OCI SDK authentication providers. Add `embabel-agent-starter-oci-genai` and set
+ `embabel.agent.platform.models.ocigenai.compartment-id`; OCI config file, instance principal, resource principal,
+ workload identity, session token and simple key authentication are supported.
+
+> We strongly recommend providing both an OpenAI and Anthropic key, as some examples require both. And it's important to
+> try to find the best LLM for a given task, rather than automatically choose a familiar provider.
+
+### Services
+
+You will need a Docker Desktop version [`>4.43.2`](https://docs.docker.com/desktop/release-notes/).
+Be sure to activate the following MCP tools from the catalog:
+
+- Brave Search
+- Fetch
+- Puppeteer
+- Wikipedia
+
+> You can also set up your own MCP tools using Spring AI conventions. See the `application-docker-desktop.yml` file for
+> an example.
+
+If you're running Ollama locally, include the `embabel ollama starter` and Embabel will automatically connect to your
+Ollama
+endpoint and make all models available.
+
+```xml
+
+
+ com.embabel.agent
+ embabel-agent-starter-ollama
+
+```
+
+### Running
+
+Create your own agent project with
+
+```
+uvx --from git+https://github.com/embabel/project-creator.git project-creator
+```
+
+### Example Agents
+
+> **📚 For examples and tutorials**, see
+> the [Embabel Agent Examples Repository](https://github.com/embabel/embabel-agent-examples)
+
+```bash
+# Clone and run examples
+git clone https://github.com/embabel/embabel-agent-examples
+cd embabel-agent-examples/scripts/kotlin
+./shell.sh
+```
+
+#### Shell Commands
+
+Spring Shell is an easy way to interact with the Embabel agent framework, especially during development.
+
+Type `help` to see available commands. Use `execute` or `x` to run an agent:
+
+```
+execute "Lynda is a Scorpio, find news for her" -p -r
+```
+
+This will look for an agent, choose the star finder agent and
+run the flow. `-p` will log prompts `-r` will log LLM responses.
+Omit these for less verbose logging.
+
+Options:
+
+- `-p` logs prompts
+- `-r` logs LLM responses
+
+Use the `chat` command to enter an interactive chat with the agent.
+It will attempt to run the most appropriate
+agent for each command.
+
+> Spring Shell supports history. Type `!!` to repeat the last command.
+> This will survive restarts, so is handy when iterating on an agent.
+
+#### Further examples
+
+Example commands within the shell:
+
+```
+# Perplexity style deep research
+# Requires both OpenAI and Anthropic keys and Docker Desktop with the MCP extension (or your own web tools)
+execute "research the recent australian federal election. what is the position of the greens party?"
+
+# x is a shortcut for execute
+x "fact check the following: holden cars are still made in australia; the koel is a bird native only to australia; fidel castro is justin trudeau's father"
+
+```
+
+### Bringing in additional LLMs
+
+#### Local models with well-known providers
+
+The Embabel Agent Framework supports local models from:
+
+- Ollama: Simply add `embabel-agent-starter-ollama` starter to your pom.xml and your local Ollama endpoint will be
+ queries. All local models will be
+ available.
+- Docker: Add the `embabel-agent-starter-dockermodels` starter to your pom.xml and your local Docker endpoint will be
+ queried. All local models will be available.
+- LMStudio: This uses the openAI compatible client. Just include LMStudio as a dependency and make sure your LMStudio
+ server is running.
+
+#### OCI Generative AI
+
+Add `embabel-agent-starter-oci-genai` to use OCI Generative AI chat and embedding models.
+
+```xml
+
+ com.embabel.agent
+ embabel-agent-starter-oci-genai
+
+```
+
+Configure `embabel.agent.platform.models.ocigenai.compartment-id` and, if needed, set
+`embabel.agent.platform.models.ocigenai.authentication-type` to `FILE`, `INSTANCE_PRINCIPAL`, `RESOURCE_PRINCIPAL`,
+`WORKLOAD_IDENTITY`, `SESSION_TOKEN` or `SIMPLE`. When the standard OpenAI provider is not on the classpath, the OCI
+starter supplies OCI defaults for Embabel's default LLM and embedding model:
+
+```properties
+embabel.models.default-llm=cohere.command-a-03-2025
+embabel.models.default-embedding-model=cohere.embed-v4.0
+```
+
+Override those values in application configuration if you want another OCI model. Use OCI model ids such as
+`cohere.command-a-03-2025` or `meta.llama-3.3-70b-instruct` for Embabel model selection.
+The Spring bean names registered by the starter are Java-friendly aliases such as `cohere_command_a` and
+`llama_33_70b`.
+If your application exposes Spring Boot Actuator `env` or `configprops` values, keep those endpoints secured and ensure
+OCI credential fields such as `pass-phrase`, `session-token` and `private-key` are sanitized.
+
+#### Custom LLMs
+
+You can define an LLM for any provider for which a Spring AI `ChatModel` is available.
+
+Simply define Spring beans of type `Llm`.
+See the `OpenAiConfiguration` class as an example.
+
+Remember:
+
+- Provide the knowledge cutoff date if you know it
+- Make the configuration class conditional on any required API key.
+
+## Roadmap
+
+This project is in its early stages, but we have big plans.
+The milestones and issues in this repository are a good reference.
+Our key goals:
+
+- **Become the natural way to Gen AI-enable Java applications**, and especially those built on Spring.
+- **Prove the power of the approach**. Demonstrate that this approach is the best way to build safe, dependable, Gen AI
+ applications.
+ In particular:
+ - Demonstrate the power of extensibility without modification, by adding goals and actions
+ - Demonstrate the potential to become the PaaS for natural language
+ - Demonstrate the potential of agent federation within the GOAP model
+ - Demonstrate budget-aware agents, such as "Research the following topic, spending up to 20c if you are still
+ learning"
+ - Integrate with data stores and demonstrate the power of surfacing existing functionality inside an organization
+- **Take the model to other platforms**: The conceptual framework is not JVM specific. Once established, we intend to
+ create TypeScript
+ and Python projects.
+
+There is a lot to do, and you are awesome. We look forward to your contribution!
+
+## Application Design
+
+### Domain objects
+
+Applications center around domain objects. These can be instantiated by LLMs or user
+code, and manipulated by user code.
+
+Use Jackson annotations to help LLMs with descriptions as well as mark fields to ignore.
+For example:
+
+```kotlin
+@JsonClassDescription("Person with astrology details")
+data class StarPerson(
+ override val name: String,
+ @get:JsonPropertyDescription("Star sign")
+ val sign: String,
+) : Person
+```
+
+See [Java Json Schema Generation - Module Jackson](https://github.com/victools/jsonschema-generator/tree/main/jsonschema-module-jackson)
+for documentation of the library used.
+
+Domain objects can have behaviors that are automatically exposed to LLMs when they are in scope. Simply annotate methods
+with the Spring AI `@Tool` annotation.
+
+> When exposing `@Tool` methods on domain objects, be sure that the tool is safe to invoke. Even the best LLMs can get
+> trigger-happy. For example, be careful about methods that can mutate or delete data. This is likely better modeled via
+> an explicit call to a non-tool method on the same domain class, in a code action.
+
+## Using Embabel as an MCP server
+
+You can use the Embabel agent platform as an MCP server from a
+UI like Claude Desktop. The Embabel MCP server is available over SSE.
+
+Configure Claude Desktop as follows in your `claude_desktop_config.yml`:
+
+```json
+{
+ "mcpServers": {
+ "embabel": {
+ "command": "npx",
+ "args": [
+ "-y",
+ "mcp-remote",
+ "http://localhost:8080/sse"
+ ]
+ }
+ }
+}
+
+```
+
+See [MCP Quickstart for Claude Desktop Users](https://modelcontextprotocol.io/quickstart/user) for how to configure
+Claude Desktop.
+
+The [MCP Inspector](https://github.com/modelcontextprotocol/inspector) is a helpful tool for interacting with your
+Embabel
+SSE server, manually invoking tools and checking the exposed prompts and resources.
+
+Start the MCP Inspector with:
+
+```bash
+npx @modelcontextprotocol/inspector
+```
+
+## Consuming MCP Servers
+
+The Embabel Agent Framework provides built-in support for consuming Model Context Protocol (MCP) servers, allowing you
+to extend your applications with powerful AI capabilities through standardized interfaces.
+
+### What is MCP?
+
+Model Context Protocol (MCP) is an open protocol that standardizes how applications provide context and extra
+functionality to large language models. Introduced by Anthropic, MCP has emerged as the de facto standard for connecting
+AI agents to tools, functioning as a client-server protocol where:
+
+- **Clients** (like Embabel Agent) send requests to servers
+- **Servers** process those requests to deliver necessary context to the AI model
+
+MCP simplifies integration between AI applications and external tools, transforming an "M×N problem" into an "M+N
+problem" through standardization - similar to what USB did for hardware peripherals.
+
+### Configuring MCP in Embabel Agent
+
+To configure MCP servers in your Embabel Agent application, add the following to your `application.yml`:
+
+```yaml
+spring:
+ ai:
+ mcp:
+ client:
+ enabled: true
+ name: embabel
+ version: 1.0.0
+ request-timeout: 30s
+ type: SYNC
+ stdio:
+ connections:
+ docker-mcp:
+ command: docker
+ args:
+ - run
+ - -i
+ - --rm
+ - alpine/socat
+ - STDIO
+ - TCP:host.docker.internal:8811
+```
+
+This configuration sets up an MCP client that connects to a Docker-based MCP server. The connection uses STDIO transport
+through Docker's socat utility to connect to a TCP endpoint.
+
+### Docker Desktop MCP Integration
+
+Docker has embraced MCP with their Docker MCP Catalog and Toolkit, which provides:
+
+1. **Centralized Discovery** - A trusted hub for discovering MCP tools integrated into Docker Hub
+2. **Containerized Deployment** - Run MCP servers as containers without complex setup
+3. **Secure Credential Management** - Centralized, encrypted credential handling
+4. **Built-in Security** - Sandbox isolation and permissions management
+
+The Docker MCP ecosystem includes over 100 verified tools from partners like Stripe, Elastic, Neo4j, and more, all
+accessible through Docker's infrastructure.
+
+### Learn More
+
+- [Docker MCP Documentation](https://docs.docker.com/desktop/features/gordon/mcp/)
+- [Docker MCP Servers Repository](https://github.com/docker/mcp-servers)
+- [Introducing Docker MCP Catalog and Toolkit](https://www.docker.com/blog/introducing-docker-mcp-catalog-and-toolkit/)
+- [MCP Introduction and Overview](https://www.philschmid.de/mcp-introduction)
+
+## A2A
+
+Embabel integrates with the [A2A](https://github.com/google-a2a/A2A) protocol, allowing you to connect to other
+A2A-enabled agents and
+services.
+
+Enable the `a2a` Spring profile to start the A2A server.
+
+You'll need the following environment variable:
+
+- `GOOGLE_STUDIO_API_KEY`: Your Google Studio API key, which is used for Gemini.
+
+Start the Google A2A web interface using the `a2a` Docker profile:
+
+```bash
+docker compose --profile a2a up
+```
+
+Go to the web interface running within the container at `http://localhost:12000/`.
+
+Connect to your agent at `host.docker.internal:8080/a2a`. Note that `localhost:8080/a2a` won't work as the server
+cannot access it when running in a Docker container.
+
+## Running Tests
+
+### Unit tests
+
+Run the unit tests via Maven. This will not require an internet connection or any external services.
+
+```bash
+mvn test
+```
+
+### Integration tests
+
+Integration tests (`*IT`) hit real provider APIs and are excluded from the default `mvn test` run.
+To run them, ensure the following environment variables are set:
+
+- `OPENAI_API_KEY`
+- `ANTHROPIC_API_KEY`
+- `DEEPSEEK_API_KEY`
+- `MISTRAL_API_KEY`
+
+Then run:
+
+```bash
+mvn -Dtest='*IT,!LLMOllama*IT' -Dsurefire.failIfNoSpecifiedTests=false test
+```
+
+This runs all integration tests except Ollama (which requires a local Ollama server).
+To run a specific module's integration tests, add `-pl`:
+
+```bash
+mvn -Dtest='*IT,!LLMOllama*IT' -Dsurefire.failIfNoSpecifiedTests=false test -pl embabel-agent-openai
+```
+
+## Spring profiles
+
+Spring profiles are used to configure the application for different environments and behaviors.
+
+Model profiles:
+
+- `docker-desktop`: Talking to Docker Desktop with the MCP extension. **This is recommended for the best experience,
+ with Docker-provided web tools.**
+
+Logging profiles:
+
+- `severance`: [Severance](https://www.youtube.com/watch?v=xEQP4VVuyrY&ab_channel=AppleTV) specific logging. Praise
+ Kier!
+- `starwars`: Star Wars specific logging. Feel the force
+- `colossus`: Colossus specific logging. The Forbin Project.
+
+## Testing
+
+A key goal of this framework is ease of testing.
+Just as Spring eased testing of early enterprise Java applications,
+this framework facilitates testing of AI applications.
+
+Types of testing:
+
+- Unit tests: All agents are unit testable, like any Spring-managed beans. Construct them with mock objects; call
+ individual action methods. The testing library facilitates testing prompts.
+- Integration tests: tbd
+
+## Logging
+
+All logging in this project is either debug logging in the relevant
+class itself, or results from the stream of events of type `AgentEvent`.
+
+Edit `application.yml` if you want to see debug logging from the relevant classes and packages.
+
+Available logging experiences:
+
+- `severance`: Severance logging. Praise Kier
+- `starwars`: Star Wars logging. Feel the force. The default as it's understood throughout the galaxy.
+- `colossus`: Colossus logging. The Forbin Project.
+- `montypython`: Monty Python logging. No one expects it.
+- `hh`: Hitchhiker's Guide to the Galaxy logging. The answer is 42.
+
+If none of these profiles is chosen, Embabel will use vanilla logging.
+This makes me sad.
+
+## Adding Embabel Agent Framework to Your Project
+
+### Maven Central Availability
+
+**Since version 0.2.0**, Embabel Agent Framework is available directly on Maven Central, simplifying dependency
+management. You no longer need to configure custom repositories for stable releases.
+
+---
+
+### Maven
+
+#### For version 0.2.0 and above (Recommended)
+
+Simply add the Embabel Spring Boot starter dependency to your `pom.xml`:
+
+```xml
+
+
+ com.embabel.agent
+ embabel-agent-starter
+ ${embabel-agent.version}
+
+```
+
+No additional repository configuration is needed! Maven Central is configured by default.
+
+#### For versions prior to 0.2.0 or for SNAPSHOT versions
+
+You need to add the Embabel repositories to your `pom.xml`:
+
+```xml
+
+
+
+ embabel-releases
+ https://repo.embabel.com/artifactory/libs-release
+
+ true
+
+
+ false
+
+
+
+ embabel-snapshots
+ https://repo.embabel.com/artifactory/libs-snapshot
+
+ false
+
+
+ true
+
+
+
+```
+
+Then add the dependency:
+
+```xml
+
+
+ com.embabel.agent
+ embabel-agent-starter
+ ${embabel-agent.version}
+
+```
+
+---
+
+### Gradle (Kotlin DSL)
+
+#### For version 0.2.0 and above (Recommended)
+
+Add the required repositories to your `build.gradle.kts`:
+
+```kotlin
+repositories {
+ mavenCentral()
+ maven {
+ name = "Spring Milestones"
+ url = uri("https://repo.spring.io/milestone")
+ }
+}
+```
+
+Add the Embabel Agent starter:
+
+```kotlin
+dependencies {
+ implementation("com.embabel.agent:embabel-agent-starter:${embabelAgentVersion}")
+}
+```
+
+#### For versions prior to 0.2.0 or for SNAPSHOT versions
+
+Add all required repositories to your `build.gradle.kts`:
+
+```kotlin
+repositories {
+ mavenCentral()
+ maven {
+ name = "embabel-releases"
+ url = uri("https://repo.embabel.com/artifactory/libs-release")
+ mavenContent {
+ releasesOnly()
+ }
+ }
+ maven {
+ name = "embabel-snapshots"
+ url = uri("https://repo.embabel.com/artifactory/libs-snapshot")
+ mavenContent {
+ snapshotsOnly()
+ }
+ }
+ maven {
+ name = "Spring Milestones"
+ url = uri("https://repo.spring.io/milestone")
+ }
+}
+```
+
+Add the Embabel Agent starter:
+
+```kotlin
+dependencies {
+ implementation("com.embabel.agent:embabel-agent-starter:${embabelAgentVersion}")
+}
+```
+
+---
+
+### Gradle (Groovy DSL)
+
+#### For version 0.2.0 and above (Recommended)
+
+Add the required repositories to your `build.gradle`:
+
+```groovy
+repositories {
+ mavenCentral()
+ maven {
+ name = 'Spring Milestones'
+ url = 'https://repo.spring.io/milestone'
+ }
+}
+```
+
+Add the Embabel Agent starter:
+
+```groovy
+dependencies {
+ implementation "com.embabel.agent:embabel-agent-starter:${embabelAgentVersion}"
+}
+```
+
+#### For versions prior to 0.2.0 or for SNAPSHOT versions
+
+Add all required repositories to your `build.gradle`:
+
+```groovy
+repositories {
+ mavenCentral()
+ maven {
+ name = 'embabel-releases'
+ url = 'https://repo.embabel.com/artifactory/libs-release'
+ mavenContent {
+ releasesOnly()
+ }
+ }
+ maven {
+ name = 'embabel-snapshots'
+ url = 'https://repo.embabel.com/artifactory/libs-snapshot'
+ mavenContent {
+ snapshotsOnly()
+ }
+ }
+ maven {
+ name = 'Spring Milestones'
+ url = 'https://repo.spring.io/milestone'
+ }
+}
+```
+
+Add the Embabel Agent starter:
+
+```groovy
+dependencies {
+ implementation 'com.embabel.agent:embabel-agent-starter:${embabelAgentVersion}'
+}
+```
+
+---
+
+### Important Notes
+
+#### Spring Milestones Repository
+
+The Spring Milestones repository is required because the Embabel BOM (`embabel-agent-dependencies`) has transitive
+dependencies on experimental Spring components, specifically the `mcp-bom`. This BOM is not available on Maven Central
+and is only published to the Spring milestone repository.
+
+**Note for Gradle users:** Unlike Maven, Gradle does not inherit repository configurations declared in parent POMs or
+BOMs. Therefore, it is necessary to explicitly declare the Spring milestone repository in your repositories block to
+ensure proper resolution of all transitive dependencies.
+
+#### Repository Types
+
+- **Maven Central** (since v0.2.0): For stable releases 0.2.0 and above
+- **Embabel Releases Repository**: For stable releases prior to 0.2.0
+- **Embabel Snapshots Repository**: For development/snapshot versions (e.g., `0.3.0-SNAPSHOT`)
+
+---
+
+### Quick Start Examples
+
+#### Maven with latest stable version
+
+```xml
+
+
+ com.embabel.agent
+ embabel-agent-starter
+ 0.3.0
+
+```
+
+#### Gradle Kotlin DSL with latest stable version
+
+```kotlin
+implementation("com.embabel.agent:embabel-agent-starter:0.3.0")
+```
+
+#### Gradle Groovy DSL with latest stable version
+
+```groovy
+implementation 'com.embabel.agent:embabel-agent-starter:0.3.0'
+```
+
+## Getting Started with Observability
+
+Add full tracing and metrics to your Embabel agents with zero code changes.
+
+### 1. Add the dependency
+
+```xml
+
+ com.embabel.agent
+ embabel-agent-starter-observability
+ ${embabel-agent.version}
+
+```
+
+### 2. Add an exporter
+
+Pick **one** (or combine multiple):
+
+**Zipkin** (simplest — no account needed):
+```xml
+
+ io.opentelemetry
+ opentelemetry-exporter-zipkin
+
+```
+
+**Langfuse** (LLM-focused observability):
+```xml
+
+ com.quantpulsar
+ opentelemetry-exporter-langfuse
+ 0.4.0
+
+```
+
+### 3. Configure
+
+```yaml
+# Enable observability
+embabel:
+ agent:
+ platform:
+ observability:
+ enabled: true
+ service-name: my-agent-app
+
+# Enable Spring Boot tracing
+management:
+ tracing:
+ export:
+ enabled: true
+ sampling:
+ probability: 1.0
+
+ # Zipkin exporter
+ zipkin:
+ tracing:
+ endpoint: http://localhost:9411/api/v2/spans
+```
+
+To use Langfuse instead of (or alongside) Zipkin:
+```yaml
+management:
+ langfuse:
+ enabled: true
+ endpoint: https://cloud.langfuse.com/api/public/otel # or your self-hosted URL
+ public-key: pk-lf-...
+ secret-key: sk-lf-...
+```
+
+### 4. Start Zipkin and run
+
+```bash
+docker run -d -p 9411:9411 openzipkin/zipkin
+./mvnw spring-boot:run
+```
+
+Open [http://localhost:9411](http://localhost:9411) — run an agent and you will see traces like:
+
+```
+Agent: CustomerServiceAgent
+├── Action: AnalyzeRequest
+│ └── ChatModel: gpt-4 (Spring AI)
+│ └── tool:searchKnowledgeBase
+├── Action: GenerateResponse
+│ └── ChatModel: gpt-4 (Spring AI)
+└── status: completed [duration=2340ms]
+```
+
+With Langfuse, you get a rich LLM-focused view of your agent traces:
+
+
+
+### What gets traced automatically
+
+- Agent lifecycle (creation, execution, completion, failures)
+- Every action as a child span
+- LLM calls with token usage (via Spring AI)
+- Tool invocations with input/output
+- Planning and replanning iterations
+- State transitions and lifecycle states
+
+### Track custom operations with `@Tracked`
+
+Use the `@Tracked` annotation to add observability spans to your own methods — inputs, outputs, duration, and errors are captured automatically:
+
+```java
+@Tracked("enrichCustomer")
+public Customer enrich(Customer input) {
+ // Your logic here
+}
+```
+
+You can specify a type and description for richer traces:
+
+```java
+@Tracked(value = "callPaymentApi", type = TrackType.EXTERNAL_CALL, description = "Payment gateway call")
+public PaymentResult processPayment(Order order) {
+ // ...
+}
+```
+
+When called within an agent execution, `@Tracked` spans are automatically nested under the current action:
+
+```
+Agent: CustomerServiceAgent
+├── Action: ProcessOrder
+│ ├── @Tracked: enrichCustomer (PROCESSING)
+│ ├── ChatModel: gpt-4
+│ └── @Tracked: callPaymentApi (EXTERNAL_CALL)
+└── status: completed
+```
+
+For the full configuration reference, MDC log correlation, and advanced options, see the [Observability Module Documentation](embabel-agent-observability/README.md).
+
+---
+
+## Contributing
+
+We welcome contributions to the Embabel Agent Framework.
+
+Look at the [coding style guide](embabel-agent-api/.embabel/coding-style.md) for style guidelines.
+This file also informs coding agent behavior.
+
+## Miscellaneous
+
+- _Why the name Embabel?_
+ The "babel" part is ultimately inspired by the story of the Tower of Babel, perhaps via Douglas
+ Adams' [babelfish](https://www.youtube.com/watch?v=iuumnjJWFO4&ab_channel=BBCStudios).
+ Per @lasuac:
+ _While Adams' fish in the ear enabled universal translation between species, Embabel aims at translating human intent
+ to JVM code, AI models, and enterprise systems._
+ "embabel" also sounds like "enable."
+- Milestone names are Australian animals. Mythical animals such as "bunyip" and "yowie" are used for futures that may or
+ not be implemented.
+- README badges come from [here](https://github.com/Ileriayo/markdown-badges)
+ and [here](https://home.aveek.io/GitHub-Profile-Badges/).
+- Don't forget to join [Discord](https://discord.gg/t6bjkyj93q) to collaborate with the Embabel community. It is a good
+ place to receive support, showcase your work, discuss ideas and connect with like-minded people.
+
+## Star History
+
+
+
+
+
+
+
+
+
+## Contributors
+
+[](https://github.com/embabel/embabel-agent/graphs/contributors)
+
+
+
+--------------------
+(c) Embabel Software Inc 2024-2026.
diff --git a/.verify/embabel_embabel-agent_README_alt.md b/.verify/embabel_embabel-agent_README_alt.md
new file mode 100644
index 0000000..bf6303c
--- /dev/null
+++ b/.verify/embabel_embabel-agent_README_alt.md
@@ -0,0 +1,1437 @@
+# [Embabel Agent Framework](https://hub.embabel.com)
+
+
+
+[](https://docs.embabel.com/embabel-agent/guide/1.5.0-SNAPSHOT/)
+[](https://central.sonatype.com/artifact/com.embabel.agent/embabel-agent-api)
+
+[](https://www.yourkit.com/)
+[](https://www.ej-technologies.com/products/jprofiler/overview.html)
+[](https://sonarcloud.io/summary/new_code?id=embabel_embabel-agent)
+[](https://discord.gg/t6bjkyj93q)
+
+[//]: # ([](https://sonarcloud.io/summary/new_code?id=embabel_embabel-agent))
+
+[//]: # ([](https://sonarcloud.io/summary/new_code?id=embabel_embabel-agent))
+
+
+
+
+
+
+
+
+
+
+
+
+
+
+[](https://www.apache.org/licenses/LICENSE-2.0)
+[](https://github.com/embabel/embabel-agent/pulse)
+
+
+
+Embabel (Em-BAY-bel) is a framework for authoring agentic flows on the JVM that seamlessly mix LLM-prompted interactions
+with code and domain models. Supports
+intelligent path finding towards goals. Written in Kotlin
+but offers a natural usage
+model from Java.
+From the creator of Spring.
+
+
+
+## Talk to the Docs
+
+Have questions? [Talk to the docs via the Embabel-powered hub](https://hub.embabel.com) — an
+Embabel agent that answers your questions about the framework in natural language.
+
+## Key Concepts
+
+Models agentic flows in terms of:
+
+- **Actions**: Steps an agent takes
+- **Goals**: What an agent is trying to achieve
+- **Conditions**: Conditions to assess before executing an action or determining that a goal has been achieved.
+ Conditions are reassessed after each action is executed.
+- **Domain model**: Objects underpinning the flow and informing Actions, Goals and Conditions.
+- **Plan**: A sequence of actions to achieve a goal. Plans are dynamically formulated by the system, not the programmer.
+ The
+ system replans after the completion of each action, allowing it to adapt to new information as well as observe the
+ effects of the previous action.
+ This is effectively an [OODA loop](https://en.wikipedia.org/wiki/OODA_loop).
+
+> Application developers don't usually have to deal with these concepts directly,
+> as most conditions result from data flow defined in code, allowing the system to infer
+> pre and post conditions.
+
+These concepts underpin these differentiators versus other agent frameworks:
+
+- **Sophisticated planning.** Goes beyond a finite state machine or sequential execution
+ with nesting by introducing a true planning step, using a
+ non-LLM AI algorithm. This enables the system to perform tasks it wasn’t programmed to do by combining known
+ steps in
+ a novel order, as well as make decisions about parallelization and other runtime behavior.
+- **Superior extensibility and reuse**: Because of dynamic planning, adding more domain objects, actions, goals and
+ conditions
+ can extend the capability of the system, _without editing FSM definitions_ or existing code.
+- **Strong typing and the benefits of object orientation**: Actions, goals and conditions are informed by a domain
+ model, which can
+ include behavior. Everything is strongly typed and prompts and
+ manually authored code interact cleanly. No more magic maps. Enjoy full refactoring support.
+
+Other benefits:
+
+- **Platform abstraction**: Clean separation between programming model and platform internals allows running locally
+ while
+ potentially offering higher QoS in production without changing application code.
+- **Designed for LLM mixing**: It is easy to build applications that mix LLMs, ensuring the most cost-effective yet
+ capable solution.
+ This enables the system to leverage the strengths of different models for different tasks. In particular, it
+ facilitates
+ the use of local models for point tasks. This can be important for cost and privacy.
+- **Built on Spring and the JVM,** making it easy to access existing enterprise functionality and capabilities.
+ For example:
+ - Spring can inject and manage agents, including using Spring AOP to decorate functions.
+ - Robust persistence and transaction management solutions are available.
+- **Designed for testability** from the ground up. Both unit testing and agent end to end testing are easy.
+
+Flows can be authored in one of two ways:
+
+- An annotation-based model similar to Spring MVC, with types annotated with the Spring stereotype `@Agent`, using
+ `@Goal`, `@Condition` and
+ `@Action` methods.
+- Idiomatic Kotlin DSL with `agent {` and `action {` blocks.
+
+Either way, flows are backed by a domain model of objects that can have rich behavior.
+
+> We are working toward allowing natural language actions and goals to be deployed.
+
+The planning step is pluggable.
+
+The default planning approach is
+[Goal Oriented Action Planning](https://medium.com/@vedantchaudhari/goal-oriented-action-planning-34035ed40d0b).
+GOAP is a popular AI planning algorithm used in gaming. It allows for dynamic decision-making and action selection based
+on the current state of the world and the goals of the agent.
+
+Goals, actions and plans are independent of GOAP. Embabel also
+supports [Utility AI](https://en.wikipedia.org/wiki/Utility_system) out of the box, which can run the same actions but
+chooses actions
+based on (potentially dynamic) utility scores rather than strict preconditions and postconditions. This is valuable for
+exploration and
+open-ended tasks, when we do not need to achieve a specific goal but want to maximize overall utility.
+
+The framework executes via an `AgentPlatform` implementation.
+
+An agent platform supports the following modes of execution:
+
+- **Focused**, where user code requests particular functionality: User code calls a method to run a particular agent,
+ passing in input. This is ideal for code-driven flows such as a flow invoked in response to an incoming event.
+- **Closed**, where user intent (or another incoming event) is classified to choose an agent. The platform tries to
+ find a
+ suitable agent among all the agents it knows about.
+ Agent choice is dynamic, but only actions defined within the particular agent
+ will run.
+- **Open**, where the user's intent is assessed and the platform uses _all_ its resources to try to achieve it. The
+ platform tries to find a
+ suitable goal among all the goals it knows about and builds a custom agent to achieve it from the start state,
+ including relevant actions and conditions. The platform will not proceed if it is unconvinced as to the applicability
+ of any goal. The `GoalChoiceApprover` interface provides developers a way to limit goal choice further.
+
+Open mode is the most powerful, but least deterministic.
+> In open mode, the platform is capable of finding novel paths that were not envisioned by developers, and even
+> combining functionality from multiple providers.
+
+Even in open mode, the platform will only perform individual steps
+that have been specified. (Of course, steps may themselves be LLM
+transforms, in which case the prompts are controlled by user code but the
+results are still non-deterministic.)
+
+Possible future modes:
+
+- **Evolving** mode: Where the platform can work with multiple goals in the same process and modify a running process to
+ add further goals and agents.
+ For example, an action can realize that it has become important to achieve additional goals.
+
+Embabel agent systems will also support federation, both with other Embabel systems (allowing planning to incorporate
+remote actions and goals) and third party agent frameworks.
+
+## Quick Start
+
+Get an agent running in under 5 minutes.
+
+Create your own agent repo from our [Java](https://github.com/embabel/java-agent-template)
+or [Kotlin](https://github.com/embabel/kotlin-agent-template) GitHub template by clicking the "Use this template"
+button.
+
+You'll have an agent running in under a minute
+if you already have an `OPENAI_API_KEY` and have Maven installed.
+
+**📚 For examples and tutorials**, see
+the [Embabel Agent Examples Repository](https://github.com/embabel/embabel-agent-examples)
+
+**🚗 For a sophisticated, realistic example application**, see
+the [Tripper travel planner agent](https://github.com/embabel/tripper)
+
+
+
+*AI-generated travel itinerary with detailed recommendations*
+
+
+
+*Map link included in output*
+
+## Why Is Embabel Needed?
+
+TL;DR Because the evolution of agent frameworks is early and there's a lot of room for improvement; because an agent
+framework on the JVM will deliver great business value.
+
+- _Why do we need an agent framework at all_? We can write code without higher level abstractions, directly invoking
+ LLMs and controlling flow directly in code. However, a higher level agent framework offers compelling benefits. For
+ example:
+ - Breaking up LLM interactions, making them simpler and more focused. This maximizes reuse and minimizes cost and
+ errors. It often allows us to use cheaper models for point interactions.
+ - Facilitating both unit and integration testing, which remain as important with agentic systems as with any other
+ software systems.
+ - Increasing composability where subflows and individual actions can be reused
+ - Making applications more manageable and robust, enabling a workflow manager to control their execution and retry
+ operations while maintaining previous state
+ - Enhancing safety through the ability to apply guardrails in many places
+- _Why do we need an agent framework for the JVM when solutions exist in Python?_: While agent frameworks initially
+ appeared predominantly Python, it's early and there's plenty of room for novel and
+ superior
+ approaches. The key adjacency is not the LLM--which is a simple HTTP call away--but existing code and
+ infrastructure
+ assets that are more valuable on the JVM than in Python.
+- _Why not use just Spring AI?_ Spring AI is great. We build on it, and embrace the Spring component model. However, we
+ believe that most applications should work with higher
+ level APIs. An analogy: Spring AI exists at the level of the Servlet API, while Embabel is more like Spring MVC.
+ Complex requirements are much easier to express and test in Embabel than with direct use of Spring AI.
+- _Why not attempt to contribute this project to Spring?_ This project requires different governance
+ from Spring, where most projects exist in stable environments and dependability and stability outweighs rapid
+ innovation. Second, the
+ concepts are not JVM-specific. We hope that Embabel will become the leading agent framework across platforms. While
+ the Spring brand is valuable in Java, it is not in TypeScript or Python.
+
+## Show Me The Code
+
+In Java or Kotlin, agent implementation code is intuitive and easy to test.
+
+
+Java
+
+```java
+
+@Agent(description = "Find news based on a person's star sign")
+public class StarNewsFinder {
+
+ private final HoroscopeService horoscopeService;
+ private final int storyCount;
+
+ // Services are injected by Spring
+ public StarNewsFinder(
+ HoroscopeService horoscopeService,
+ @Value("${star-news-finder.story.count:5}") int storyCount) {
+ this.horoscopeService = horoscopeService;
+ this.storyCount = storyCount;
+ }
+
+ @Action
+ public StarPerson extractStarPerson(UserInput userInput, Ai ai) {
+ return ai
+ .withLlm(OpenAiModels.GPT_41)
+ .createObjectIfPossible(
+ """
+ Create a person from this user input, extracting their name and star sign:
+ %s""".formatted(userInput.getContent()),
+ StarPerson.class
+ );
+ }
+
+ @Action
+ public Horoscope retrieveHoroscope(StarPerson starPerson) {
+ return new Horoscope(horoscopeService.dailyHoroscope(starPerson.sign()));
+ }
+
+ // toolGroups specifies tools that are required for this action to run
+ @Action(toolGroups = {CoreToolGroups.WEB})
+ public RelevantNewsStories findNewsStories(
+ StarPerson person,
+ Horoscope horoscope,
+ Ai ai) {
+ var prompt = """
+ %s is an astrology believer with the sign %s.
+ Their horoscope for today is:
+ %s
+ Given this, use web tools and generate search queries
+ to find %d relevant news stories summarize them in a few sentences.
+ Include the URL for each story.
+ Do not look for another horoscope reading or return results directly about astrology;
+ find stories relevant to the reading above.
+
+ For example:
+ - If the horoscope says that they may
+ want to work on relationships, you could find news stories about
+ novel gifts
+ - If the horoscope says that they may want to work on their career,
+ find news stories about training courses.""".formatted(
+ person.name(), person.sign(), horoscope.summary(), storyCount);
+ return ai
+ .withDefaultLlm()
+ .createObject(prompt, RelevantNewsStories.class);
+ }
+
+ // The @AchievesGoal annotation indicates that completing this action
+ // achieves the given goal, so the agent can be complete
+ @AchievesGoal(
+ description = "Write an amusing writeup for the target person based on their horoscope and current news stories",
+ export = @Export(
+ remote = true,
+ name = "starNewsWriteupJava",
+ startingInputTypes = {StarPerson.class, UserInput.class})
+ )
+ @Action
+ public Writeup writeup(
+ StarPerson person,
+ RelevantNewsStories relevantNewsStories,
+ Horoscope horoscope,
+ Ai ai) {
+ var llm = LlmOptions
+ .withModel(OpenAiModels.GPT_41_MINI)
+ // High temperature for creativity
+ .withTemperature(0.9);
+
+ var newsItems = relevantNewsStories.getItems().stream()
+ .map(item -> "- " + item.getUrl() + ": " + item.getSummary())
+ .collect(Collectors.joining("\n"));
+
+ var prompt = """
+ Take the following news stories and write up something
+ amusing for the target person.
+
+ Begin by summarizing their horoscope in a concise, amusing way, then
+ talk about the news. End with a surprising signoff.
+
+ %s is an astrology believer with the sign %s.
+ Their horoscope for today is:
+ %s
+ Relevant news stories are:
+ %s
+
+ Format it as Markdown with links.""".formatted(
+ person.name(), person.sign(), horoscope.summary(), newsItems);
+ return ai
+ .withLlm(llm)
+ .createObject(prompt, Writeup.class);
+ }
+}
+```
+
+
+
+
+Kotlin
+
+```kotlin
+@Agent(description = "Find news based on a person's star sign")
+class StarNewsFinder(
+ // Services such as Horoscope are injected by Spring
+ private val horoscopeService: HoroscopeService,
+ // Potentially externalized by Spring
+ @param:Value("\${star-news-finder.story.count:5}")
+ private val storyCount: Int = 5,
+) {
+
+ @Action
+ fun extractPerson(
+ userInput: UserInput,
+ ai: Ai
+ ): StarPerson =
+ // All prompts are typesafe
+ ai.withDefaultLlm()
+ .createObject("Create a person from this user input, extracting their name and star sign: $userInput")
+
+ // This action doesn't use an LLM
+ // Embabel makes it easy to mix LLM use with regular code
+ @Action
+ fun retrieveHoroscope(starPerson: StarPerson) =
+ Horoscope(horoscopeService.dailyHoroscope(starPerson.sign))
+
+ // This action uses tools
+ // "toolGroups" specifies tools that are required for this action to run
+ @Action(toolGroups = [ToolGroup.WEB])
+ fun findNewsStories(
+ person: StarPerson,
+ horoscope: Horoscope,
+ ai: Ai,
+ ): RelevantNewsStories =
+ ai.withDefaultLlm().createObject(
+ """
+ ${person.name} is an astrology believer with the sign ${person.sign}.
+ Their horoscope for today is:
+ ${horoscope.summary}
+ Given this, use web tools and generate search queries
+ to find $storyCount relevant news stories summarize them in a few sentences.
+ Include the URL for each story.
+ Do not look for another horoscope reading or return results directly about astrology;
+ find stories relevant to the reading above.
+
+ For example:
+ - If the horoscope says that they may
+ want to work on relationships, you could find news stories about
+ novel gifts
+ - If the horoscope says that they may want to work on their career,
+ find news stories about training courses.
+ """.trimIndent()
+ )
+
+ // The @AchievesGoal annotation indicates that completing this action
+ // achieves the given goal, so the agent run will be complete
+ @AchievesGoal(
+ description = "Write an amusing writeup for the target person based on their horoscope and current news stories",
+ )
+ @Action
+ fun writeup(
+ person: StarPerson,
+ relevantNewsStories: RelevantNewsStories,
+ horoscope: Horoscope,
+ ai: Ai,
+ ): Writeup =
+ ai
+ .withLlm(
+ LlmOptions
+ .withModel(model)
+ .withTemperature(0.9)
+ )
+ .createObject(
+ """
+ Take the following news stories and write up something
+ amusing for the target person.
+
+ Begin by summarizing their horoscope in a concise, amusing way, then
+ talk about the news. End with a surprising signoff.
+
+ ${person.name} is an astrology believer with the sign ${person.sign}.
+ Their horoscope for today is:
+ ${horoscope.summary}
+ Relevant news stories are:
+ ${relevantNewsStories.items.joinToString("\n") { "- ${it.url}: ${it.summary}" }}
+
+ Format it as Markdown with links.
+ """.trimIndent()
+ )
+
+}
+```
+
+
+
+
+The following domain classes ensure type safety:
+
+
+Java
+
+```java
+
+@JsonClassDescription("Person with astrology details")
+@JsonDeserialize(as = StarPerson.class)
+public record StarPerson(
+ String name,
+ @JsonPropertyDescription("Star sign") String sign
+) implements Person {
+
+ @JsonCreator
+ public StarPerson(
+ @JsonProperty("name") String name,
+ @JsonProperty("sign") String sign
+ ) {
+ this.name = name;
+ this.sign = sign;
+ }
+
+ @Override
+ public String getName() {
+ return name;
+ }
+}
+
+public record Horoscope(String summary) {
+}
+
+@JsonClassDescription("Writeup relating to a person's horoscope and relevant news")
+public record Writeup(String text) implements HasContent {
+
+ @JsonCreator
+ public Writeup(@JsonProperty("text") String text) {
+ this.text = text;
+ }
+
+ @Override
+ public String getContent() {
+ return text;
+ }
+}
+
+```
+
+
+
+
+Kotlin
+
+```kotlin
+data class RelevantNewsStories(
+ val items: List
+)
+
+data class NewsStory(
+ val url: String,
+
+ val summary: String,
+)
+
+data class Subject(
+ val name: String,
+ val sign: String,
+)
+
+data class Horoscope(
+ val summary: String,
+)
+
+data class FunnyWriteup(
+ override val text: String,
+) : HasContent
+```
+
+
+
+It's easy to unit test your agents to ensure that they correctly execute logic
+and pass the correct prompts and hyperparameters to LLMs. For example:
+
+```java
+public class StarNewsFinderTest {
+
+ @Test
+ void writeupPromptMustContainKeyData() {
+ HoroscopeService horoscopeService = mock(HoroscopeService.class);
+ StarNewsFinder starNewsFinder = new StarNewsFinder(horoscopeService, 5);
+ var context = new FakeOperationContext();
+ context.expectResponse(new com.embabel.example.horoscope.Writeup("Gonna be a good day"));
+
+ NewsStory cockatoos = new NewsStory(
+ "https://fake.com.au",
+ "Cockatoo behavior",
+ "Cockatoos are eating cabbages"
+ );
+
+ NewsStory emus = new NewsStory(
+ "https://morefake.com.au",
+ "Emu movements",
+ "Emus are massing"
+ );
+
+ StarPerson starPerson = new StarPerson("Lynda", "Scorpio");
+ RelevantNewsStories relevantNewsStories = new RelevantNewsStories(Arrays.asList(cockatoos, emus));
+ Horoscope horoscope = new Horoscope("This is a good day for you");
+
+ starNewsFinder.writeup(starPerson, relevantNewsStories, horoscope, context);
+
+ var prompt = context.getLlmInvocations().getFirst().getPrompt();
+ var toolGroups = context.getLlmInvocations().getFirst().getInteraction().getToolGroups();
+
+
+ assertTrue(prompt.contains(starPerson.getName()));
+ assertTrue(prompt.contains(starPerson.sign()));
+ assertTrue(prompt.contains(cockatoos.getSummary()));
+ assertTrue(prompt.contains(emus.getSummary()));
+
+ assertTrue(toolGroups.isEmpty(), "The LLM should not have been given any tool groups");
+ }
+}
+```
+
+## Dog Food Policy
+
+We believe that all aspects of software development and business can and should
+be greatly accelerated through the use of AI agents. The ultimate decision
+makers remain human, but they can and should be greatly augmented.
+
+> This project practices extreme dogfooding.
+
+
+
+Our key principles:
+
+1. **We will use AI agents to help every aspect of the project:** coding, documentation, community management, producing
+ marketing copy etc.
+ Any
+ human performing a task should ask why it cannot be automated, and strive toward maximum automation.
+2. **Developers retain ultimate control.** Developers are responsible for guiding agents toward the solution and
+ iterating
+ as necessary. A developer who commits or merges an agent contribution
+ is responsible for ensuring that it meets the project coding standards, which are
+ independent of the use of agents. For example, code must be human-readable.
+3. **We will favour open source agents built on the Embabel platform,** and contribute improvements. While
+ commercial agents
+ may be more advanced in some areas, we believe that our
+ platform is the best general solution for automation and by dogfooding we will improve it fastest.
+ By open sourcing agents used on our open source projects, we will maximize benefit to the community.
+4. **We will prioritize agents that help accelerate our progress.** Per the flight safety advice to fit your own mask
+ before helping others, we will prioritize
+ agents that help us accelerate our own progress. This will not only produce useful examples, but increase overall
+ project velocity.
+
+Developers must carefully read all code they commit and improve generated code if possible.
+
+> Coding agents are a special case. While the `embabel-agent-code` submodule offers support for project modification
+> that is useful for project bootstrapping, coding agents are the most mature of commercial agents, and their vendors
+> are
+> heavily subsidising their users, making it economically irrational to insist on our own platform.
+
+## Getting Started
+
+- Get the bits
+- Set up your environment
+- Run the application
+
+### Getting the bits
+
+Choose one of the following:
+
+- Clone the repository via `git clone https://github.com/embabel/embabel-agent`
+- Create a new Spring Boot project and add the necessary dependencies (see "Using Embabel Agent Framework in Your
+ Project" below)
+
+### Environment variables
+
+> Environment variables are consistent with common usage, rather than Spring AI.
+> For example, we prefer `OPENAI_API_KEY` to `SPRING_AI_OPENAI_API_KEY`.
+
+Required:
+
+- `OPENAI_API_KEY`: For the OpenAI API
+
+Optional:
+
+- `ANTHROPIC_API_KEY`: For the Anthropic API. Necessary for the coding agent.
+- `MINIMAX_API_KEY`: For the [MiniMax](https://www.minimax.io) API. Supports MiniMax-M3, MiniMax-M2.7 and MiniMax-M2.7-highspeed models.
+- `ZAI_API_KEY`: For the [Z.ai](https://z.ai) (Zhipu AI) API. Supports GLM-5.2, GLM-4.7, GLM-4.6, GLM-4.5-Air and GLM-4.7-Flash models.
+- OCI Generative AI uses OCI SDK authentication providers. Add `embabel-agent-starter-oci-genai` and set
+ `embabel.agent.platform.models.ocigenai.compartment-id`; OCI config file, instance principal, resource principal,
+ workload identity, session token and simple key authentication are supported.
+
+> We strongly recommend providing both an OpenAI and Anthropic key, as some examples require both. And it's important to
+> try to find the best LLM for a given task, rather than automatically choose a familiar provider.
+
+### Services
+
+You will need a Docker Desktop version [`>4.43.2`](https://docs.docker.com/desktop/release-notes/).
+Be sure to activate the following MCP tools from the catalog:
+
+- Brave Search
+- Fetch
+- Puppeteer
+- Wikipedia
+
+> You can also set up your own MCP tools using Spring AI conventions. See the `application-docker-desktop.yml` file for
+> an example.
+
+If you're running Ollama locally, include the `embabel ollama starter` and Embabel will automatically connect to your
+Ollama
+endpoint and make all models available.
+
+```xml
+
+
+ com.embabel.agent
+ embabel-agent-starter-ollama
+
+```
+
+### Running
+
+Create your own agent project with
+
+```
+uvx --from git+https://github.com/embabel/project-creator.git project-creator
+```
+
+### Example Agents
+
+> **📚 For examples and tutorials**, see
+> the [Embabel Agent Examples Repository](https://github.com/embabel/embabel-agent-examples)
+
+```bash
+# Clone and run examples
+git clone https://github.com/embabel/embabel-agent-examples
+cd embabel-agent-examples/scripts/kotlin
+./shell.sh
+```
+
+#### Shell Commands
+
+Spring Shell is an easy way to interact with the Embabel agent framework, especially during development.
+
+Type `help` to see available commands. Use `execute` or `x` to run an agent:
+
+```
+execute "Lynda is a Scorpio, find news for her" -p -r
+```
+
+This will look for an agent, choose the star finder agent and
+run the flow. `-p` will log prompts `-r` will log LLM responses.
+Omit these for less verbose logging.
+
+Options:
+
+- `-p` logs prompts
+- `-r` logs LLM responses
+
+Use the `chat` command to enter an interactive chat with the agent.
+It will attempt to run the most appropriate
+agent for each command.
+
+> Spring Shell supports history. Type `!!` to repeat the last command.
+> This will survive restarts, so is handy when iterating on an agent.
+
+#### Further examples
+
+Example commands within the shell:
+
+```
+# Perplexity style deep research
+# Requires both OpenAI and Anthropic keys and Docker Desktop with the MCP extension (or your own web tools)
+execute "research the recent australian federal election. what is the position of the greens party?"
+
+# x is a shortcut for execute
+x "fact check the following: holden cars are still made in australia; the koel is a bird native only to australia; fidel castro is justin trudeau's father"
+
+```
+
+### Bringing in additional LLMs
+
+#### Local models with well-known providers
+
+The Embabel Agent Framework supports local models from:
+
+- Ollama: Simply add `embabel-agent-starter-ollama` starter to your pom.xml and your local Ollama endpoint will be
+ queries. All local models will be
+ available.
+- Docker: Add the `embabel-agent-starter-dockermodels` starter to your pom.xml and your local Docker endpoint will be
+ queried. All local models will be available.
+- LMStudio: This uses the openAI compatible client. Just include LMStudio as a dependency and make sure your LMStudio
+ server is running.
+
+#### OCI Generative AI
+
+Add `embabel-agent-starter-oci-genai` to use OCI Generative AI chat and embedding models.
+
+```xml
+
+ com.embabel.agent
+ embabel-agent-starter-oci-genai
+
+```
+
+Configure `embabel.agent.platform.models.ocigenai.compartment-id` and, if needed, set
+`embabel.agent.platform.models.ocigenai.authentication-type` to `FILE`, `INSTANCE_PRINCIPAL`, `RESOURCE_PRINCIPAL`,
+`WORKLOAD_IDENTITY`, `SESSION_TOKEN` or `SIMPLE`. When the standard OpenAI provider is not on the classpath, the OCI
+starter supplies OCI defaults for Embabel's default LLM and embedding model:
+
+```properties
+embabel.models.default-llm=cohere.command-a-03-2025
+embabel.models.default-embedding-model=cohere.embed-v4.0
+```
+
+Override those values in application configuration if you want another OCI model. Use OCI model ids such as
+`cohere.command-a-03-2025` or `meta.llama-3.3-70b-instruct` for Embabel model selection.
+The Spring bean names registered by the starter are Java-friendly aliases such as `cohere_command_a` and
+`llama_33_70b`.
+If your application exposes Spring Boot Actuator `env` or `configprops` values, keep those endpoints secured and ensure
+OCI credential fields such as `pass-phrase`, `session-token` and `private-key` are sanitized.
+
+#### Custom LLMs
+
+You can define an LLM for any provider for which a Spring AI `ChatModel` is available.
+
+Simply define Spring beans of type `Llm`.
+See the `OpenAiConfiguration` class as an example.
+
+Remember:
+
+- Provide the knowledge cutoff date if you know it
+- Make the configuration class conditional on any required API key.
+
+## Roadmap
+
+This project is in its early stages, but we have big plans.
+The milestones and issues in this repository are a good reference.
+Our key goals:
+
+- **Become the natural way to Gen AI-enable Java applications**, and especially those built on Spring.
+- **Prove the power of the approach**. Demonstrate that this approach is the best way to build safe, dependable, Gen AI
+ applications.
+ In particular:
+ - Demonstrate the power of extensibility without modification, by adding goals and actions
+ - Demonstrate the potential to become the PaaS for natural language
+ - Demonstrate the potential of agent federation within the GOAP model
+ - Demonstrate budget-aware agents, such as "Research the following topic, spending up to 20c if you are still
+ learning"
+ - Integrate with data stores and demonstrate the power of surfacing existing functionality inside an organization
+- **Take the model to other platforms**: The conceptual framework is not JVM specific. Once established, we intend to
+ create TypeScript
+ and Python projects.
+
+There is a lot to do, and you are awesome. We look forward to your contribution!
+
+## Application Design
+
+### Domain objects
+
+Applications center around domain objects. These can be instantiated by LLMs or user
+code, and manipulated by user code.
+
+Use Jackson annotations to help LLMs with descriptions as well as mark fields to ignore.
+For example:
+
+```kotlin
+@JsonClassDescription("Person with astrology details")
+data class StarPerson(
+ override val name: String,
+ @get:JsonPropertyDescription("Star sign")
+ val sign: String,
+) : Person
+```
+
+See [Java Json Schema Generation - Module Jackson](https://github.com/victools/jsonschema-generator/tree/main/jsonschema-module-jackson)
+for documentation of the library used.
+
+Domain objects can have behaviors that are automatically exposed to LLMs when they are in scope. Simply annotate methods
+with the Spring AI `@Tool` annotation.
+
+> When exposing `@Tool` methods on domain objects, be sure that the tool is safe to invoke. Even the best LLMs can get
+> trigger-happy. For example, be careful about methods that can mutate or delete data. This is likely better modeled via
+> an explicit call to a non-tool method on the same domain class, in a code action.
+
+## Using Embabel as an MCP server
+
+You can use the Embabel agent platform as an MCP server from a
+UI like Claude Desktop. The Embabel MCP server is available over SSE.
+
+Configure Claude Desktop as follows in your `claude_desktop_config.yml`:
+
+```json
+{
+ "mcpServers": {
+ "embabel": {
+ "command": "npx",
+ "args": [
+ "-y",
+ "mcp-remote",
+ "http://localhost:8080/sse"
+ ]
+ }
+ }
+}
+
+```
+
+See [MCP Quickstart for Claude Desktop Users](https://modelcontextprotocol.io/quickstart/user) for how to configure
+Claude Desktop.
+
+The [MCP Inspector](https://github.com/modelcontextprotocol/inspector) is a helpful tool for interacting with your
+Embabel
+SSE server, manually invoking tools and checking the exposed prompts and resources.
+
+Start the MCP Inspector with:
+
+```bash
+npx @modelcontextprotocol/inspector
+```
+
+## Consuming MCP Servers
+
+The Embabel Agent Framework provides built-in support for consuming Model Context Protocol (MCP) servers, allowing you
+to extend your applications with powerful AI capabilities through standardized interfaces.
+
+### What is MCP?
+
+Model Context Protocol (MCP) is an open protocol that standardizes how applications provide context and extra
+functionality to large language models. Introduced by Anthropic, MCP has emerged as the de facto standard for connecting
+AI agents to tools, functioning as a client-server protocol where:
+
+- **Clients** (like Embabel Agent) send requests to servers
+- **Servers** process those requests to deliver necessary context to the AI model
+
+MCP simplifies integration between AI applications and external tools, transforming an "M×N problem" into an "M+N
+problem" through standardization - similar to what USB did for hardware peripherals.
+
+### Configuring MCP in Embabel Agent
+
+To configure MCP servers in your Embabel Agent application, add the following to your `application.yml`:
+
+```yaml
+spring:
+ ai:
+ mcp:
+ client:
+ enabled: true
+ name: embabel
+ version: 1.0.0
+ request-timeout: 30s
+ type: SYNC
+ stdio:
+ connections:
+ docker-mcp:
+ command: docker
+ args:
+ - run
+ - -i
+ - --rm
+ - alpine/socat
+ - STDIO
+ - TCP:host.docker.internal:8811
+```
+
+This configuration sets up an MCP client that connects to a Docker-based MCP server. The connection uses STDIO transport
+through Docker's socat utility to connect to a TCP endpoint.
+
+### Docker Desktop MCP Integration
+
+Docker has embraced MCP with their Docker MCP Catalog and Toolkit, which provides:
+
+1. **Centralized Discovery** - A trusted hub for discovering MCP tools integrated into Docker Hub
+2. **Containerized Deployment** - Run MCP servers as containers without complex setup
+3. **Secure Credential Management** - Centralized, encrypted credential handling
+4. **Built-in Security** - Sandbox isolation and permissions management
+
+The Docker MCP ecosystem includes over 100 verified tools from partners like Stripe, Elastic, Neo4j, and more, all
+accessible through Docker's infrastructure.
+
+### Learn More
+
+- [Docker MCP Documentation](https://docs.docker.com/desktop/features/gordon/mcp/)
+- [Docker MCP Servers Repository](https://github.com/docker/mcp-servers)
+- [Introducing Docker MCP Catalog and Toolkit](https://www.docker.com/blog/introducing-docker-mcp-catalog-and-toolkit/)
+- [MCP Introduction and Overview](https://www.philschmid.de/mcp-introduction)
+
+## A2A
+
+Embabel integrates with the [A2A](https://github.com/google-a2a/A2A) protocol, allowing you to connect to other
+A2A-enabled agents and
+services.
+
+Enable the `a2a` Spring profile to start the A2A server.
+
+You'll need the following environment variable:
+
+- `GOOGLE_STUDIO_API_KEY`: Your Google Studio API key, which is used for Gemini.
+
+Start the Google A2A web interface using the `a2a` Docker profile:
+
+```bash
+docker compose --profile a2a up
+```
+
+Go to the web interface running within the container at `http://localhost:12000/`.
+
+Connect to your agent at `host.docker.internal:8080/a2a`. Note that `localhost:8080/a2a` won't work as the server
+cannot access it when running in a Docker container.
+
+## Running Tests
+
+### Unit tests
+
+Run the unit tests via Maven. This will not require an internet connection or any external services.
+
+```bash
+mvn test
+```
+
+### Integration tests
+
+Integration tests (`*IT`) hit real provider APIs and are excluded from the default `mvn test` run.
+To run them, ensure the following environment variables are set:
+
+- `OPENAI_API_KEY`
+- `ANTHROPIC_API_KEY`
+- `DEEPSEEK_API_KEY`
+- `MISTRAL_API_KEY`
+
+Then run:
+
+```bash
+mvn -Dtest='*IT,!LLMOllama*IT' -Dsurefire.failIfNoSpecifiedTests=false test
+```
+
+This runs all integration tests except Ollama (which requires a local Ollama server).
+To run a specific module's integration tests, add `-pl`:
+
+```bash
+mvn -Dtest='*IT,!LLMOllama*IT' -Dsurefire.failIfNoSpecifiedTests=false test -pl embabel-agent-openai
+```
+
+## Spring profiles
+
+Spring profiles are used to configure the application for different environments and behaviors.
+
+Model profiles:
+
+- `docker-desktop`: Talking to Docker Desktop with the MCP extension. **This is recommended for the best experience,
+ with Docker-provided web tools.**
+
+Logging profiles:
+
+- `severance`: [Severance](https://www.youtube.com/watch?v=xEQP4VVuyrY&ab_channel=AppleTV) specific logging. Praise
+ Kier!
+- `starwars`: Star Wars specific logging. Feel the force
+- `colossus`: Colossus specific logging. The Forbin Project.
+
+## Testing
+
+A key goal of this framework is ease of testing.
+Just as Spring eased testing of early enterprise Java applications,
+this framework facilitates testing of AI applications.
+
+Types of testing:
+
+- Unit tests: All agents are unit testable, like any Spring-managed beans. Construct them with mock objects; call
+ individual action methods. The testing library facilitates testing prompts.
+- Integration tests: tbd
+
+## Logging
+
+All logging in this project is either debug logging in the relevant
+class itself, or results from the stream of events of type `AgentEvent`.
+
+Edit `application.yml` if you want to see debug logging from the relevant classes and packages.
+
+Available logging experiences:
+
+- `severance`: Severance logging. Praise Kier
+- `starwars`: Star Wars logging. Feel the force. The default as it's understood throughout the galaxy.
+- `colossus`: Colossus logging. The Forbin Project.
+- `montypython`: Monty Python logging. No one expects it.
+- `hh`: Hitchhiker's Guide to the Galaxy logging. The answer is 42.
+
+If none of these profiles is chosen, Embabel will use vanilla logging.
+This makes me sad.
+
+## Adding Embabel Agent Framework to Your Project
+
+### Maven Central Availability
+
+**Since version 0.2.0**, Embabel Agent Framework is available directly on Maven Central, simplifying dependency
+management. You no longer need to configure custom repositories for stable releases.
+
+---
+
+### Maven
+
+#### For version 0.2.0 and above (Recommended)
+
+Simply add the Embabel Spring Boot starter dependency to your `pom.xml`:
+
+```xml
+
+
+ com.embabel.agent
+ embabel-agent-starter
+ ${embabel-agent.version}
+
+```
+
+No additional repository configuration is needed! Maven Central is configured by default.
+
+#### For versions prior to 0.2.0 or for SNAPSHOT versions
+
+You need to add the Embabel repositories to your `pom.xml`:
+
+```xml
+
+
+
+ embabel-releases
+ https://repo.embabel.com/artifactory/libs-release
+
+ true
+
+
+ false
+
+
+
+ embabel-snapshots
+ https://repo.embabel.com/artifactory/libs-snapshot
+
+ false
+
+
+ true
+
+
+
+```
+
+Then add the dependency:
+
+```xml
+
+
+ com.embabel.agent
+ embabel-agent-starter
+ ${embabel-agent.version}
+
+```
+
+---
+
+### Gradle (Kotlin DSL)
+
+#### For version 0.2.0 and above (Recommended)
+
+Add the required repositories to your `build.gradle.kts`:
+
+```kotlin
+repositories {
+ mavenCentral()
+ maven {
+ name = "Spring Milestones"
+ url = uri("https://repo.spring.io/milestone")
+ }
+}
+```
+
+Add the Embabel Agent starter:
+
+```kotlin
+dependencies {
+ implementation("com.embabel.agent:embabel-agent-starter:${embabelAgentVersion}")
+}
+```
+
+#### For versions prior to 0.2.0 or for SNAPSHOT versions
+
+Add all required repositories to your `build.gradle.kts`:
+
+```kotlin
+repositories {
+ mavenCentral()
+ maven {
+ name = "embabel-releases"
+ url = uri("https://repo.embabel.com/artifactory/libs-release")
+ mavenContent {
+ releasesOnly()
+ }
+ }
+ maven {
+ name = "embabel-snapshots"
+ url = uri("https://repo.embabel.com/artifactory/libs-snapshot")
+ mavenContent {
+ snapshotsOnly()
+ }
+ }
+ maven {
+ name = "Spring Milestones"
+ url = uri("https://repo.spring.io/milestone")
+ }
+}
+```
+
+Add the Embabel Agent starter:
+
+```kotlin
+dependencies {
+ implementation("com.embabel.agent:embabel-agent-starter:${embabelAgentVersion}")
+}
+```
+
+---
+
+### Gradle (Groovy DSL)
+
+#### For version 0.2.0 and above (Recommended)
+
+Add the required repositories to your `build.gradle`:
+
+```groovy
+repositories {
+ mavenCentral()
+ maven {
+ name = 'Spring Milestones'
+ url = 'https://repo.spring.io/milestone'
+ }
+}
+```
+
+Add the Embabel Agent starter:
+
+```groovy
+dependencies {
+ implementation "com.embabel.agent:embabel-agent-starter:${embabelAgentVersion}"
+}
+```
+
+#### For versions prior to 0.2.0 or for SNAPSHOT versions
+
+Add all required repositories to your `build.gradle`:
+
+```groovy
+repositories {
+ mavenCentral()
+ maven {
+ name = 'embabel-releases'
+ url = 'https://repo.embabel.com/artifactory/libs-release'
+ mavenContent {
+ releasesOnly()
+ }
+ }
+ maven {
+ name = 'embabel-snapshots'
+ url = 'https://repo.embabel.com/artifactory/libs-snapshot'
+ mavenContent {
+ snapshotsOnly()
+ }
+ }
+ maven {
+ name = 'Spring Milestones'
+ url = 'https://repo.spring.io/milestone'
+ }
+}
+```
+
+Add the Embabel Agent starter:
+
+```groovy
+dependencies {
+ implementation 'com.embabel.agent:embabel-agent-starter:${embabelAgentVersion}'
+}
+```
+
+---
+
+### Important Notes
+
+#### Spring Milestones Repository
+
+The Spring Milestones repository is required because the Embabel BOM (`embabel-agent-dependencies`) has transitive
+dependencies on experimental Spring components, specifically the `mcp-bom`. This BOM is not available on Maven Central
+and is only published to the Spring milestone repository.
+
+**Note for Gradle users:** Unlike Maven, Gradle does not inherit repository configurations declared in parent POMs or
+BOMs. Therefore, it is necessary to explicitly declare the Spring milestone repository in your repositories block to
+ensure proper resolution of all transitive dependencies.
+
+#### Repository Types
+
+- **Maven Central** (since v0.2.0): For stable releases 0.2.0 and above
+- **Embabel Releases Repository**: For stable releases prior to 0.2.0
+- **Embabel Snapshots Repository**: For development/snapshot versions (e.g., `0.3.0-SNAPSHOT`)
+
+---
+
+### Quick Start Examples
+
+#### Maven with latest stable version
+
+```xml
+
+
+ com.embabel.agent
+ embabel-agent-starter
+ 0.3.0
+
+```
+
+#### Gradle Kotlin DSL with latest stable version
+
+```kotlin
+implementation("com.embabel.agent:embabel-agent-starter:0.3.0")
+```
+
+#### Gradle Groovy DSL with latest stable version
+
+```groovy
+implementation 'com.embabel.agent:embabel-agent-starter:0.3.0'
+```
+
+## Getting Started with Observability
+
+Add full tracing and metrics to your Embabel agents with zero code changes.
+
+### 1. Add the dependency
+
+```xml
+
+ com.embabel.agent
+ embabel-agent-starter-observability
+ ${embabel-agent.version}
+
+```
+
+### 2. Add an exporter
+
+Pick **one** (or combine multiple):
+
+**Zipkin** (simplest — no account needed):
+```xml
+
+ io.opentelemetry
+ opentelemetry-exporter-zipkin
+
+```
+
+**Langfuse** (LLM-focused observability):
+```xml
+
+ com.quantpulsar
+ opentelemetry-exporter-langfuse
+ 0.4.0
+
+```
+
+### 3. Configure
+
+```yaml
+# Enable observability
+embabel:
+ agent:
+ platform:
+ observability:
+ enabled: true
+ service-name: my-agent-app
+
+# Enable Spring Boot tracing
+management:
+ tracing:
+ export:
+ enabled: true
+ sampling:
+ probability: 1.0
+
+ # Zipkin exporter
+ zipkin:
+ tracing:
+ endpoint: http://localhost:9411/api/v2/spans
+```
+
+To use Langfuse instead of (or alongside) Zipkin:
+```yaml
+management:
+ langfuse:
+ enabled: true
+ endpoint: https://cloud.langfuse.com/api/public/otel # or your self-hosted URL
+ public-key: pk-lf-...
+ secret-key: sk-lf-...
+```
+
+### 4. Start Zipkin and run
+
+```bash
+docker run -d -p 9411:9411 openzipkin/zipkin
+./mvnw spring-boot:run
+```
+
+Open [http://localhost:9411](http://localhost:9411) — run an agent and you will see traces like:
+
+```
+Agent: CustomerServiceAgent
+├── Action: AnalyzeRequest
+│ └── ChatModel: gpt-4 (Spring AI)
+│ └── tool:searchKnowledgeBase
+├── Action: GenerateResponse
+│ └── ChatModel: gpt-4 (Spring AI)
+└── status: completed [duration=2340ms]
+```
+
+With Langfuse, you get a rich LLM-focused view of your agent traces:
+
+
+
+### What gets traced automatically
+
+- Agent lifecycle (creation, execution, completion, failures)
+- Every action as a child span
+- LLM calls with token usage (via Spring AI)
+- Tool invocations with input/output
+- Planning and replanning iterations
+- State transitions and lifecycle states
+
+### Track custom operations with `@Tracked`
+
+Use the `@Tracked` annotation to add observability spans to your own methods — inputs, outputs, duration, and errors are captured automatically:
+
+```java
+@Tracked("enrichCustomer")
+public Customer enrich(Customer input) {
+ // Your logic here
+}
+```
+
+You can specify a type and description for richer traces:
+
+```java
+@Tracked(value = "callPaymentApi", type = TrackType.EXTERNAL_CALL, description = "Payment gateway call")
+public PaymentResult processPayment(Order order) {
+ // ...
+}
+```
+
+When called within an agent execution, `@Tracked` spans are automatically nested under the current action:
+
+```
+Agent: CustomerServiceAgent
+├── Action: ProcessOrder
+│ ├── @Tracked: enrichCustomer (PROCESSING)
+│ ├── ChatModel: gpt-4
+│ └── @Tracked: callPaymentApi (EXTERNAL_CALL)
+└── status: completed
+```
+
+For the full configuration reference, MDC log correlation, and advanced options, see the [Observability Module Documentation](embabel-agent-observability/README.md).
+
+---
+
+## Contributing
+
+We welcome contributions to the Embabel Agent Framework.
+
+Look at the [coding style guide](embabel-agent-api/.embabel/coding-style.md) for style guidelines.
+This file also informs coding agent behavior.
+
+## Miscellaneous
+
+- _Why the name Embabel?_
+ The "babel" part is ultimately inspired by the story of the Tower of Babel, perhaps via Douglas
+ Adams' [babelfish](https://www.youtube.com/watch?v=iuumnjJWFO4&ab_channel=BBCStudios).
+ Per @lasuac:
+ _While Adams' fish in the ear enabled universal translation between species, Embabel aims at translating human intent
+ to JVM code, AI models, and enterprise systems._
+ "embabel" also sounds like "enable."
+- Milestone names are Australian animals. Mythical animals such as "bunyip" and "yowie" are used for futures that may or
+ not be implemented.
+- README badges come from [here](https://github.com/Ileriayo/markdown-badges)
+ and [here](https://home.aveek.io/GitHub-Profile-Badges/).
+- Don't forget to join [Discord](https://discord.gg/t6bjkyj93q) to collaborate with the Embabel community. It is a good
+ place to receive support, showcase your work, discuss ideas and connect with like-minded people.
+
+## Star History
+
+
+
+
+
+
+
+
+
+## Contributors
+
+[](https://github.com/embabel/embabel-agent/graphs/contributors)
+
+
+
+--------------------
+(c) Embabel Software Inc 2024-2026.
diff --git a/.verify/rtk-ai_rtk.json b/.verify/rtk-ai_rtk.json
new file mode 100644
index 0000000..3fd8645
--- /dev/null
+++ b/.verify/rtk-ai_rtk.json
@@ -0,0 +1 @@
+{"id":1139971460,"node_id":"R_kgDOQ_KVhA","name":"rtk","full_name":"rtk-ai/rtk","private":false,"owner":{"login":"rtk-ai","id":258253854,"node_id":"O_kgDOD2SkHg","avatar_url":"https://avatars.githubusercontent.com/u/258253854?v=4","gravatar_id":"","url":"https://api.github.com/users/rtk-ai","html_url":"https://github.com/rtk-ai","followers_url":"https://api.github.com/users/rtk-ai/followers","following_url":"https://api.github.com/users/rtk-ai/following{/other_user}","gists_url":"https://api.github.com/users/rtk-ai/gists{/gist_id}","starred_url":"https://api.github.com/users/rtk-ai/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/rtk-ai/subscriptions","organizations_url":"https://api.github.com/users/rtk-ai/orgs","repos_url":"https://api.github.com/users/rtk-ai/repos","events_url":"https://api.github.com/users/rtk-ai/events{/privacy}","received_events_url":"https://api.github.com/users/rtk-ai/received_events","type":"Organization","user_view_type":"public","site_admin":false},"html_url":"https://github.com/rtk-ai/rtk","description":"CLI proxy that reduces LLM token consumption by 60-90% on common dev commands. Single Rust binary, zero dependencies","fork":false,"url":"https://api.github.com/repos/rtk-ai/rtk","forks_url":"https://api.github.com/repos/rtk-ai/rtk/forks","keys_url":"https://api.github.com/repos/rtk-ai/rtk/keys{/key_id}","collaborators_url":"https://api.github.com/repos/rtk-ai/rtk/collaborators{/collaborator}","teams_url":"https://api.github.com/repos/rtk-ai/rtk/teams","hooks_url":"https://api.github.com/repos/rtk-ai/rtk/hooks","issue_events_url":"https://api.github.com/repos/rtk-ai/rtk/issues/events{/number}","events_url":"https://api.github.com/repos/rtk-ai/rtk/events","assignees_url":"https://api.github.com/repos/rtk-ai/rtk/assignees{/user}","branches_url":"https://api.github.com/repos/rtk-ai/rtk/branches{/branch}","tags_url":"https://api.github.com/repos/rtk-ai/rtk/tags","blobs_url":"https://api.github.com/repos/rtk-ai/rtk/git/blobs{/sha}","git_tags_url":"https://api.github.com/repos/rtk-ai/rtk/git/tags{/sha}","git_refs_url":"https://api.github.com/repos/rtk-ai/rtk/git/refs{/sha}","trees_url":"https://api.github.com/repos/rtk-ai/rtk/git/trees{/sha}","statuses_url":"https://api.github.com/repos/rtk-ai/rtk/statuses/{sha}","languages_url":"https://api.github.com/repos/rtk-ai/rtk/languages","stargazers_url":"https://api.github.com/repos/rtk-ai/rtk/stargazers","contributors_url":"https://api.github.com/repos/rtk-ai/rtk/contributors","subscribers_url":"https://api.github.com/repos/rtk-ai/rtk/subscribers","subscription_url":"https://api.github.com/repos/rtk-ai/rtk/subscription","commits_url":"https://api.github.com/repos/rtk-ai/rtk/commits{/sha}","git_commits_url":"https://api.github.com/repos/rtk-ai/rtk/git/commits{/sha}","comments_url":"https://api.github.com/repos/rtk-ai/rtk/comments{/number}","issue_comment_url":"https://api.github.com/repos/rtk-ai/rtk/issues/comments{/number}","contents_url":"https://api.github.com/repos/rtk-ai/rtk/contents/{+path}","compare_url":"https://api.github.com/repos/rtk-ai/rtk/compare/{base}...{head}","merges_url":"https://api.github.com/repos/rtk-ai/rtk/merges","archive_url":"https://api.github.com/repos/rtk-ai/rtk/{archive_format}{/ref}","downloads_url":"https://api.github.com/repos/rtk-ai/rtk/downloads","issues_url":"https://api.github.com/repos/rtk-ai/rtk/issues{/number}","pulls_url":"https://api.github.com/repos/rtk-ai/rtk/pulls{/number}","milestones_url":"https://api.github.com/repos/rtk-ai/rtk/milestones{/number}","notifications_url":"https://api.github.com/repos/rtk-ai/rtk/notifications{?since,all,participating}","labels_url":"https://api.github.com/repos/rtk-ai/rtk/labels{/name}","releases_url":"https://api.github.com/repos/rtk-ai/rtk/releases{/id}","deployments_url":"https://api.github.com/repos/rtk-ai/rtk/deployments","created_at":"2026-01-22T16:54:16Z","updated_at":"2026-08-12T14:39:54Z","pushed_at":"2026-08-12T00:10:53Z","git_url":"git://github.com/rtk-ai/rtk.git","ssh_url":"git@github.com:rtk-ai/rtk.git","clone_url":"https://github.com/rtk-ai/rtk.git","svn_url":"https://github.com/rtk-ai/rtk","homepage":"https://www.rtk-ai.app","size":6515,"stargazers_count":75861,"watchers_count":75861,"language":"Rust","has_issues":true,"has_projects":true,"has_downloads":false,"has_wiki":true,"has_pages":false,"has_discussions":true,"forks_count":4770,"mirror_url":null,"archived":false,"disabled":false,"open_issues_count":1953,"license":{"key":"apache-2.0","name":"Apache License 2.0","spdx_id":"Apache-2.0","url":"https://api.github.com/licenses/apache-2.0","node_id":"MDc6TGljZW5zZTI="},"allow_forking":true,"is_template":false,"web_commit_signoff_required":false,"has_pull_requests":true,"pull_request_creation_policy":"all","topics":["agentic-coding","ai-coding","anthropic","claude-code","cli","command-line-tool","cost-reduction","developer-tools","llm","open-source","productivity","rust","token-optimization"],"visibility":"public","forks":4770,"open_issues":1953,"watchers":75861,"default_branch":"develop","temp_clone_token":null,"custom_properties":{},"organization":{"login":"rtk-ai","id":258253854,"node_id":"O_kgDOD2SkHg","avatar_url":"https://avatars.githubusercontent.com/u/258253854?v=4","gravatar_id":"","url":"https://api.github.com/users/rtk-ai","html_url":"https://github.com/rtk-ai","followers_url":"https://api.github.com/users/rtk-ai/followers","following_url":"https://api.github.com/users/rtk-ai/following{/other_user}","gists_url":"https://api.github.com/users/rtk-ai/gists{/gist_id}","starred_url":"https://api.github.com/users/rtk-ai/starred{/owner}{/repo}","subscriptions_url":"https://api.github.com/users/rtk-ai/subscriptions","organizations_url":"https://api.github.com/users/rtk-ai/orgs","repos_url":"https://api.github.com/users/rtk-ai/repos","events_url":"https://api.github.com/users/rtk-ai/events{/privacy}","received_events_url":"https://api.github.com/users/rtk-ai/received_events","type":"Organization","user_view_type":"public","site_admin":false},"network_count":4770,"subscribers_count":206}
\ No newline at end of file
diff --git a/.verify/rtk-ai_rtk_README.md b/.verify/rtk-ai_rtk_README.md
new file mode 100644
index 0000000..1becba2
--- /dev/null
+++ b/.verify/rtk-ai_rtk_README.md
@@ -0,0 +1 @@
+404: Not Found
\ No newline at end of file
diff --git a/.verify/rtk-ai_rtk_README_alt.md b/.verify/rtk-ai_rtk_README_alt.md
new file mode 100644
index 0000000..420a237
--- /dev/null
+++ b/.verify/rtk-ai_rtk_README_alt.md
@@ -0,0 +1,527 @@
+
+
+
+
+
+ High-performance CLI proxy that cuts up to 90% of the bash output your agent reads
+
+
+---
+
+rtk filters and compresses command outputs before they reach your LLM context. Single Rust binary, 100+ supported commands, <10ms overhead.
+
+## What RTK Does
+
+RTK intercepts shell commands and compresses their output before your agent reads it.
+
+| Operation | What RTK does to the output |
+|-----------|-----------------------------|
+| `ls` / `tree` | Tree format with file counts instead of one line per entry |
+| `cat` / `read` | Smart file reading: signatures and structure over full bodies |
+| `grep` / `rg` | Truncates long lines, groups matches by file |
+| `git status` | Compact stat format, grouped by state |
+| `git diff` | Reduced context, headers stripped |
+| `git log` | Hash, author and subject only |
+| `git add/commit/push` | Confirmation line instead of full progress output |
+| `cargo test` / `npm test` | Failures only, passing tests collapsed to a count |
+| `ruff check` | Grouped by rule and file |
+| `pytest` | Failures only, traceback trimmed |
+| `go test` | NDJSON parsed, failures only |
+| `docker ps` | Essential fields only |
+
+## How Savings Work
+
+RTK cuts **up to 90% of the bash output** your agent reads. That is what RTK measures, and it is not the same as cutting your bill by 90%.
+
+Bash output is **one contributor to input tokens**, alongside your prompt, the system prompt and conversation history. Input tokens are in turn **only part of the bill**, which also counts output tokens. The reduction dilutes at every step.
+
+The token counts RTK reports are estimated as `bytes / 4` — RTK ships no tokenizer, so the **percentages are reliable but the absolute token numbers are approximate**.
+
+> Full explanation: [How RTK Savings Work](docs/guide/resources/savings-explained.md)
+
+## Installation
+
+### Homebrew (recommended)
+
+```bash
+brew install rtk
+```
+
+### Quick Install (Linux/macOS)
+
+```bash
+curl -fsSL https://raw.githubusercontent.com/rtk-ai/rtk/refs/heads/master/install.sh | sh
+```
+
+> Installs to `~/.local/bin`. Add to PATH if needed:
+> ```bash
+> echo 'export PATH="$HOME/.local/bin:$PATH"' >> ~/.bashrc # or ~/.zshrc
+> ```
+
+### Cargo
+
+```bash
+cargo install --git https://github.com/rtk-ai/rtk
+```
+
+### Pre-built Binaries
+
+Download from [releases](https://github.com/rtk-ai/rtk/releases):
+- macOS: `rtk-x86_64-apple-darwin.tar.gz` / `rtk-aarch64-apple-darwin.tar.gz`
+- Linux: `rtk-x86_64-unknown-linux-musl.tar.gz` / `rtk-aarch64-unknown-linux-gnu.tar.gz`
+- Windows: `rtk-x86_64-pc-windows-msvc.zip`
+
+> **Windows users**: Extract the zip and place `rtk.exe` somewhere in your PATH (e.g. `C:\Users\\.local\bin`). Run RTK from **Command Prompt**, **PowerShell**, or **Windows Terminal** — do not double-click the `.exe` (it will flash and close). The full hook system works natively on Windows (and in [WSL](https://learn.microsoft.com/en-us/windows/wsl/install)). See [Windows setup](#windows) below for details.
+
+### Verify Installation
+
+```bash
+rtk --version # Should show "rtk 0.28.2"
+rtk gain # Should show the savings dashboard
+```
+
+> **Name collision warning**: Another project named "rtk" (Rust Type Kit) exists on crates.io. If `rtk gain` fails, you have the wrong package. Use `cargo install --git` above instead.
+
+## Quick Start
+
+```bash
+# 1. Install for your AI tool
+rtk init -g # Claude Code / Copilot (default)
+rtk init -g --gemini # Gemini CLI
+rtk init -g --codex # Codex (OpenAI)
+rtk init -g --agent cursor # Cursor
+rtk init -g --agent windsurf # Windsurf
+rtk init --agent cline # Cline / Roo Code
+rtk init --agent kilocode # Kilo Code
+rtk init --agent antigravity # Google Antigravity
+rtk init --agent kimi # Kimi AI
+rtk init -g --agent pi # Pi
+rtk init --agent hermes # Hermes
+rtk init -g --agent droid # Factory Droid
+
+# 2. Restart your AI tool, then test
+git status # Automatically rewritten to rtk git status
+```
+
+Hook-based agents rewrite Bash commands (e.g., `git status` -> `rtk git status`) before execution. Plugin-based agents, including Hermes, use their plugin API to rewrite commands before execution. The agent receives compact output without needing to call `rtk` explicitly.
+
+**Important:** the hook only runs on Bash tool calls. Claude Code built-in tools like `Read`, `Grep`, and `Glob` do not pass through the Bash hook, so they are not auto-rewritten. To get RTK's compact output for those workflows, use shell commands (`cat`/`head`/`tail`, `rg`/`grep`, `find`) or call `rtk read`, `rtk grep`, or `rtk find` directly.
+
+## How It Works
+
+```
+ Without rtk: With rtk:
+
+ Claude --git status--> shell --> git Claude --git status--> RTK --> git
+ ^ | ^ | |
+ | full raw output | | compact output | filter |
+ +-----------------------------------+ +------- (filtered) ---+----------+
+```
+
+Four strategies applied per command type:
+
+1. **Smart Filtering** - Removes noise (comments, whitespace, boilerplate)
+2. **Grouping** - Aggregates similar items (files by directory, errors by type)
+3. **Truncation** - Keeps relevant context, cuts redundancy
+4. **Deduplication** - Collapses repeated log lines with counts
+
+## Commands
+
+> Percentages below are **reductions in bash output**, not reductions in your bill. See [How Savings Work](#how-savings-work).
+
+### Files
+```bash
+rtk ls . # Compact directory tree
+rtk read file.rs # Smart file reading
+rtk read file.rs -l aggressive # Signatures only (strips bodies)
+rtk smart file.rs # 2-line heuristic code summary
+rtk find "*.rs" . # Compact find results
+rtk grep "pattern" . # Grouped search results
+rtk diff file1 file2 # Condensed diff (exit 1 if files differ)
+```
+
+### Git
+```bash
+rtk git status # Compact status
+rtk git log -n 10 # One-line commits
+rtk git diff # Condensed diff
+rtk git add # -> "ok"
+rtk git commit -m "msg" # -> "ok abc1234"
+rtk git push # -> "ok main"
+rtk git pull # -> "ok 3 files +10 -2"
+```
+
+### GitHub CLI
+```bash
+rtk gh pr list # Compact PR listing
+rtk gh pr view 42 # PR details + checks
+rtk gh issue list # Compact issue listing
+rtk gh run list # Workflow run status
+```
+
+### Test Runners
+```bash
+rtk jest # Jest compact (failures only)
+rtk vitest # Vitest compact (failures only)
+rtk playwright test # E2E results (failures only)
+rtk pytest # Python tests (-90%)
+rtk go test # Go tests (NDJSON, -90%)
+rtk cargo test # Cargo tests (-90%)
+rtk rake test # Ruby minitest (-90%)
+rtk rspec # RSpec tests (JSON, -60%+)
+rtk err # Filter errors only from any command
+rtk test # Generic test wrapper - failures only (-90%)
+```
+
+### Build & Lint
+```bash
+rtk lint # ESLint grouped by rule/file
+rtk lint biome # Supports other linters
+rtk tsc # TypeScript errors grouped by file
+rtk next build # Next.js build compact
+rtk prettier --check . # Files needing formatting
+rtk cargo build # Cargo build (-80%)
+rtk cargo clippy # Cargo clippy (-80%)
+rtk ruff check # Python linting (JSON, -80%)
+rtk golangci-lint run # Go linting (JSON, -85%)
+rtk rubocop # Ruby linting (JSON, -60%+)
+rtk sbt test # ScalaTest output (-90%)
+rtk sbt compile # Compilation errors only (-75%)
+rtk sbt run # Strip SBT preamble noise
+```
+
+### Package Managers
+```bash
+rtk pnpm list # Compact dependency tree
+rtk uv run pytest # Preserve uv env, keep program output
+rtk pip list # Python packages (auto-detect uv)
+rtk pip outdated # Outdated packages
+rtk bundle install # Ruby gems (strip Using lines)
+rtk prisma generate # Schema generation (no ASCII art)
+```
+
+### AWS
+```bash
+rtk aws sts get-caller-identity # One-line identity
+rtk aws ec2 describe-instances # Compact instance list
+rtk aws lambda list-functions # Name/runtime/memory (strips secrets)
+rtk aws logs get-log-events # Timestamped messages only
+rtk aws cloudformation describe-stack-events # Failures first
+rtk aws dynamodb scan # Unwraps type annotations
+rtk aws iam list-roles # Strips policy documents
+rtk aws s3 ls # Truncated with tee recovery
+```
+
+### Containers
+```bash
+rtk docker ps # Compact container list
+rtk docker images # Compact image list
+rtk docker logs # Deduplicated logs
+rtk docker compose ps # Compose services
+rtk kubectl pods # Compact pod list
+rtk kubectl logs # Deduplicated logs
+rtk kubectl services # Compact service list
+rtk oc get pods # OpenShift pod summary
+rtk oc get services # OpenShift service list
+rtk oc logs # Deduplicated logs
+```
+
+### Infrastructure as Code
+```bash
+rtk pulumi preview # Strip header/URL/duration noise
+rtk pulumi up # Compact apply output
+rtk pulumi destroy # Compact destroy output
+rtk pulumi refresh # Drift summary
+rtk pulumi stack # Stack metadata (strips owner/timestamps)
+```
+
+### Data & Analytics
+```bash
+rtk json config.json # Structure without values
+rtk deps # Dependencies summary
+rtk env -f AWS # Filtered env vars
+rtk log app.log # Deduplicated logs
+rtk curl # Truncate + save full output
+rtk wget # Download, strip progress bars
+rtk summary # Heuristic summary
+rtk proxy # Raw passthrough + tracking
+```
+
+### Token Savings Analytics
+```bash
+rtk gain # Summary stats
+rtk gain --graph # ASCII graph (last 30 days)
+rtk gain --history # Recent command history
+rtk gain --daily # Day-by-day breakdown
+rtk gain --all --format json # JSON export for dashboards
+
+rtk discover # Find missed savings opportunities
+rtk discover --all --since 7 # All projects, last 7 days
+
+rtk session # Show RTK adoption across recent sessions
+```
+
+## Global Flags
+
+```bash
+-u, --ultra-compact # ASCII icons, inline format (further output reduction)
+-v, --verbose # Increase verbosity (-v, -vv, -vvv)
+```
+
+## Examples
+
+**Directory listing:**
+```
+# ls -la (45 lines) # rtk ls (12 lines)
+drwxr-xr-x 15 user staff 480 ... my-project/
+-rw-r--r-- 1 user staff 1234 ... +-- src/ (8 files)
+... | +-- main.rs
+ +-- Cargo.toml
+```
+
+**Git operations:**
+```
+# git push (15 lines) # rtk git push (1 line)
+Enumerating objects: 5, done. ok main
+Counting objects: 100% (5/5), done.
+Delta compression using up to 8 threads
+...
+```
+
+**Test output:**
+```
+# cargo test (200+ lines on failure) # rtk test cargo test (~20 lines)
+running 15 tests FAILED: 2/15 tests
+test utils::test_parse ... ok test_edge_case: assertion failed
+test utils::test_format ... ok test_overflow: panic at utils.rs:18
+...
+```
+
+## Auto-Rewrite Hook
+
+The most effective way to use rtk. The hook transparently intercepts Bash commands and rewrites them to rtk equivalents before execution.
+
+**Result**: 100% rtk adoption across all conversations and subagents, with no per-command context overhead.
+
+**Scope note:** this only applies to Bash tool calls. Claude Code built-in tools such as `Read`, `Grep`, and `Glob` bypass the hook, so use shell commands or explicit `rtk` commands when you want RTK filtering there.
+
+### Setup
+
+```bash
+rtk init -g # Install hook + RTK.md (recommended)
+rtk init -g --opencode # OpenCode plugin (instead of Claude Code)
+rtk init -g --auto-patch # Non-interactive (CI/CD)
+rtk init -g --hook-only # Hook only, no RTK.md
+rtk init --show # Verify installation
+```
+
+After install, **restart Claude Code**.
+
+## Windows
+
+RTK works fully on native Windows. Since **v0.37.2** the auto-rewrite hook runs as a **native binary command** (`rtk hook claude`) — no Unix shell, bash, or jq required — so commands are rewritten transparently on Command Prompt, PowerShell, and Windows Terminal, just like on Linux and macOS.
+
+### Native Windows
+
+```powershell
+# 1. Download and extract rtk-x86_64-pc-windows-msvc.zip from releases
+# 2. Add rtk.exe to your PATH (e.g. C:\Users\\.local\bin)
+# 3. Initialize — installs the native binary hook
+rtk init -g
+```
+
+**Upgrading from an older install?** If you set RTK up before v0.37.2 you may still have the legacy `rtk-rewrite.sh` shell hook (which does need a Unix shell). Re-run `rtk init -g` to migrate to the native binary hook.
+
+**Prerequisites**: some filters shell out to [ripgrep](https://github.com/BurntSushi/ripgrep) (`rg`). Install it and keep it on your PATH (e.g. `winget install BurntSushi.ripgrep.MSVC`) to avoid `Binary 'rg' not found on PATH` warnings.
+
+**Important**: Do not double-click `rtk.exe` — it is a CLI tool that prints usage and exits immediately. Always run it from a terminal (Command Prompt, PowerShell, or Windows Terminal).
+
+### WSL
+
+[WSL](https://learn.microsoft.com/en-us/windows/wsl/install) also works and behaves exactly like Linux:
+
+```bash
+# Inside WSL
+curl -fsSL https://raw.githubusercontent.com/rtk-ai/rtk/refs/heads/master/install.sh | sh
+rtk init -g
+```
+
+| Feature | Native Windows | WSL |
+|---------|----------------|-----|
+| Filters (cargo, git, etc.) | Full | Full |
+| Auto-rewrite hook | Yes (native binary) | Yes |
+| `rtk init -g` | Hook mode | Hook mode |
+| `rtk gain` / analytics | Full | Full |
+
+## Supported AI Tools
+
+RTK supports 16 AI coding tools. Each integration rewrites shell commands to `rtk` equivalents, reducing the bash output the agent reads where the agent supports command interception.
+
+| Tool | Install | Method |
+|------|---------|--------|
+| **Claude Code** | `rtk init -g` | PreToolUse hook (native binary) |
+| **GitHub Copilot (VS Code)** | `rtk init -g --copilot` | PreToolUse hook — transparent rewrite |
+| **GitHub Copilot CLI** | `rtk init -g --copilot` | PreToolUse deny-with-suggestion (CLI limitation) |
+| **Cursor** | `rtk init -g --agent cursor` | preToolUse hook (hooks.json) |
+| **Gemini CLI** | `rtk init -g --gemini` | BeforeTool hook |
+| **Codex** | `rtk init -g --codex` | AGENTS.md + RTK.md instructions |
+| **Windsurf** | `rtk init -g --agent windsurf` | .windsurfrules (project-scoped) |
+| **Cline / Roo Code** | `rtk init --agent cline` | .clinerules (project-scoped) |
+| **OpenCode** | `rtk init -g --opencode` | Plugin TS (tool.execute.before) |
+| **OpenClaw** | `openclaw plugins install ./openclaw` | Plugin TS (before_tool_call) |
+| **Pi** | `rtk init -g --agent pi` (global) | TypeScript extension (tool_call) |
+| **Hermes** | `rtk init --agent hermes` | Python plugin adapter (terminal command mutation via `rtk rewrite`) |
+| **Mistral Vibe** | `rtk init -g --agent vibe` | `pre_tool` hook (hooks.toml) |
+| **Kilo Code** | `rtk init --agent kilocode` | .kilocode/rules/rtk-rules.md (project-scoped) |
+| **Google Antigravity** | `rtk init --agent antigravity` | .agents/rules/antigravity-rtk-rules.md (project-scoped) |
+| **Kimi AI** | `rtk init --agent kimi` | AGENTS.md (project-scoped) |
+| **Factory Droid** | `rtk init -g --agent droid` (or per-project) | PreToolUse hook in `~/.factory/hooks.json` (matcher `Execute`) |
+
+For per-agent setup details, override controls, and graceful degradation, see the [Supported Agents guide](https://www.rtk-ai.app/guide/getting-started/supported-agents). The Hermes plugin source and tests live in `hooks/hermes/`; installed Hermes runtime files still live under `~/.hermes/plugins/rtk-rewrite/`.
+
+## Configuration
+
+`~/.config/rtk/config.toml` (macOS: `~/Library/Application Support/rtk/config.toml`):
+
+```toml
+[hooks]
+exclude_commands = ["curl", "playwright"] # skip rewrite for these
+
+[tee]
+enabled = true # save raw output on failure (default: true)
+mode = "failures" # "failures", "always", or "never"
+```
+
+When a command fails, RTK saves the full unfiltered output so the LLM can read it without re-executing:
+
+```
+FAILED: 2/15 tests
+[full output: ~/.local/share/rtk/tee/1707753600_cargo_test.log]
+```
+
+For the full config reference (all sections, env vars, per-project filters), see the [Configuration guide](https://www.rtk-ai.app/guide/getting-started/configuration).
+
+### Uninstall
+
+```bash
+rtk init -g --uninstall # Remove hook, RTK.md, settings.json entry
+cargo uninstall rtk # Remove binary
+brew uninstall rtk # If installed via Homebrew
+```
+
+## Documentation
+
+- **[rtk-ai.app/guide](https://www.rtk-ai.app/guide)** — full user guide (installation, supported agents, what gets optimized, analytics, configuration, troubleshooting)
+- **[INSTALL.md](INSTALL.md)** — detailed installation reference
+- **[ARCHITECTURE.md](docs/contributing/ARCHITECTURE.md)** — system design and technical decisions
+- **[CONTRIBUTING.md](CONTRIBUTING.md)** — contribution guide
+- **[SECURITY.md](SECURITY.md)** — security policy
+
+## Privacy & Telemetry
+
+RTK can collect **anonymous, aggregate usage metrics** once per day. Telemetry is **disabled by default** and requires **explicit opt-in consent** (GDPR Art. 6, 7) during `rtk init` or via `rtk telemetry enable`. This data helps us build a better product: identifying which commands need filters, which filters need improvement, and how much value RTK delivers. For the full list of fields, data handling, and contributor guidelines, see **[docs/TELEMETRY.md](docs/TELEMETRY.md)**.
+
+**What is collected and why:**
+
+| Category | Data | Why |
+|----------|------|-----|
+| Identity | Salted device hash (SHA-256, not reversible) | Count unique installations without tracking individuals |
+| Environment | RTK version, OS, architecture, install method | Know which platforms to support and test |
+| Usage volume | Command count (24h), total commands, estimated tokens saved (24h/30d/total) | Measure adoption and value delivered |
+| Quality | Top 5 passthrough commands (0% reduction), parse failure count, commands with <30% reduction | Identify missing filters and weak ones to improve |
+| Ecosystem | Command category distribution (e.g. git 45%, cargo 20%, js 15%) | Prioritize filter development for popular ecosystems |
+| Retention | Days since first use, active days in last 30 | Understand engagement and detect churn |
+| Adoption | AI agent hook type (claude/gemini/codex), custom TOML filter count | Track integration coverage and DSL adoption |
+| Configuration | Whether config.toml exists, number of excluded commands, project count | Understand user maturity and customization patterns |
+| Features | Usage counts for meta-commands (gain, discover, proxy, verify) | Know which RTK features are valued vs unused |
+| Economics | Estimated USD value, derived from the estimated tokens saved and a fixed internal constant | Quantify the value RTK provides to users |
+
+All data is **aggregate counts or anonymized command names** (first 3 words, no arguments). Top commands report only tool names (e.g. "git", "cargo"), never full command lines.
+
+**What is NOT collected:** source code, file paths, command arguments, secrets, environment variables, personal data, or repository contents.
+
+**Manage telemetry:**
+```bash
+rtk telemetry status # Check current consent state
+rtk telemetry enable # Give consent (interactive prompt)
+rtk telemetry disable # Withdraw consent — stops all collection immediately
+rtk telemetry forget # Withdraw consent + delete all local data + request server-side erasure
+```
+
+**Override via environment:**
+```bash
+export RTK_TELEMETRY_DISABLED=1 # Blocks telemetry regardless of consent
+```
+
+## Star History
+
+
+
+
+
+
+
+
+
+## StarMapper
+
+
+
+
+
+
+
+
+
+## Core team
+
+- **Patrick Szymkowiak** — Founder
+ [GitHub](https://github.com/pszymkowiak) · [LinkedIn](https://www.linkedin.com/in/patrick-szymkowiak/)
+- **Florian Bruniaux** — Core contributor
+ [GitHub](https://github.com/FlorianBruniaux) · [LinkedIn](https://www.linkedin.com/in/florian-bruniaux-43408b83/)
+- **Adrien Eppling** — Core contributor
+ [GitHub](https://github.com/aeppling) · [LinkedIn](https://www.linkedin.com/in/adrien-eppling/)
+- **Nicolas Le Cam** — Core contributor
+ [Github](https://github.com/kush) · [LinkedIn](https://www.linkedin.com/in/nicolas-le-cam-386387160/)
+- **Takayuki Maeda** — Core contributor
+ [GitHub](https://github.com/TaKO8Ki) · [LinkedIn](https://www.linkedin.com/in/tako8ki/)
+
+## Contributing
+
+Contributions welcome! Please open an issue or PR on [GitHub](https://github.com/rtk-ai/rtk).
+
+Join the community on [Discord](https://discord.gg/RySmvNF5kF).
+
+## License
+
+Apache License 2.0 - see [LICENSE](LICENSE) for details.
+
+## Disclaimer
+
+See [DISCLAIMER.md](DISCLAIMER.md).
diff --git a/src/content/ensayos/el-stack-de-infraestructura-para-coding-agents-en-2026.md b/src/content/ensayos/el-stack-de-infraestructura-para-coding-agents-en-2026.md
new file mode 100644
index 0000000..3a4df28
--- /dev/null
+++ b/src/content/ensayos/el-stack-de-infraestructura-para-coding-agents-en-2026.md
@@ -0,0 +1,377 @@
+---
+title: "El stack de infraestructura para coding agents en 2026: schema, routing, framework y economía de tokens"
+description: "LangChain y OpenAI ya no bastan. En 2026, cualquier setup serio de coding agents necesita cuatro capas resolviendo problemas distintos: contratos de salida, traducción entre APIs, framework de aplicación y recorte del contexto. Repaso por las piezas OSS más veteranas y más recientes del momento."
+fecha: 2026-08-12
+tags: ["ia", "agentes", "infraestructura", "llm", "baml", "switchyard", "embabel", "rtk"]
+tipo: ensayo
+autor: "Alejandro de la Fuente"
+---
+
+Hace un año, montar un coding agent era cuestión de tres líneas: eliges modelo, eliges framework, eliges prompt. Hoy esa simplicidad se ha evaporado. En su lugar hay un ecosistema que se parece más al de una plataforma backend tradicional —con su capa de esquemas, su capa de traducción de protocolos, su capa de aplicación y su capa de costes— que al script minimalista que arrancó todo esto.
+
+Este artículo es un mapa de las cuatro capas que necesitas si quieres que tu setup agent aguante producción más de un trimestre. No es un tutorial. Es la respuesta a una pregunta que llevo semanas haciéndome: **¿qué huecos quedaron entre LangChain y el LLM, y quién los está cubriendo?**
+
+La respuesta corta: cuatro huecos, cuatro categorías de herramientas, todas ellas con tracción real en 2026.
+
+## El problema que ya tienes y no ves
+
+Si tu coding agent actual es "LangChain + OpenAI directo" o "Claude Code apuntando a Anthropic", tienes tres puntos ciegos que probablemente no has cuantificado:
+
+1. **No controlas el contrato de salida.** Le pides al LLM que devuelva un JSON con campos concretos. Te devuelve algo parecido, con un campo que sobra, otro que falta, y el tipo mal en un porcentaje variable de los casos (medido en benchmarks de tool calling entre 5-15% según el modelo y la complejidad del schema). Le añades un retry. El retry cuesta tokens. Los tokens cuestan dinero. Nadie mira ese porcentaje hasta que se convierte en un caso de soporte.
+2. **No controlas el modelo que responde.** Cuando Anthropic tiene outage, tu agente se para. Cuando OpenAI deprecia un modelo, refactorizas el código. Si quieres cambiar de proveedor porque bajó el precio, descubres que cada agente habla su propia API.
+3. **No controlas el contexto que le llega al LLM.** Tu agente ejecuta `git status` y el output ocupa 400 tokens. Ejecuta `cargo test` y le llegan 4.000 líneas de output que el agente tiene que masticar. Cada interacción cuesta más de lo que debería.
+
+Ninguno de los tres es un fallo del modelo. Son huecos del stack. Y en 2026 ya hay piezas OSS que los cubren, una por capa.
+
+## Las cuatro capas
+
+El stack mínimo de un coding agent serio en 2026 tiene esta forma:
+
+```
+[framework de aplicación] ← Embabel / LangGraph / LangChain
+ ↓
+[capa de routing] ← Switchyard / LiteLLM
+ ↓
+[capa de schema/contrato] ← BAML / instructor / lm-format-enforcer
+ ↓
+[proveedor LLM] ← Anthropic, OpenAI, vLLM, Ollama
+ ↓
+[capa de economía de tokens] ← rtk (recorta output de bash)
+```
+
+Las cuatro capas son ortogonales entre sí. Puedes empezar por una y añadir las demás cuando duela. Y la mayoría de setups que conozco hoy solo tienen la del framework — por eso fallan en producción.
+
+---
+
+## Capa 1 — Schema y contrato: BAML como DSL de agentes
+
+[`BoundaryML/baml`](https://github.com/BoundaryML/baml) (8.927★, Apache-2.0, creado en 2023) es el veterano del grupo. Tres años sin venderse, sin pivotes, sin hype. Es un lenguaje de programación específicamente diseñado para que un LLM cometa menos errores al devolverte datos estructurados.
+
+El ángulo es este: en lugar de pedirle al modelo "dame un JSON con estos campos", declaras una función en BAML con tipos como los de Rust. BAML genera automáticamente la gramática que constrains al modelo a emitir solo outputs válidos. Cuando el modelo invoca tu función, los argumentos llegan ya tipados, validados, listos para usar en Python, TypeScript, Go, Ruby, Java o C#.
+
+```baml
+// excerpt.baml
+function ClassifySupportTicket(ticket: string) -> Category {
+ category Category {
+ area "billing" | "technical" | "account" | "other"
+ urgency 1 | 2 | 3 | 4 | 5
+ needs_human bool
+ }
+
+ client GPT4
+ prompt #"
+ Classify this support ticket:
+ {{ ticket }}
+
+ {{ ctx.output_format }}
+ "#
+}
+```
+
+Lo que el código de arriba declara: una función `ClassifySupportTicket` que toma texto, devuelve un objeto con tres campos (área, urgencia 1-5, y un booleano). El LLM nunca puede devolver `urgency: "alta"` o `urgency: 7` — la gramática generada por BAML lo bloquea a nivel de token. Tu código Python recibe un objeto tipado, no un `dict` que validar a mano.
+
+**Por qué importa para tu setup.** El problema del JSON que casi encaja está resuelto por docenas de librerías — `instructor` en Python, `lm-format-enforcer`, `outlines`, `guidance`. BAML no es la única solución. Es la más veterana, la que más lenguajes de salida cubre, y la que integra un framework de tests y un DSL completo en vez de ser una librería de validación.
+
+### Quick start aislado (solo BAML)
+
+```bash
+# Instalar
+brew install baml
+
+# En tu proyecto Python
+pip install baml-py
+
+# Inicializar y declarar una función
+baml init
+# Editar baml_src/main.baml con la función ClassifySupportTicket
+baml generate
+```
+
+```python
+from baml_client import b
+result = b.ClassifySupportTicket(ticket="Llevo 3 días sin acceso")
+print(result.urgency) # -> 3
+```
+
+Si esto funciona, tienes la capa 1 funcionando en 10 minutos. Si no, el problema está en tu instalación de BAML, no en el resto del stack.
+
+**Cuándo no lo necesitas.** Si solo tienes 2-3 funciones tipadas y la mayoría de las llamadas las haces con `gpt-4o-mini`, `instructor` en Python hace lo mismo con menos ceremonia. BAML brilla cuando tienes docenas de funciones y necesitas generar clientes a varios lenguajes.
+
+---
+
+## Capa 2 — Routing y traducción: Switchyard como proxy institucional
+
+[`NVIDIA-NeMo/Switchyard`](https://github.com/NVIDIA-NeMo/Switchyard) (656★, Apache-2.0, **pre-alpha** — estado del proyecto más temprano que alpha; API y comportamiento pueden cambiar sin aviso) es el recién llegado.
+
+El hueco que cubre: cuando lanzas Claude Code, Codex CLI o cualquier coding agent, ese agente habla una API concreta (Anthropic Messages, OpenAI Chat Completions, OpenAI Responses). Si quieres que el request termine siendo servido por un modelo OSS local — vLLM (motor de inferencia open source), NVIDIA NIM (microservicio de inferencia optimizado para GPUs NVIDIA), Ollama (runner local de modelos) — necesitas un proxy que traduzca entre formatos. Switchyard hace exactamente eso, con un plus: registra métricas Prometheus (estándar de facto para monitorizar servicios en producción) y soporta varios algoritmos de routing componibles.
+
+```bash
+# instalar
+uv tool install --python 3.12 "nemo-switchyard[cli]"
+
+# configurar OpenRouter (agregador que da una sola API para acceder a
+# múltiples proveedores LLM: Anthropic, OpenAI, Google, etc.)
+export OPENROUTER_API_KEY="sk-or-..."
+switchyard launch claude --model switchyard
+
+# lanzar Codex a través del mismo proxy
+switchyard launch codex --model switchyard
+```
+
+El caso de uso real: tienes un equipo de desarrolladores. Una parte usa Claude Code, otra usa Codex CLI, otra usa OpenClaw (el espacio personal AI assistant). Cada uno configurado para hablar con su proveedor por defecto. Cuando un proveedor tiene un incidente, todos se paran. Cuando un proveedor sube precios, todos refactorizan.
+
+Con Switchyard delante, todos apuntan al mismo endpoint (`http://localhost:4000`). Cambias un TOML y el 80% del tráfico va a Anthropic, el 20% a un modelo local. El día que Anthropic tiene outage, rotas a OpenRouter con un cambio de config. Cada agente sigue hablando su API nativa; Switchyard traduce.
+
+**El caveat importante.** Switchyard está explícitamente marcado como **pre-alpha**. El README dice literalmente: "Experimental software. Not for production use". La API va a cambiar antes de 1.0.
+
+**Comparación con la competencia.** [`BerriAI/litellm`](https://github.com/BerriAI/litellm) lleva dos años haciendo esto en Python, tiene más de 10.000 estrellas y una tracción muy superior a la de Switchyard a día de hoy (656★ es el 5-6% de LiteLLM). Lo que Switchyard aporta sobre LiteLLM: implementación en Rust (latencia más baja, menor overhead) y el respaldo institucional de NVIDIA. Compra Switchyard si la latencia de proxy o la supervivencia a largo plazo del vendor te importan más que la estabilidad inmediata. Para la mayoría de equipos Python hoy, LiteLLM es la opción más segura.
+
+### Quick start aislado (solo Switchyard)
+
+Necesitas tres cosas: la tool instalada, una clave de API válida, y un fichero `routes.toml`. La sintaxis real de Switchyard es más rica que LiteLLM; la documentación oficial está en `docs/getting_started.md` del repo.
+
+```toml
+# routes.toml — ejemplo mínimo con primaria Anthropic + fallback OpenRouter
+# Cada bloque [model.] define un target concreto al que Switchyard sabe hablar.
+
+[model.claude-sonnet]
+provider = "anthropic"
+api_key = "${ANTHROPIC_API_KEY}" # se lee de la variable de entorno
+
+[model.openrouter-mix]
+provider = "openai_compatible"
+base_url = "https://openrouter.ai/api/v1"
+api_key = "${OPENROUTER_API_KEY}"
+
+[route.default]
+# Passthrough: enruta todo al modelo "claude-sonnet".
+# Cambia el target a "openrouter-mix" para hacer fallback.
+type = "passthrough"
+target = "claude-sonnet"
+
+[route.fallback]
+# Si "default" devuelve error, enruta a OpenRouter.
+type = "passthrough"
+target = "openrouter-mix"
+```
+
+```bash
+# arrancar el proxy
+export ANTHROPIC_API_KEY="sk-ant-..."
+export OPENROUTER_API_KEY="sk-or-..."
+switchyard-server --config routes.toml --host 127.0.0.1 --port 4000
+
+# en otra terminal: verificar
+curl http://localhost:4000/health
+```
+
+**Variables de entorno por coding agent.** Switchyard soporta los agentes más comunes pero cada uno necesita su variable apuntando al proxy:
+
+| Coding agent | Variable de entorno | Notas |
+|---|---|---|
+| Claude Code | `ANTHROPIC_BASE_URL=http://localhost:4000` | Conserva `ANTHROPIC_API_KEY` o usa la del proxy |
+| Codex CLI | `OPENAI_BASE_URL=http://localhost:4000/v1` | Más `OPENAI_API_KEY` apuntando a una clave dummy |
+| OpenClaw | Sigue la convención OpenAI | `OPENAI_BASE_URL` |
+| Gemini CLI | `GOOGLE_API_BASE` o `OPENAI_BASE_URL` según modo | Ver docs |
+
+Si esto funciona y `curl http://localhost:4000/health` responde `200 OK`, tienes la capa 2 funcionando. Si no, el problema está en `routes.toml` o en las variables de entorno.
+
+---
+
+## Capa 3 — Framework de aplicación: Embabel como respuesta JVM
+
+[`embabel/embabel-agent`](https://github.com/embabel/embabel-agent) (4.175★, Apache-2.0) es la pieza que más me ha hecho pensar este mes.
+
+El dato relevante: Rod Johnson, autor de Spring, lidera el proyecto. Spring es el framework que sostiene la mayor parte del enterprise Java del planeta. Cuando un perfil así decide construir un framework de agentes para JVM, no es un random jugando a los wrappers de LangChain. El framework hereda los patrones de inyección de dependencias (el sistema te pasa los objetos que necesitas, no los buscas tú), AOP (puedes añadir comportamiento a métodos sin tocarlos) y transacciones de Spring. Eso le da una base sólida que otros frameworks de agentes no tienen.
+
+El hueco que Embabel cubre es específico y real: **la mayoría de empresas grandes del mundo corren JVM**. Su código no está en Python. Su equipo no va a migrar a Python para usar LangChain. Y hasta ahora, las alternativas serias para ellos eran Semantic Kernel (de Microsoft, tira a .NET/C#) y Spring AI (de Pivotal/VMware, integración directa con Spring). Embabel es la apuesta más ambiciosa: un framework agent-native, no un wrapper de LLM sobre Spring.
+
+> **Si tu equipo es 100% Python, sáltate esta sección.** Embabel no te concierne salvo que estéis evaluando mover carga a JVM. Para ti, LangChain, LangGraph o CrewAI son opciones equivalentes con menos fricción.
+
+Lo que aporta técnicamente:
+
+```kotlin
+// El plan se formula dinámicamente, no lo escribes tú
+@Agent
+class TravelPlannerAgent {
+ @Goal
+ fun planTrip(request: TripRequest): TripPlan { /* ... */ }
+
+ @Action
+ fun searchFlights(origin: String, destination: String): List { /* ... */ }
+
+ @Action
+ fun bookHotel(flight: Flight): Hotel { /* ... */ }
+}
+```
+
+El framework usa GOAP (Goal Oriented Action Planning) por defecto. GOAP es un algoritmo clásico de planificación que viene de los juegos: tú declaras un estado actual, una meta, y un conjunto de acciones con precondiciones y efectos. El algoritmo busca la secuencia de acciones que lleva del estado actual a la meta. A diferencia de ReAct (donde el LLM decide paso a paso qué hacer y razona sobre cada observación) o plan-and-execute (donde un LLM genera un plan completo upfront), GOAP busca el plan dinámicamente usando un algoritmo determinista — el LLM solo se invoca para ejecutar las acciones que requieren comprensión, no para decidir el orden. Esto lo hace más robusto y barato cuando las acciones tienen efectos claros sobre el estado.
+
+**Por qué importa para enterprise.** Tres cosas que el stack Python no te da gratis:
+
+1. **Tipado fuerte.** Acciones, metas, condiciones y planes están todos tipados. Refactor seguro. Si cambias el dominio, el compilador te dice qué acciones quedan rotas.
+2. **Integración nativa con Spring.** Inyección de dependencias, AOP, transacciones, persistencia. Todo lo que ya tienes en tu stack funciona.
+3. **Modos de ejecución.** Focused (codriven), Closed (clasificación a un agente), Open (la plataforma busca el goal y construye el agente). Open mode es el más potente y el menos determinista; usa con cuidado.
+
+**Cuándo no lo necesitas.** Si tu equipo es de tres personas en Python, Embabel no compite con LangChain ni tiene por qué. Embabel brilla cuando tienes 50 desarrolladores Java y necesitas que adopten agentes sin tirar el stack actual.
+
+---
+
+## Capa 4 — Economía de tokens: rtk como proxy de salida
+
+[`rtk-ai/rtk`](https://github.com/rtk-ai/rtk) (75.861★, Apache-2.0) ha crecido de forma sostenida: de 72.400 a 75.800 estrellas en tres días a mediados de agosto de 2026.
+
+El problema es simple y cuantificable. Cuando tu coding agent ejecuta `git status`, el output es razonable. Cuando ejecuta `cargo test` y hay 200 tests pasando, el output son 4.000 líneas de ruido. Cuando ejecuta `ls -la` en un repo grande, son 800 entradas. Cada una de esas líneas va al contexto del LLM. Cada token cuenta.
+
+rtk se interpone entre el shell y el agente y filtra el output antes de que llegue al contexto:
+
+```bash
+# instalar
+brew install rtk
+
+# activar para Claude Code
+rtk init -g
+rtk init -g --codex
+rtk init -g --gemini
+rtk init --agent hermes # ← funciona con Hermes
+rtk init --agent cursor
+```
+
+A partir de ese momento, cuando el agente ejecuta `git status`, rtk reescribe el comando a `rtk git status` antes de pasarlo al shell. El output que vuelve al contexto es compacto:
+
+| Comando sin rtk | Con rtk |
+|---|---|
+| `git status` | Stat por estado, agrupado |
+| `ls -la` | Tree con contadores |
+| `cat archivo_largo.py` | Solo signatures y estructura |
+| `cargo test` (200 OK) | "200 passed" |
+| `pytest` (mismos) | "12 passed" + traceback de los que fallan |
+
+La métrica que cita el README de rtk: hasta un 90% de reducción del output de bash. Pero el propio repo aclara (en `docs/guide/resources/savings-explained.md`): **esa cifra mide reducción del output de bash, no reducción de la factura completa**. El output de bash es un input más; el sistema prompt, el historial y el output del modelo también cuentan. La reducción real de factura es menor. Aún así, en un setup donde cada sesión cuesta 2-5€ en tokens, recortar un 30-50% del input no es trivial.
+
+**Por qué importa más de lo que parece.** Los modelos grandes de 2026 son buenos ignorando ruido. Pero "buenos ignorando ruido" no es "gratis ignorar ruido". El README de rtk afirma que recortar el ruido también sube la calidad de las respuestas, especialmente en sesiones largas. Es plausible — más señal por token dedicado a la ventana de contexto debería ayudar al modelo a enfocarse — pero es un claim de marketing, no un benchmark publicado. Tómalo como hipótesis a validar con tus propias métricas, no como hecho.
+
+**Cuándo no lo necesitas.** Si ejecutas comandos de forma interactiva y no con un agente, rtk te estorba (el output recortado es menos legible para humanos). Si tu agente hace dos comandos por sesión, el ROI es marginal. Si tu agente hace 50+ comandos por sesión, rtk es la palanca con más retorno por hora invertida de toda la lista.
+
+---
+
+## Cómo se conectan las cuatro capas
+
+Un setup mínimo que cubra las cuatro, end-to-end, en local:
+
+```bash
+# 1. Instalar rtk y activarlo para tu coding agent
+brew install rtk
+rtk init --agent claude
+
+# 2. Arrancar Switchyard como proxy
+uv tool install --python 3.12 "nemo-switchyard[cli]"
+export OPENROUTER_API_KEY="..."
+switchyard-server --config routes.toml --host 127.0.0.1 --port 4000
+
+# 3. Apuntar Claude Code a Switchyard
+export ANTHROPIC_BASE_URL=http://localhost:4000
+claude
+
+# 4. Declarar tus funciones con BAML
+baml init
+# editar baml_src/main.baml con tus funciones
+baml generate
+
+# 5. Importar desde Python
+# from baml_client import b
+# result = b.ClassifySupportTicket(ticket="...")
+```
+
+El flujo: Claude Code habla Anthropic Messages → Switchyard traduce a OpenAI Chat Completions → el modelo (vLLM, Ollama, OpenAI, Anthropic directo vía OpenRouter) responde → Switchyard traduce de vuelta → BAML valida la estructura final → rtk recortó el output de bash que llenó el contexto.
+
+Cuatro capas independientes. Cada una se puede quitar o sustituir sin romper las demás.
+
+## El grafo de infraestructura que se está formando
+
+Si te paras a mirar el trending de GitHub en las últimas semanas, la categoría "infraestructura para coding agents" se ha consolidado con nombres reconocibles:
+
+| Capa | Veteranos 2024 | Recién llegados 2026 |
+|---|---|---|
+| Schema/contrato | instructor, outlines, guidance | **BAML** (8.9k★) |
+| Routing | LiteLLM, Portkey | **Switchyard** (NVIDIA oficial), codex-router |
+| Framework | LangChain, LangGraph, CrewAI, AutoGen | **Embabel** (JVM), OpenClaw, Waku |
+| Economía de tokens | (vacío) | **rtk** (75k★, 0→75k en meses) |
+
+Lo que me llama la atención es la fila de abajo. **rtk ha popularizado una categoría que no existía**: el proxy de salida para reducir el coste por sesión del agente. Antes de rtk, la conversación sobre reducir tokens se centraba en prompts más cortos o modelos más baratos; nadie había puesto el foco en el output de bash. El éxito del repo demuestra que el hueco existía — y que la solución obvia en retrospectiva ("recortar el output antes de que llegue al contexto") resuelve un problema real.
+
+Y la fila del framework: la JVM por fin tiene una respuesta seria, y OpenClaw/Waku entran como alternativas desktop-first al stack web (LangGraph Studio, AgentKit). El espacio se está diversificando.
+
+## Cuándo NO necesitas este stack
+
+Honestidad. La mayoría de proyectos personales y prototipos no necesitan nada de esto. Si tu agente hace 5-10 llamadas al LLM por sesión, con funciones simples, y no tienes usuarios reales pendientes del coste, te basta con:
+
+- `instructor` (Python) en vez de BAML
+- Nada en vez de Switchyard (apunta directo al proveedor)
+- LangChain en vez de Embabel
+- Nada en vez de rtk
+
+**Caso concreto del lector probable:** equipo de 6-12 devs Python con mix OpenAI + Anthropic, sin presupuesto para JVM ni para Rust. Tu setup mínimo sensato hoy es:
+
+- BAML o `instructor` para las 5-10 funciones críticas (las que afectan a datos de usuarios o que llaman APIs externas).
+- LiteLLM como proxy (no Switchyard): más tracción, Python-native, API estable.
+- LangChain o LangGraph como framework de aplicación (no Embabel: irrelevante en Python).
+- rtk como capa de ahorro de tokens. Esta sí aplica a todo el mundo.
+
+Eso te da el 80% del valor con piezas estables. Switchyard y Embabel son para cuando ya tienes lo anterior funcionando y el problema siguiente es otro: outage de proveedor (entonces evalúas Switchyard) o mover carga a JVM (entonces evalúas Embabel).
+
+La señal de que necesitas el stack completo es una combinación de:
+
+- **Volumen.** Más de 50 invocaciones LLM al día, o sesiones largas (multi-turn con historial).
+- **Criticidad.** El output del agente afecta a usuarios reales (no solo a ti en un script).
+- **Coste.** Estás viendo la factura subir y necesitas optimizarla sin cambiar de modelo.
+- **Volatilidad.** Has tenido un outage de proveedor en los últimos 6 meses que te costó algo.
+
+Si marcas dos o más de esas, el stack vale la pena. Si no, estás añadiendo complejidad por adelantado.
+
+## Roadmap incremental
+
+La mayoría de setups no necesitan las cuatro capas desde el día uno. La progresión natural que veo en equipos reales:
+
+**Crawl (1-2 días):** Empieza con **rtk** solo. Instalación de 5 minutos, beneficio inmediato en sesiones largas. Si usas Claude Code, Codex, Cursor, Windsurf o Hermes, `rtk init` te lo activa.
+
+**Walk (1 semana):** Añade **BAML** o `instructor` para tus 5-10 funciones más críticas. Las que invocan APIs externas, las que afectan a tu base de datos, las que tienen branches de error no triviales. Deja las funciones simples (resúmenes, clasificaciones binarias) con JSON prompting normal.
+
+**Run (1-2 semanas):** Añade **Switchyard** o LiteLLM como proxy. Configura dos rutas: primaria (proveedor principal) y fallback (proveedor secundario o modelo local). Activa métricas Prometheus. Ahora tienes visibilidad de qué se gasta y dónde.
+
+**Run+ (1 mes):** Migra el framework a **Embabel** (si JVM) o consolida en LangGraph (si Python). En este punto ya tienes cuatro capas y un sistema que aguanta producción.
+
+## Modos de fallo y mitigaciones
+
+**Fallo 1: BAML genera gramáticas que rechazan outputs válidos.**
+Síntoma: tu agente falla más a menudo con BAML que sin él. Causa: la gramática generada es demasiado estricta para tu caso. Mitigación: usa `Field` para relajar constraints; empieza con solo `description` y ve añadiendo constraints de uno en uno.
+
+**Fallo 2: Switchyard añade latencia perceptible.**
+Síntoma: el agente tarda más en responder. Causa: dos saltos HTTP en vez de uno (agente → proxy → modelo). Mitigación: corre Switchyard en el mismo host, sin red. La latencia añadida debería ser <5ms. Si no, hay un bug.
+
+**Fallo 3: Embabel en modo Open genera planes absurdos.**
+Síntoma: el agente invoca acciones en un orden que no tiene sentido. Causa: el dominio tiene metas ambiguas o acciones con efectos secundarios no modelados. Mitigación: empieza en modo Focused o Closed, no Open. Open es para cuando ya conoces bien tu dominio.
+
+**Fallo 4: rtk rompe comandos que tu agente ejecuta.**
+Síntoma: el output filtrado pierde información que el agente necesita. Causa: rtk no conoce ese comando y lo pasa tal cual, o lo conoce pero filtra demasiado. Mitigación: usa `rtk read` o `rtk find` directamente cuando necesites el output completo; o añade un wrapper específico.
+
+## Cierre
+
+El ecosistema de coding agents en 2026 ya no es "framework + LLM". Es cuatro capas resolviendo problemas distintos, cada una con su propia historia y su propio vendor. BAML lleva tres años cubriendo el nicho de DSL para agentes. Switchyard lleva meses como el proxy institucional de NVIDIA (con todas las caveats de pre-alpha). Embabel lleva uno como la respuesta JVM al stack Python. rtk ha demostrado que la categoría "recorte del output de bash" tenía demanda real.
+
+Si tu setup actual tiene solo la capa de framework, no estás tarde. Pero cada mes que pasa, los huecos que cubren las otras tres capas cuestan más dinero y más tiempo de depuración. La pregunta no es si adoptarlas. Es en qué orden.
+
+Y si tu setup ya tiene las cuatro: probablemente no necesitas leer este artículo. Probablemente lo escribiste tú.
+
+---
+
+## Apéndice — Resumen de las cuatro piezas
+
+| Pieza | Capa | ★ | Licencia | Madurez | Instalación |
+|---|---|---|---|---|---|
+| [BAML](https://github.com/BoundaryML/baml) | Schema/contrato | 8.927 | Apache-2.0 | Estable (3 años) | `brew install baml` |
+| [Switchyard](https://github.com/NVIDIA-NeMo/Switchyard) | Routing | 656 | Apache-2.0 | Pre-alpha | `uv tool install "nemo-switchyard[cli]"` |
+| [Embabel](https://github.com/embabel/embabel-agent) | Framework JVM | 4.175 | Apache-2.0 | Estable (1 año) | Maven Central: `com.embabel.agent:embabel-agent-api` |
+| [rtk](https://github.com/rtk-ai/rtk) | Economía de tokens | 75.861 | Apache-2.0 | Estable | `brew install rtk` |
+
+Estrellas verificadas vía GitHub REST API el 2026-08-12. Madurez cualitativa basada en el README de cada proyecto. "Maven Central" (en la fila de Embabel) es el repositorio de paquetes estándar de JVM, equivalente a PyPI para Python.
diff --git a/src/content/ensayos/tiny-foundation-models-14-mb-45m-parametros-agente-bolsillo.md b/src/content/ensayos/tiny-foundation-models-14-mb-45m-parametros-agente-bolsillo.md
new file mode 100644
index 0000000..477d12e
--- /dev/null
+++ b/src/content/ensayos/tiny-foundation-models-14-mb-45m-parametros-agente-bolsillo.md
@@ -0,0 +1,331 @@
+---
+title: "Tiny foundation models: 14 MB, 45M parámetros y un agente entero en tu bolsillo"
+description: "El 'agente local' en 2026 ya no necesita una M3 Max ni un A100. cactus-compute/needle ejecuta tool calling en 28 MB de RAM con un binario de 14 MB. Análisis técnico del paper, ejemplos hands-on, comparativa con FunctionGemma, Apple FM y LFM2.5, y por qué este cambio de escala importa más de lo que parece."
+fecha: 2026-08-12
+tags: ["ia", "edge", "on-device", "tiny-ml", "agentes", "wearables", "cactus"]
+tipo: ensayo
+autor: "Alejandro de la Fuente"
+---
+
+El mes pasado estuve en una conversación con un equipo que está construyendo un asistente para personas mayores en smart speakers y gateways de hogar. El requisito era claro: el modelo tiene que correr en el dispositivo del usuario, sin red, sin mandar datos a la nube, y consumir menos batería que la linterna. Cualquier modelo de 7B parámetros quedaba descartado de entrada.
+
+Cuando les mencioné [`cactus-compute/needle`](https://github.com/cactus-compute/needle) —45 millones de parámetros, 14 MB de binario, 28 MB de RAM en sesión completa, soporte nativo de tool calling— la conversación cambió. Hasta entonces habían evaluado Phi-4 mini y modelos quantized de 7B sobre hardware edge, con resultados poco prometedores en latencia y batería. Por primera vez, el "asistente local con acciones" tenía un modelo específico para ese nicho.
+
+Este artículo es sobre lo que significa ese cambio de escala. Y sobre por qué, cuando hablamos de "agente en local", ya no nos referimos a lo mismo que hace un año.
+
+## La nueva frontera de tamaño
+
+Enero de 2025. "Agente local" significaba "LLaMA 7B cuantizado corriendo en una M3 Max". Funcionaba, pero exigía hardware específico, RAM abundante, y batería a tope. Cualquier despliegue real requería un servidor.
+
+Agosto de 2026. "Agente local" puede significar "modelo de 45M parámetros corriendo en un Pixel de gama media con 28 MB de RAM ocupados". La conversación sobre dónde corre el modelo ya no es principalmente técnica (qué SoC, qué runtime): es de producto (qué margen, qué regulación, qué experiencia offline, qué coste de actualización).
+
+La convergencia es esto: teléfonos, wearables, smart home y vehículos llevan años acumulando capacidad de cómputo y sensores. Lo que les faltaba era un modelo pequeño, ejecutable, con tool calling nativo. Needle 2 es la primera apuesta seria que cumple los cuatro requisitos a la vez.
+
+## Qué es un tiny foundation model y qué no es
+
+Antes de entrar en Needle específicamente, una definición operativa. Un **tiny foundation model** (TFM) en 2026 es un modelo que cumple cuatro condiciones:
+
+1. **Menos de 50M parámetros.** Por debajo de ese umbral, el binario cabe en la flash de dispositivos básicos y se puede distribuir como un asset normal.
+2. **Binario único.** No hay modelo por un lado, tokenizer por otro, y configuración por otro. Un solo archivo `.cact` o equivalente que se carga y se ejecuta.
+3. **Soporte nativo de tool calling y structured extraction.** No es un modelo de lenguaje al que le enchufas un wrapper. El modelo fue entrenado para emitir function calls y extraer JSON con gramática.
+4. **Memoria acotada y predecible.** La sesión completa —contexto, herramientas, KV cache— cabe en un orden de magnitud conocido (decenas de MB), no en "depende del prompt".
+
+Lo que un TFM **no es**:
+
+- No es un LLM grande cuantizado a 4 bits. Eso da un modelo de 4 GB con un comportamiento similar al original. Un TFM es arquitectura y entrenamiento distintos, no compresión.
+- No es un SLM (small language model) clásico tipo Phi-2 o TinyLlama. Esos son de 1-3B parámetros, diseñados para lenguaje general. Un TFM está especializado en una tarea (tool calling, extracción) y optimizado para footprint.
+- No es un modelo de embedding. No genera texto libre; emite calls o JSON estructurado. La diferencia importa para el diseño de producto.
+
+## Anatomía de cactus-compute/needle
+
+[`cactus-compute/needle`](https://github.com/cactus-compute/needle) (3.963★, MIT) es la implementación de referencia de Needle 2, documentada en el paper [arXiv:2607.18363](https://arxiv.org/abs/2607.18363). El README empieza con un resumen que merece la pena citar entero:
+
+> Needle 2 is an open 45M-parameter model for tool calling, device use and structured extraction. The whole model is a single 14MB binary that runs a full session in about 28MB of RAM. It is built on our Simple Attention Network findings, compressed to CQ2-bit with Cactus Quants, and baked into its own engine.
+
+Lo que eso significa pieza a pieza:
+
+| Métrica | Valor | Comparación |
+|---|---|---|
+| Parámetros | 45M | ~16x menor que FunctionGemma 270M |
+| Binario único | 14 MB | Cabe en la flash de cualquier dispositivo moderno |
+| RAM en sesión | ~28 MB | 28 MB totales, contexto incluido |
+| Cuantización | CQ2-bit (2 bits) | vs f16 de FunctionGemma |
+| Tool calling | Nativo, grammar-constrained | vs prompting con JSON |
+| Confidence scoring | Calibrado, aprendido | vs heurística |
+
+La arquitectura del modelo se llama **Simple Attention Network** y combina cuatro ingredientes:
+
+1. **Hadamard MLP en lugar de FFN.** Una matriz ortonormal fija de Walsh-Hadamard reemplaza la red feed-forward estándar. Sin pesos que cargar, se aplica en n log n. Reduce parámetros y cómputo sin perder capacidad.
+2. **GQA (Grouped Query Attention).** Múltiples heads de query comparten la misma key-value. Reduce la huella de la KV cache — clave para mantener la sesión en 28 MB.
+3. **Engram key-value memory.** Sitios a dos capas disparan filas (kₜ, vₜ) recuperadas de tablas hash de n-gramas. Memoria explícita inyectada en el cómputo, sin coste de inferencia sobre los n-gramas.
+4. **Multi-lane hyper-connections.** Varios streams residuales en paralelo, normalizados por Sinkhorn. Permite que el modelo combine información de varias rutas sin que una domine.
+
+El resultado: un modelo que en benchmarks está al nivel de FunctionGemma 270M (de Google) y Apple FM, siendo 5-70x más pequeño y cuantizado a 2 bits donde los demás están a f16.
+
+**Lo que el paper no te dice en el README, pero importa para tu producto:**
+
+- El contexto es una ventana deslizante de **256 tokens** con las herramientas fijadas como KV sinks. Por eso la memoria total no crece con la conversación. Sesión larga = mismo consumo de RAM.
+- El modelo **rechaza peticiones fuera de catálogo** devolviendo la lista vacía `[]`. No hay fallback a texto libre. Si declaras tres herramientas, el modelo solo puede llamar a esas tres.
+- El campo `reasoning` se genera sin restricción gramatical. Solo el `call` está constrained. La consecuencia: la derivación puede ser legible aunque el JSON de la call siempre esté bien formado.
+
+## Grammar-constrained decoding nativo y confidence scoring calibrado
+
+Dos features técnicas que parecen menores pero que cambian el diseño de producto.
+
+**Grammar-constrained decoding nativo.** Cuando declaras una herramienta con tipos, Needle genera automáticamente una gramática a nivel de byte. El decoder solo puede emitir tokens que produzcan strings válidos según esa gramática. Si declaras `Literal["heat", "cool", "auto"]` para un campo, el modelo no puede emitir `"heating"`. Si declaras `Annotated[float, Field(gt=0, le=10000)]` para amount, no puede emitir `-50` ni `10001`.
+
+Comparación: el approach alternativo es "pídele al LLM que devuelva JSON válido y luego valida en Python". Funciona, pero un 5-15% de las veces tienes que reintentar. Cada reintento cuesta tokens, latencia y batería. Con grammar-constrained decoding, el modelo emite JSON válido **a la primera**. La diferencia en latencia media es de 2-4x.
+
+```python
+# del README de Needle
+from typing import Annotated
+
+@needle.tool
+def send_money(
+ amount: Annotated[float, needle.Field(gt=0, le=10000, description="USD, hasta 10.000")],
+ to: Annotated[str, needle.Field(pattern=r"^@[a-z0-9_]+$", description="handle destinatario")],
+ memo: Annotated[str, needle.Field(max_length=80)] = "",
+):
+ """Send money to a handle."""
+ return {"sent": amount, "to": to}
+```
+
+El `pattern=r"^@[a-z0-9_]+$"` se compila a una gramática. El modelo emite handles válidos o no emite nada.
+
+**Confidence scoring calibrado.** Cada respuesta del modelo lleva un campo `confidence` entre 0 y 1. Es la mínima entre dos señales: una cabeza calibrada post-hoc que puntúa el prompt completo más la call producida, y la probabilidad de decoding de los tokens de la call. Si las dos señales no coinciden, la confianza baja.
+
+```json
+{
+ "type": "call",
+ "success": true,
+ "function_calls": [
+ {"name": "set_lights", "arguments": {"room": "living room", "on": true, "brightness": 30}}
+ ],
+ "reasoning": "'living room' -> room; 'dim' -> on true, brightness 30",
+ "confidence": 0.94
+}
+```
+
+El contrato es directo: fijas un threshold para tu producto (digamos 0.85). Por encima, ejecutas. Por debajo, escalas a un modelo más grande o repreguntas al usuario. Esto convierte "el modelo a veces falla" en "el modelo falla con probabilidad N, y la decisión de qué hacer la tomas tú".
+
+Para los que vienen de la generación de texto con LLMs grandes: esta es la diferencia entre "esperar que el modelo alucine" y "medir programáticamente la incertidumbre". El diseño de producto cambia completamente cuando tienes una señal cuantitativa.
+
+## Hands-on: `pip install cactus-needle`
+
+La forma más rápida de probar Needle es instalar el paquete Python y declarar dos o tres herramientas:
+
+```bash
+pip install cactus-needle
+```
+
+El primer arranque descarga los pesos desde HuggingFace (`Cactus-Compute/needle2`) y los cachea. A partir de ahí, todo es offline.
+
+```python
+import needle
+
+@needle.tool
+def get_weather(city: str):
+ """Get the current weather for a city."""
+ return {"city": city, "temp_c": 27, "sky": "clear"}
+
+agent = needle.Needle(tools=[get_weather])
+result = agent.run("qué tiempo hace ahora en Lagos")
+
+print(result["results"])
+# [{'city': 'Lagos', 'temp_c': 27, 'sky': 'clear'}]
+```
+
+El modelo lee la firma y el docstring, decide qué herramienta llamar, ejecuta la función con los argumentos que extrajo, y devuelve los resultados. Todo en local. La latencia medida en el README: **4300 tokens/s en prefill, 850 tokens/s en decode** en hardware de referencia (probablemente M-class).
+
+El caso más interesante es con varias herramientas y constraints:
+
+```python
+from typing import Literal
+
+@needle.tool
+def set_thermostat(
+ temperature: int,
+ mode: Literal["heat", "cool", "auto"] = "auto",
+):
+ """Configura el termostato.
+
+ Args:
+ temperature: temperatura objetivo en Celsius
+ mode: estrategia de climatización
+ """
+ return {"temperature": temperature, "mode": mode}
+
+agent = needle.Needle(tools=[set_thermostat])
+agent.run("ponlo a 21 y refresca la habitación")
+```
+
+El `Literal["heat", "cool", "auto"]` se compila a una gramática con tres opciones. El modelo **no puede** emitir `mode: "calor"`. Si lo intenta, el decoder lo bloquea a nivel de token. Es la diferencia entre "el modelo respeta el esquema" y "el esquema es físicamente imposible de violar".
+
+## El contexto: dos rutas al inference local-first
+
+Hay una conversación complementaria que merece la pena traer aquí. El espacio "inferencia local en dispositivo" se está consolidando con dos rutas distintas, que se complementan en lugar de competir:
+
+**Ruta A — Tool-calling edge (cactus-compute/needle).** Modelos especializados en emitir calls estructuradas. No generan texto libre. Optimizados para footprint mínimo y latencia predecible. Casos de uso: asistentes de dispositivo, smart home, asistentes para personas mayores, IoT industrial.
+
+**Ruta B — Multimodal edge (antirez/h3.c).** Modelos multimodales pequeños escritos en C puro, diseñados para correr en cualquier sitio (incluso microcontroladores). Procesan imagen y texto. No están optimizados para tool calling, sino para comprensión sensorial local.
+
+Las dos rutas atacan el mismo problema ("inferencia en el dispositivo") desde dos ángulos distintos. Needle es para cuando tu dispositivo necesita **actuar**. h3.c es para cuando tu dispositivo necesita **percibir**. En un producto real, probablemente acabas combinando las dos: h3.c para procesar la imagen del termómetro, Needle para decidir si subir o bajar la temperatura según el resultado.
+
+## Cuándo un tiny FM es la elección correcta
+
+La decisión no es técnica, es de producto. Un TFM es la elección correcta cuando cumples al menos dos de estas condiciones:
+
+**Restricción dura de privacidad.** Tu producto maneja datos que no pueden salir del dispositivo. Historial médico, finanzas personales, conversaciones íntimas. Cualquier modelo que toque un servidor remoto queda descartado por diseño.
+
+**Restricción dura de latencia.** Tu producto necesita responder en menos de 200ms. La latencia de red ya consume 50-150ms en buenas condiciones. Si el modelo corre en el dispositivo, te ahorras la red.
+
+**Restricción dura de coste.** Tu producto hace miles de inferencias por usuario al día. A 0.001€ por inferencia, mil inferencias son 1€ por usuario. Multiplica por 100k usuarios y tienes un problema de margen.
+
+**Restricción dura de hardware.** Tu dispositivo tiene margen para correr un binario de 14 MB y mantener la sesión en 28-64 MB de RAM. Esto incluye teléfonos de gama baja, smart speakers, gateways de smart home, vehículos con SoC dedicado, microcontroladores con PSRAM. Una pulsera o wearable compacto típico (Nordic nRF5340, Apollo4) tiene 256 KB-512 KB de RAM y 1-4 MB de flash: ahí un TFM de 28 MB no entra sin rediseñar el SoC. Si tu dispositivo está en ese rango, este artículo te dice qué hardware necesitas diseñar, no que el modelo actual entre.
+
+**Caso de uso narrow.** Tu producto hace una cosa concreta (clasificar tickets, extraer datos de facturas, controlar luces) y no necesita razonamiento general. Los TFM están optimizados para tareas específicas.
+
+Si marcas tres o más, el TFM no es solo la elección correcta — es probablemente la única opción viable.
+
+**Ejemplo aterrizado: pulsera para personas mayores.**
+
+| Condición | ¿Se cumple? | Notas |
+|---|---|---|
+| Restricción de privacidad | Sí | Datos médicos y de localización, no pueden salir del dispositivo |
+| Restricción de latencia | Sí | Alertas críticas en menos de 200ms |
+| Restricción de coste | Sí | 1.000 inferencias/día a 0.001€ = 1€/usuario/día; insostenible en cloud |
+| Restricción de hardware | **No** | Pulsera típica (Nordic nRF5340) tiene 256 KB-512 KB de RAM; Needle necesita 28 MB |
+| Caso de uso narrow | Sí | Alertas, recordatorios, llamadas; razonamiento general no requerido |
+
+Resultado: 4 de 5, pero la condición de hardware es bloqueadora. **Veredicto explícito: las pulseras compactas actuales no pueden correr Needle.** El caso sí encajaría en smart speakers, gateways de hogar, tablets fijas o teléfonos. Si tu producto es wearable barato, este artículo te sirve para diseñar el SoC del siguiente modelo, no para desplegar Needle hoy.
+
+## La frontier size-vs-quality
+
+Datos del README y del paper. Advertencia honesta: los benchmarks son del propio equipo (cactus-compute), así que tómalos con la cautela habitual. La columna "RAM en inferencia" incluye modelo + KV cache + overhead del runtime; "RAM del peso" es solo el binario del modelo. Las cifras de competidores son estimaciones de la documentación pública de cada uno, no medidas propias.
+
+| Modelo | Parámetros | Bit-width | RAM peso | RAM en inferencia | Tool calling nativo | Notas |
+|---|---|---|---|---|---|---|
+| **Needle 2** | 45M | 2 bits | ~14 MB | ~28 MB | Sí | Binario único, grammar-constrained, confidence scoring |
+| FunctionGemma 270M | 270M | f16 | ~540 MB | ~1.0-1.5 GB | Sí | De Google, formato gemma estándar |
+| Apple FM | ~3B (estimado, no público) | mixto (4-8 bits) | ~1.5 GB | ~3-5 GB | Limitado | Privado, solo Apple Intelligence |
+| LFM2.5 230M | 230M | f16 por defecto | ~460 MB | ~700 MB-1 GB | Parcial | De Liquid AI, multilingüe |
+| Phi-4 mini | ~3.8B | mixto | ~2 GB | ~6-8 GB | Sí | De Microsoft, razonamiento general |
+
+Lo que la tabla muestra: Needle es 5-70x más pequeño que la competencia a tool calling equivalente. La pregunta razonable es cuánto pierdes en calidad por esa reducción.
+
+La respuesta honesta: en benchmarks de tool calling estandarizados (Berkeley Function Calling Leaderboard y similares), Needle está al nivel de FunctionGemma en métricas principales, con dos diferencias cualitativas:
+
+1. Needle es **mejor** en consistencia: la varianza entre ejecuciones es menor, porque la gramática fuerza outputs válidos y la cuantización a 2 bits afecta menos al reasoning que al lenguaje libre.
+2. Needle es **peor** en tareas que requieren razonamiento largo o chaining multi-paso. El contexto deslizante de 256 tokens limita la profundidad de las cadenas.
+
+La implicación práctica: Needle es ideal como **primera capa** en una arquitectura agent. El agente principal (Claude, GPT-4, lo que sea) hace el planning y el reasoning. Needle hace el 90% de las ejecuciones mecánicas que no necesitan contexto largo: clasificar, extraer, transformar, decidir entre A y B.
+
+## Limitaciones reales y modos de fallo
+
+Honestidad. Un modelo de 45M parámetros tiene límites que un LLM grande no tiene. Identificarlos antes evita frustración en producción.
+
+**Limitación 1: contexto corto.** La ventana es 256 tokens. Cualquier tarea que requiera mirar más allá de un párrafo se queda fuera. Needle está diseñado para turnos cortos, no para conversaciones largas.
+
+**Limitación 2: razonamiento multi-paso limitado.** El modelo puede encadenar dos o tres llamadas si las dependencias están claras (`search_for_contact` → `send_message` con el `contact_id`). Más allá de eso, pierde el hilo.
+
+**Limitación 3: zero-shot fuera de catálogo.** Si declaras cinco herramientas y el usuario pide algo que ninguna resuelve, Needle devuelve `[]`. No intenta improvisar. Es lo correcto para producción (no ejecuta acciones no autorizadas), pero requiere que el catálogo de herramientas cubra los casos de uso reales.
+
+**Limitación 4: languages.** El modelo está entrenado mayoritariamente en inglés. En español funciona, pero con menos precisión en tools con nombres y descripciones complicadas. El README sugiere que el multilingual coverage se expandirá en versiones futuras. Para mercados con multilingüismo fuerte (catalán + español, euskera + español) hay que hacer fine-tune por mercado y testear la calidad por idioma antes de desplegar.
+
+**Limitación 5: latencia real en hardware con thermal throttling y batería baja.** Los benchmarks del README (`4300 tokens/s` en prefill, `850 tokens/s` en decode) están medidos en hardware de referencia. En un SoC de dispositivo wearable bajo carga térmica o con batería al 20%, esa cifra puede caer 5-10x. Si tu producto promete una latencia fija ("alerta en menos de 200ms"), el cálculo debe hacerse en el peor caso, no en el banco de pruebas. Sin perfilado térmico, no puedes calcular autonomía.
+
+**Limitación 6: actualizaciones OTA del modelo.** Un `.cact` de 14 MB en una flota de 100k dispositivos no se actualiza como una app. Necesitas decidir: ¿delta-updates para reducir el payload? ¿particionado A/B para no romper dispositivos a mitad de update? ¿cómo verificas que la nueva versión se cargó correctamente antes de borrar la antigua? No hay respuestas estándar — cada equipo de firmware lo resuelve a su manera, y Needle no trae tooling para esto.
+
+**Limitación 7: certificación regulatoria.** Si tu producto toma decisiones críticas para la salud del usuario (alertas médicas, recordatorios de medicación, detección de caídas), puede reclasificarse como producto sanitario: Clase IIa en la UE (MDR), 510(k) en FDA. Eso obliga a documentar el comportamiento del modelo entre versiones, validar que las actualizaciones no introducen regresiones, y mantener un registro de cambios firmados por el responsable regulatorio. Needle cambia de versión cada pocas semanas; tu proceso regulatorio probablemente necesita cadencia trimestral. La desalineación es un riesgo de cumplimiento.
+
+**Mitigación: arquitectura híbrida.** El patrón que mejor funciona en producción:
+
+```
+[input] → [tiny FM para clasificación/intent]
+ ↓ (confidence > threshold?)
+[ejecutar herramienta directamente]
+ ↓ (confidence < threshold?)
+[escalar a LLM grande para planning + ejecución]
+```
+
+Tiny FM hace el 80% del trabajo en local, gratis y rápido. LLM grande solo se invoca para el 20% ambiguo. Resultado: coste por interacción 5-10x menor, latencia media menor, y la garantía de que las acciones críticas se deciden en local.
+
+### Caso de uso aterrizado: asistente en smart speaker para personas mayores
+
+Para que el caso no quede abstracto, una concreción. Imagina un smart speaker de gama media (1 GB de RAM, Cortex-A53, batería a red eléctrica sin restricción de energía) que asiste a personas mayores en casa:
+
+```python
+from typing import Literal, Annotated
+import needle
+
+@needle.tool
+def trigger_emergency_alert(
+ type: Literal["fall", "no_movement", "low_heart_rate", "gas_smell"],
+ confidence_threshold: Annotated[float, needle.Field(ge=0.0, le=1.0)] = 0.85,
+):
+ """Lanza una alerta a servicios de emergencia o familiares.
+
+ Args:
+ type: tipo de incidente detectado
+ confidence_threshold: confianza mínima para actuar sin repreguntar
+ """
+ return {"alert_type": type, "escalated": True}
+
+@needle.tool
+def schedule_medication_reminder(
+ medication: str,
+ time_iso: str,
+):
+ """Programa un recordatorio de medicación.
+
+ Args:
+ medication: nombre del medicamento
+ time_iso: hora en formato ISO 8601
+ """
+ return {"scheduled": True, "medication": medication, "at": time_iso}
+
+@needle.tool
+def call_emergency_contact(name: Annotated[str, needle.Field(pattern=r"^[A-Za-záéíóúñ ]{2,40}$")]):
+ """Llama al contacto de emergencia configurado."""
+ return {"calling": name, "status": "ringing"}
+
+agent = needle.Needle(tools=[trigger_emergency_alert, schedule_medication_reminder, call_emergency_contact])
+```
+
+Las métricas de aceptación razonables para este caso: >95% de las alertas de caída detectadas correctamente, <2% de falsos positivos que generen ansiedad al usuario. El confidence score permite fijar el threshold: si la pulsera externa (que detecta la caída) reporta "caída probable" con 0.92 y el modelo confirma "entorno coherente con caída" con 0.88, se ejecuta la alerta. Si el modelo devuelve `[]` o confidence <0.7, el sistema repregunta al usuario por voz antes de escalar.
+
+**Limitación hardware explícita:** este caso funciona en smart speaker, gateway de hogar o tablet fija. Una pulsera compacta típica (Nordic nRF5340, Apollo4) tiene 100-1000x menos RAM de la que Needle necesita. Si tu producto es wearable barato, este artículo no te sirve para producción hoy; te sirve para decidir qué hardware diseñar.
+
+## Implicaciones para el stack de agents
+
+El cambio relevante no es "existe un modelo pequeño". Es lo que ese modelo pequeño habilita en el diseño de productos.
+
+**Edge-first como decisión de producto, no técnica.** Cuando el modelo cabe en el dispositivo, ya no estás decidiendo entre "servidor en la nube" y "servidor on-premise". Estás decidiendo si la lógica corre **en el dispositivo del usuario o en tu infraestructura**. Eso cambia el cálculo de márgenes, el modelo de privacidad, la regulación aplicable, y la experiencia de uso cuando no hay red.
+
+**El modelo no se descarga. Ya viene pre-instalado.** Es plausible que la distribución de modelos TFMs evolucione hacia patrones ya conocidos: fuentes tipográficas, codecs de audio, bibliotecas estándar del sistema. El modelo vendría embebido en el sistema operativo, en el firmware, o en el bundle de la app. El usuario no "instala un LLM"; usa el dispositivo. Si esa dinámica se consolida, los marketplaces de modelos abiertos quedan como una capa de nicho (developers, customización, fine-tuning), no como el canal principal de distribución — la mayor parte del uso saldría por defecto del dispositivo del usuario.
+
+**El agent runtime se fragmenta por hardware.** Los frameworks de agentes ya no diseñan solo para servidores potentes. Diseñan para gradiente: el mismo agente decide en tiempo real si una subtarea la hace en local con un TFM, en edge con un modelo mediano, o en la nube con un modelo grande. Es la evolución natural de lo que se llamó "cascade routing" pero a nivel de dispositivo.
+
+**Las herramientas cambian de granularidad.** Un LLM grande puede razonar sobre herramientas abstractas ("busca información sobre el clima"). Un TFM necesita herramientas concretas con contratos precisos. El diseño de productos agent-first en edge va a empujar hacia APIs más estrechas, más tipadas, más parecidas a function calls que a endpoints REST ambiguos.
+
+## Cierre
+
+La pregunta ya no es "qué modelo grande uso". Es "qué decisiones dejo en el dispositivo del usuario". Y la respuesta, en 2026, depende de qué tan estrechas sean tus herramientas y qué tan predecible sea tu latencia.
+
+cactus-compute/needle no es la única apuesta en el espacio. Habrá competidores —Apple FM es el más obvio cuando se abra más allá del ecosistema Apple, y los modelos chinos (Qwen, GLM) sacarán versiones sub-100M tarde o temprano. Pero la dirección está clara: el modelo pequeño, especializado, ejecutable en local, ya es un componente realista de producto.
+
+Si tu próximo producto tiene un componente agent, y ese componente va a correr en un smart speaker, gateway o teléfono: **diseña desde el principio para que parte de la lógica viva en el dispositivo**. El modelo existe, el tooling está maduro, los cuellos de botella que quedan son de hardware y regulación, no de IA.
+
+---
+
+## Apéndice — Recursos
+
+| Recurso | URL | Notas |
+|---|---|---|
+| Repositorio | https://github.com/cactus-compute/needle | MIT, 3.963★, 268 líneas de README |
+| Pesos | https://huggingface.co/Cactus-Compute/needle2 | Auto-descarga en el primer `import needle` |
+| Paper | https://arxiv.org/abs/2607.18363 | Simple Attention Network, 2026 |
+| Cita BibTeX | (en README, sección Citation) | Cactus Compute, Inc. 2026 |
+| Playground | `needle playground` | Servidor local en `127.0.0.1:7860` |
+
+Estrellas verificadas vía GitHub REST API el 2026-08-12. Datos de arquitectura del paper arXiv 2607.18363.
diff --git a/src/data/site.ts b/src/data/site.ts
index 42d7aef..67db501 100644
--- a/src/data/site.ts
+++ b/src/data/site.ts
@@ -10,6 +10,10 @@ export const NAV = [
{ href: '/', label: 'Inicio' },
{ href: '/rutas', label: 'Guías' },
{ href: '/ensayos', label: 'Artículos' },
+ // El boletín es otro despliegue bajo el mismo dominio (/tecnoboletin),
+ // no una ruta de este sitio Astro: por eso es un enlace normal y no
+ // aparece en el sitemap de aquí.
+ { href: '/tecnoboletin/', label: 'Boletín' },
{ href: '/talleres', label: 'Talleres' },
];
diff --git a/src/pages/index.astro b/src/pages/index.astro
index 40f7020..8e2913d 100644
--- a/src/pages/index.astro
+++ b/src/pages/index.astro
@@ -138,6 +138,27 @@ const proof = [
+
+
+
cada día
+
Tecnoboletín
+
+ El radar diario de repos, herramientas y papers de IA agéntica: cada hallazgo
+ con su ficha, por qué importa y qué hacer con él. Todo lo publicado se cruza
+ además en un grafo de conocimiento navegable.
+