Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 6 additions & 6 deletions package.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"name": "@tangle-network/braid",
"version": "0.2.0",
"version": "0.2.1",
"description": "A universal terminal interface for portable agent profiles",
"type": "module",
"bin": {
Expand Down Expand Up @@ -100,11 +100,11 @@
"@dataiku/uv": "0.12.0",
"@earendil-works/pi-tui": "0.84.2",
"@napi-rs/keyring": "1.3.0",
"@tangle-network/agent-eval": "0.149.0",
"@tangle-network/agent-interface": "1.3.0",
"@tangle-network/agent-provider-cli-bridge": "0.9.4",
"@tangle-network/agent-provider-tangle": "0.13.0",
"@tangle-network/agent-runtime": "0.143.0",
"@tangle-network/agent-eval": "0.163.2",
"@tangle-network/agent-interface": "1.4.0",
"@tangle-network/agent-provider-cli-bridge": "0.9.5",
"@tangle-network/agent-provider-tangle": "0.13.1",
"@tangle-network/agent-runtime": "0.153.2",
"@tangle-network/sandbox": "0.31.0",
"better-sqlite3-multiple-ciphers": "13.0.3",
"chalk": "6.0.0",
Expand Down
94 changes: 52 additions & 42 deletions pnpm-lock.yaml

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

14 changes: 7 additions & 7 deletions pnpm-workspace.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -9,13 +9,13 @@ allowBuilds:
node-pty: true
ignoreScripts: false
minimumReleaseAgeExclude:
- '@tangle-network/agent-runtime@0.126.0 || 0.128.0 || 0.131.0 || 0.131.1 || 0.131.2 || 0.131.3 || 0.131.4 || 0.131.5 || 0.131.6 || 0.131.7 || 0.132.0 || 0.132.4 || 0.132.6 || 0.132.9 || 0.132.10 || 0.132.11 || 0.132.12 || 0.133.6 || 0.134.2 || 0.134.4 || 0.134.5 || 0.134.9 || 0.135.0 || 0.135.3 || 0.142.0 || 0.142.1 || 0.142.3 || 0.143.0'
- '@tangle-network/agent-eval@0.143.0 || 0.144.0 || 0.144.1 || 0.144.3 || 0.144.4 || 0.144.6 || 0.144.8 || 0.144.10 || 0.144.11 || 0.144.13 || 0.145.0 || 0.145.1 || 0.145.2 || 0.145.3 || 0.145.9 || 0.145.10 || 0.145.11 || 0.145.14 || 0.145.15 || 0.145.21 || 0.149.0'
- '@tangle-network/agent-interface@0.46.0 || 0.46.1 || 0.47.0 || 0.51.0 || 0.52.0 || 0.53.0 || 0.54.0 || 1.0.0 || 1.1.0 || 1.3.0'
- '@tangle-network/agent-knowledge@7.0.8 || 7.1.2 || 7.1.3 || 7.2.0 || 7.2.2 || 7.2.3 || 7.2.4 || 7.2.6 || 8.0.0 || 8.0.1 || 8.0.5 || 8.0.10'
- '@tangle-network/agent-profile-materialize@0.10.2 || 0.13.1 || 0.14.0 || 0.14.2 || 0.15.1 || 0.15.2'
- '@tangle-network/agent-provider-cli-bridge@0.4.2 || 0.4.3 || 0.5.0 || 0.6.0 || 0.6.4 || 0.7.1 || 0.7.2 || 0.7.3 || 0.7.6 || 0.8.0 || 0.8.1 || 0.9.0 || 0.9.2 || 0.9.3 || 0.9.4'
- '@tangle-network/agent-provider-tangle@0.13.0'
- '@tangle-network/agent-runtime@0.126.0 || 0.128.0 || 0.131.0 || 0.131.1 || 0.131.2 || 0.131.3 || 0.131.4 || 0.131.5 || 0.131.6 || 0.131.7 || 0.132.0 || 0.132.4 || 0.132.6 || 0.132.9 || 0.132.10 || 0.132.11 || 0.132.12 || 0.133.6 || 0.134.2 || 0.134.4 || 0.134.5 || 0.134.9 || 0.135.0 || 0.135.3 || 0.142.0 || 0.142.1 || 0.142.3 || 0.143.0 || 0.153.2'
- '@tangle-network/agent-eval@0.143.0 || 0.144.0 || 0.144.1 || 0.144.3 || 0.144.4 || 0.144.6 || 0.144.8 || 0.144.10 || 0.144.11 || 0.144.13 || 0.145.0 || 0.145.1 || 0.145.2 || 0.145.3 || 0.145.9 || 0.145.10 || 0.145.11 || 0.145.14 || 0.145.15 || 0.145.21 || 0.149.0 || 0.163.2'
- '@tangle-network/agent-interface@0.46.0 || 0.46.1 || 0.47.0 || 0.51.0 || 0.52.0 || 0.53.0 || 0.54.0 || 1.0.0 || 1.1.0 || 1.3.0 || 1.4.0'
- '@tangle-network/agent-knowledge@7.0.8 || 7.1.2 || 7.1.3 || 7.2.0 || 7.2.2 || 7.2.3 || 7.2.4 || 7.2.6 || 8.0.0 || 8.0.1 || 8.0.5 || 8.0.10 || 10.7.0'
- '@tangle-network/agent-profile-materialize@0.10.2 || 0.13.1 || 0.14.0 || 0.14.2 || 0.15.1 || 0.15.2 || 0.17.1'
- '@tangle-network/agent-provider-cli-bridge@0.4.2 || 0.4.3 || 0.5.0 || 0.6.0 || 0.6.4 || 0.7.1 || 0.7.2 || 0.7.3 || 0.7.6 || 0.8.0 || 0.8.1 || 0.9.0 || 0.9.2 || 0.9.3 || 0.9.4 || 0.9.5'
- '@tangle-network/agent-provider-tangle@0.13.0 || 0.13.1'
- '@tangle-network/sandbox@0.30.1 || 0.31.0'
- hono@4.13.0
- '@esbuild/aix-ppc64@0.28.2'
Expand Down
9 changes: 6 additions & 3 deletions src/eval/execution.ts
Original file line number Diff line number Diff line change
Expand Up @@ -205,9 +205,14 @@ export function evalJudgeProfile(config: EvalRouteConfig): AgentProfile {
provider: 'tangle-router',
default: config.model,
reasoningEffort: 'none',
// One budget bounds the visible answer and the whole completion alike: the judge
// returns a short verdict and must never spend the run on reasoning. Runtime lowers
// these onto the Router route as `max_tokens` and `max_completion_tokens`, which this
// profile used to spell out by hand in `extraBody` and a request-level cap.
maxVisibleOutputTokens: EVAL_TOTAL_COMPLETION_TOKENS,
maxTotalOutputTokens: EVAL_TOTAL_COMPLETION_TOKENS,
metadata: {
temperature: 0,
maxTokens: EVAL_TOTAL_COMPLETION_TOKENS,
retry: {
maxAttempts: 3,
initialBackoffMs: 1_000,
Expand All @@ -216,8 +221,6 @@ export function evalJudgeProfile(config: EvalRouteConfig): AgentProfile {
requestTimeoutMs: config.timeoutMs,
},
extraBody: {
// Router treats this field as the hard ceiling over visible and reasoning tokens.
max_completion_tokens: EVAL_TOTAL_COMPLETION_TOKENS,
// GLM enables thinking by default and does not honor reasoning_effort: none.
...(disablesThinking ? { thinking: { type: 'disabled' } } : {}),
},
Expand Down
12 changes: 6 additions & 6 deletions test/eval.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -183,9 +183,10 @@ test('semantic judge execution is one exact AgentProfile owned by Runtime', asyn
provider: 'tangle-router',
default: DEFAULT_EVAL_MODEL,
reasoningEffort: 'none',
maxVisibleOutputTokens: EVAL_TOTAL_COMPLETION_TOKENS,
maxTotalOutputTokens: EVAL_TOTAL_COMPLETION_TOKENS,
metadata: {
temperature: 0,
maxTokens: EVAL_TOTAL_COMPLETION_TOKENS,
retry: {
maxAttempts: 3,
initialBackoffMs: 1_000,
Expand All @@ -194,7 +195,6 @@ test('semantic judge execution is one exact AgentProfile owned by Runtime', asyn
requestTimeoutMs: 120_000,
},
extraBody: {
max_completion_tokens: EVAL_TOTAL_COMPLETION_TOKENS,
thinking: { type: 'disabled' },
},
},
Expand Down Expand Up @@ -261,9 +261,9 @@ test('semantic judge execution is one exact AgentProfile owned by Runtime', asyn
})

const nonGlm = evalJudgeProfile({ ...config, model: 'openai-codex/gpt-5.6-luna' })
assert.deepEqual(nonGlm.model?.metadata?.extraBody, {
max_completion_tokens: EVAL_TOTAL_COMPLETION_TOKENS,
})
assert.deepEqual(nonGlm.model?.metadata?.extraBody, {})
assert.equal(nonGlm.model?.maxVisibleOutputTokens, EVAL_TOTAL_COMPLETION_TOKENS)
assert.equal(nonGlm.model?.maxTotalOutputTokens, EVAL_TOTAL_COMPLETION_TOKENS)
})

test('semantic judge retries transient Router failures with the same operation identity', async (context) => {
Expand Down Expand Up @@ -494,7 +494,7 @@ test('semantic judge refuses an ambiguous smaller request cap before provider sp
messages: [{ role: 'user', content: 'candidate' }],
maxTokens: 320,
}),
/request maxTokens conflicts with AgentProfile model metadata/u,
/request maxTokens 320 conflicts with AgentProfile\.model\.maxVisibleOutputTokens 2048/u,
)
assert.equal(calls, 0)
})
Expand Down
6 changes: 5 additions & 1 deletion test/trace-analysis-configuration.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -587,7 +587,11 @@ test('runtime-owned trace model call preserves canonical messages, limits, usage
content: JSON.stringify({ messages: optimizerRequest().request.messages }),
},
])
assert.equal(receivedBody?.max_tokens, undefined)
// The request's cap is the visible ceiling, and Runtime now lowers it onto the Router
// route as `max_tokens`; it used to accept the cap and never send it. This profile
// declares no total ceiling, so no `max_completion_tokens` is sent.
assert.equal(receivedBody?.max_tokens, 64)
assert.equal(receivedBody?.max_completion_tokens, undefined)
assert.equal(receivedBody?.temperature, 0.2)
assert.deepEqual(receivedBody?.response_format, {
type: 'json_schema',
Expand Down