Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
4 changes: 2 additions & 2 deletions package-lock.json

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

2 changes: 1 addition & 1 deletion package.json
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
{
"name": "@agentmuxai/muxcode",
"version": "0.7.0",
"version": "0.8.0",
"description": "First-party agentic loop CLI for AgentMux — local GGUF inference via llama-server, cloud APIs, full MCP tool execution",
"type": "module",
"bin": {
Expand Down
4 changes: 3 additions & 1 deletion src/backends/local.ts
Original file line number Diff line number Diff line change
@@ -1,6 +1,6 @@
import path from 'path';
import { existsSync } from 'fs';
import { getServerUrl } from '../llama-server/manager.js';
import { contextSizeFor, getServerUrl } from '../llama-server/manager.js';
import { muxHome } from '../llama-server/acquire.js';
import { listInstalled } from '../models/list.js';
import type { CompleteOptions, CompletionResponse, IBackend, McpTool, Message, StreamSink } from '../types.js';
Expand All @@ -11,13 +11,15 @@ export class LocalBackend implements IBackend {
private modelPath: string;
private onProgress?: (pct: number, label: string) => void;
readonly model: string;
readonly contextWindow: number;

constructor(
modelName: string,
onProgress?: (pct: number, label: string) => void
) {
this.modelPath = resolveModelPath(modelName);
this.model = path.basename(this.modelPath, '.gguf');
this.contextWindow = contextSizeFor(this.modelPath);
this.onProgress = onProgress;
}

Expand Down
1 change: 1 addition & 0 deletions src/cli.ts
Original file line number Diff line number Diff line change
Expand Up @@ -114,6 +114,7 @@ export function buildCli(): Command {
baseUrl: opts.baseUrl,
onProgress: (pct, label) => emitter.loading(`${label} (${pct}%)`),
});
if (backend.contextWindow) emitter.setContextWindow(backend.model, backend.contextWindow);
} catch (err) {
// Ensure init always precedes the error event
if (!initEmitted) emitter.init(opts.model ?? 'auto', [], [], opts.permissionMode);
Expand Down
17 changes: 17 additions & 0 deletions src/emit/stream-json.ts
Original file line number Diff line number Diff line change
Expand Up @@ -35,6 +35,11 @@ export class StreamJsonEmitter implements StreamSink {
readonly sessionId: string;
private startMs: number;
private format: OutputFormat;
// The model's context window, when the backend knows it, and every model id
// it has been reported under: the backend's own name and whatever
// message_start carried (a server may answer under another id).
private contextWindow?: number;
private windowModels = new Set<string>();

constructor(sessionId?: string, format: OutputFormat = 'stream-json') {
this.sessionId = sessionId ?? `mux-${randomBytes(8).toString('hex')}`;
Expand Down Expand Up @@ -66,7 +71,14 @@ export class StreamJsonEmitter implements StreamSink {

// ── StreamSink: the model's response as it streams ─────────────────────

/** The context window `model` runs with; reported in the result's `modelUsage`. */
setContextWindow(model: string, contextWindow: number) {
this.contextWindow = contextWindow;
this.windowModels.add(model);
}

messageStart(id: string, model: string, usage: Usage) {
if (this.contextWindow != null && model) this.windowModels.add(model);
this.streamEvent({
type: 'message_start',
message: { id, type: 'message', role: 'assistant', model, content: [], stop_reason: null, usage: wireUsage(usage) },
Expand Down Expand Up @@ -167,6 +179,11 @@ export class StreamJsonEmitter implements StreamSink {
total_cost_usd: t.costUsd,
cost_usd: t.costUsd,
usage: wireUsage(t.usage),
// Claude Code's per-model report; AgentMux's context meter takes the
// window from `modelUsage[<message.model>].contextWindow`.
...(this.contextWindow != null
? { modelUsage: Object.fromEntries([...this.windowModels].map(m => [m, { contextWindow: this.contextWindow }])) }
: {}),
});
}

Expand Down
12 changes: 10 additions & 2 deletions src/llama-server/manager.ts
Original file line number Diff line number Diff line change
Expand Up @@ -36,8 +36,7 @@ export async function getServerUrl(

const binPath = await ensureLlamaServer(onProgress);
const port = await findFreePort(8080);
const meta = readModelMeta(modelPath);
const ctxSize = meta?.context_window ?? 4096;
const ctxSize = contextSizeFor(modelPath);

const proc = spawn(binPath, [
'--model', modelPath,
Expand Down Expand Up @@ -139,6 +138,15 @@ function isPortFree(port: number): Promise<boolean> {
});
}

/**
* The context size llama-server runs `modelPath` with: the model's
* `context_window` from its sidecar metadata, else 4096. The effective window
* of a local model, so it is also what `run` reports as the window.
*/
export function contextSizeFor(modelPath: string): number {
return readModelMeta(modelPath)?.context_window ?? 4096;
}

function readModelMeta(modelPath: string): { context_window?: number } | null {
const metaPath = modelPath + '.json';
try {
Expand Down
7 changes: 7 additions & 0 deletions src/types.ts
Original file line number Diff line number Diff line change
Expand Up @@ -83,6 +83,13 @@ export interface CompleteOptions {
export interface IBackend {
/** The model requested (backends report the model that answered on each response). */
readonly model: string;
/**
* The context window the model runs with, when the backend knows it (a local
* model: the size llama-server is started with). Reported in the result's
* `modelUsage`, where AgentMux's context meter reads it. Absent for remote
* backends, whose window muxcode doesn't know.
*/
readonly contextWindow?: number;
complete(messages: Message[], tools: McpTool[], sink: StreamSink, opts?: CompleteOptions): Promise<CompletionResponse>;
}

Expand Down
60 changes: 60 additions & 0 deletions test/context-window.test.mjs
Original file line number Diff line number Diff line change
@@ -0,0 +1,60 @@
// The result frame reports the model's context window the way Claude Code
// does (`modelUsage[<model>].contextWindow`), so AgentMux's context meter
// shows "tokens / window" for a muxcode pane instead of "tokens ctx".
import assert from 'node:assert/strict';
import { test } from 'node:test';
import { StreamJsonEmitter } from '../dist/emit/stream-json.js';

const usage = { inputTokens: 1200, outputTokens: 5, cacheCreationInputTokens: 0, cacheReadInputTokens: 0 };
const totals = { usage, costUsd: 0, numTurns: 1, durationApiMs: 10 };

/** The frames `run` writes to stdout. */
function capture(fn) {
const frames = [];
const write = process.stdout.write;
process.stdout.write = (chunk) => {
for (const line of String(chunk).split('\n')) if (line.trim()) frames.push(JSON.parse(line));
return true;
};
try {
fn();
} finally {
process.stdout.write = write;
}
return frames;
}

test('no window known (a remote backend): no modelUsage', () => {
const frames = capture(() => {
const e = new StreamJsonEmitter('s1');
e.messageStart('m1', 'gpt-4o', usage);
e.done('ok', totals);
});
const result = frames.find(f => f.type === 'result');
assert.equal(result.modelUsage, undefined);
});

test('a local model reports its window under its own id and the id message_start carried', () => {
const frames = capture(() => {
const e = new StreamJsonEmitter('s1');
e.setContextWindow('qwen2.5-coder-7b-q4', 32768);
e.messageStart('m1', 'served-model-alias', usage);
e.done('ok', totals);
});
const result = frames.find(f => f.type === 'result');
assert.deepEqual(result.modelUsage, {
'qwen2.5-coder-7b-q4': { contextWindow: 32768 },
'served-model-alias': { contextWindow: 32768 },
});
});

test('an error result reports it too', () => {
const frames = capture(() => {
const e = new StreamJsonEmitter('s1');
e.setContextWindow('local-model', 4096);
e.error('boom');
});
const result = frames.find(f => f.type === 'result');
assert.equal(result.is_error, true);
assert.deepEqual(result.modelUsage, { 'local-model': { contextWindow: 4096 } });
});
Loading