feat(freebuff): update model catalog + fable offer claims

This commit is contained in:
MUH. IQRAM BAHRING
2026-09-05 23:18:42 +08:00
parent cdbdcd3448
commit 9cd61d585c
4 changed files with 225 additions and 14 deletions
+112 -2
View File
@@ -40,6 +40,18 @@ const SESSION_DEFAULT_TTL_MS = 60 * 60 * 1000; // active sessions live ~1h
// before retrying (mirrors the CLI's FreebuffGateErrorKind statuses).
const SESSION_STALE_CODES = new Set([428, 409, 410]);
// Models the backend runs as a CAPACITY-LIMITED OFFER rather than a standing
// picker row. Claude Fable 5 is not in the client catalog at all: the server
// advertises it per-session-response (`limitedModelOffers`) only while its
// shared wave pool has sessions left, and a request without a live offer is
// refused. A claim must therefore peek at the current offers first instead of
// POSTing blind (mirrors the CLI: the "Claude Fable 5 · N of M left" row only
// renders from that payload). Offer state is per-account and cached briefly —
// the pool can reopen at any time, so a closed offer must NOT set a long
// cooldown.
const OFFER_GATED_MODELS = new Set(["anthropic/claude-fable-5"]);
const OFFER_CACHE_TTL_MS = 45_000;
// The free tier rejects requests whose first system message doesn't open with
// the canonical Freebuff CLI root prompt (server gate
// requestHasFreebuffSystemMarker → 403 free_mode_cli_required). The check is a
@@ -103,12 +115,21 @@ function injectEndTurnTool(body) {
// FREEBUFF_CLI_BASE3_AGENT_ID_BY_MODEL — the CLI harness moved from base2 to
// base3, and the backend can return 404 "No endpoints found" for the old
// base2 roots during the transition).
//
// Withdrawn upstream models (deepseek-v4-pro, minimax-m3, stealth/ox-alpha,
// google/gemini-3.8-flash) are deliberately absent: no new session can be
// admitted on them, so mapping them would only hide a dead pick behind a
// wrong root. z-ai/glm-5.2 stays mapped (referral-earned accounts can still
// run it) even though it is not a standing picker row.
const FREE_ROOT_AGENT_BY_MODEL = {
"deepseek/deepseek-v4-flash": "base3-free-deepseek-flash",
"deepseek/deepseek-v4-pro": "base3-free-deepseek",
"z-ai/glm-5.2": "base3-free-glm",
"z-ai/glm-5.3-flash": "base3-free-glm-5-3-flash",
"mimo/mimo-v2.5": "base3-free-mimo",
"minimax/minimax-m3": "base3-free-minimax-m3",
"openai/gpt-5.6-luna": "base3-free-luna",
"upstage/solar-pro4": "base3-free-solar-pro4",
"meta/muse-spark-1.3-contributor": "base3-free-muse-spark-1-3",
"anthropic/claude-fable-5": "base3-free-fable",
};
// Per-token+model session cache (in-memory; keyed so multi-account setups
@@ -122,11 +143,13 @@ const fbState = (globalThis[FB_STATE_KEY] ??= {
inflight: new Map(), // dedupe concurrent claims for the same key
modelLockCooldowns: new Map(), // `${token}::${model}` -> expiresAt (ms)
poolLimitCooldowns: new Map(), // `${proxyKey}::${model}` -> expiresAt (ms)
offerCache: new Map(), // `${token}` -> { fetchedAt, offers: [] } (limited-offer rows)
});
const sessionCache = fbState.sessionCache;
const inflight = fbState.inflight;
const modelLockCooldowns = fbState.modelLockCooldowns;
const poolLimitCooldowns = fbState.poolLimitCooldowns;
const offerCache = fbState.offerCache;
const MODEL_LOCK_COOLDOWN_MS = 10 * 60 * 1000; // session bound to another model (~1h) — re-check every 10 min
const POOL_LIMITED_COOLDOWN_MS = 5 * 60 * 1000; // IP tier refuses this model — try a different pool/relay
@@ -256,6 +279,10 @@ async function fetchWithNetworkRetry(url, options, proxyOptions, attempts = 3, t
}
async function requestSession(token, model, proxyOptions) {
// Offer-gated models (Fable) refuse claims while their wave pool is closed —
// checked before the POST so a closed offer never burns a claim attempt.
await guardOfferClaim(token, model, proxyOptions);
const response = await fetchWithNetworkRetry(`${sessionOrigin()}${SESSION_PATH}`, {
method: "POST",
headers: {
@@ -313,6 +340,78 @@ async function requestSession(token, model, proxyOptions) {
throw new Error(`Freebuff session rejected (${status || response.status}): ${JSON.stringify(data).slice(0, 200)}`);
}
// Fetch the account's current limited-model offers (GET — never claims).
// Cached per token for OFFER_CACHE_TTL_MS: the wave pool changes on server
// time, not ours, and a claim only needs to know "is it open right now".
async function fetchSessionOffers(token, proxyOptions) {
const now = Date.now();
const cached = offerCache.get(token);
if (cached && now - cached.fetchedAt < OFFER_CACHE_TTL_MS) {
return cached.offers;
}
const response = await fetchWithNetworkRetry(`${sessionOrigin()}${SESSION_PATH}`, {
method: "GET",
headers: {
Authorization: `Bearer ${token}`,
"User-Agent": "codebuff-cli/0.0.138",
Accept: "application/json",
},
}, proxyOptions);
let data = {};
try { data = await response.json(); } catch { data = {}; }
if (response.status === 401) {
const err = new Error("Freebuff session auth failed (401) — re-login in the dashboard");
err.status = 401;
throw err;
}
if (!response.ok) {
const err = new Error(`Freebuff offer check failed: ${response.status} ${JSON.stringify(data).slice(0, 200)}`);
err.status = response.status;
throw err;
}
const offers = Array.isArray(data?.limitedModelOffers)
? data.limitedModelOffers.filter((o) => o && typeof o.model === "string")
: [];
offerCache.set(token, { fetchedAt: now, offers });
return offers;
}
// For an offer-gated model (Fable), refuse the claim BEFORE the POST when the
// backend is not currently advertising it. Returns the matching offer when the
// claim may proceed. Throws a plain Error (no JSON tail) so the executor's
// sessionGateFromError stays null and the cooldown maps are never touched —
// a closed offer is availability, not a lock, and the pool can reopen any time.
async function guardOfferClaim(token, model, proxyOptions) {
if (!OFFER_GATED_MODELS.has(model)) return null;
const offers = await fetchSessionOffers(token, proxyOptions);
const offer = offers.find((o) => o.model === model);
if (!offer || Number(offer.remaining) <= 0) {
const err = new Error(
`Claude Fable 5 is not being offered right now — it is a capacity-limited trial served in waves, and freebuff's shared Fable pool is currently empty. Watch the official freebuff CLI for the "Claude Fable 5 · N of M left" row, or retry later.`,
);
err.status = 409;
err.code = "offer_closed";
throw err;
}
const userLeft = Number(offer.userRemaining);
if (Number.isFinite(userLeft) && userLeft <= 0) {
const resetAt = Date.parse(offer.userResetAt || "");
const err = new Error(
`Your Freebuff account has used its Claude Fable 5 sessions for today (pool: ${offer.remaining} of ${offer.total} left)${Number.isFinite(resetAt) ? ` — next slot ${new Date(resetAt).toLocaleString()}` : ""}.`,
);
err.status = 409;
err.code = "offer_user_capped";
if (Number.isFinite(resetAt)) err.resetsAtMs = resetAt;
throw err;
}
return offer;
}
async function ensureSession(token, model, proxyOptions, force = false) {
const key = sessionCacheKey(token, model);
// Lazy prune: drop stale rows so the cache never accumulates expired entries.
@@ -394,6 +493,7 @@ async function finishRun(token, runId, status, proxyOptions) {
export function resetSessionCache() {
sessionCache.clear();
inflight.clear();
offerCache.clear();
}
// Snapshot sizes of in-memory freebuff state (for the dashboard memory panel).
@@ -403,6 +503,7 @@ export function sessionStateSize() {
inflight: inflight.size,
modelLocks: modelLockCooldowns.size,
poolLimits: poolLimitCooldowns.size,
offerCaches: offerCache.size,
};
}
@@ -417,6 +518,12 @@ export function pruneSessionState(now = Date.now()) {
removed += 1;
}
}
for (const [key, entry] of offerCache) {
if (now - entry.fetchedAt >= OFFER_CACHE_TTL_MS) {
offerCache.delete(key);
removed += 1;
}
}
for (const [key, until] of modelLockCooldowns) {
if (until <= now) {
modelLockCooldowns.delete(key);
@@ -686,6 +793,9 @@ export const __test__ = {
injectFreebuffMarker,
injectEndTurnTool,
fetchWithNetworkRetry,
fetchSessionOffers,
guardOfferClaim,
OFFER_GATED_MODELS,
FREEBUFF_SYSTEM_MARKER,
SESSION_STALE_CODES,
};
+15 -6
View File
@@ -62,15 +62,24 @@ export default {
features: {
usage: true,
},
// Mirrors the CLI's free picker (FREEBUFF_ROOT_AGENT_ID_BY_MODEL).
// mimo/mimo-v2.5-pro is intentionally absent — it is not a free-tier model
// and would bill credits or be rejected under the base2-free agent.
// Mirrors the Freebuff waiting-room picker (upstream FREEBUFF_MODELS) as of
// 2026-09-05, plus the capacity-limited Fable trial. deepseek-v4-pro and
// minimax-m3 were withdrawn upstream (2026-08-26 / 2026-08-20) and ox-alpha
// (2026-08-27) + gemini-3.8-flash (2026-09-03) never stuck — none are
// claimable anymore. z-ai/glm-5.2 is a referral reward (not a free pick),
// luna-es / kimi-k3-eco are god-only rows, and the `-max` variants are
// provisioned per-account — all intentionally omitted. Fable is a
// capacity-limited WAVE trial: sessions only claim while the backend
// advertises it via limitedModelOffers on the session status (the executor
// auto-checks before claiming); the model is otherwise refused.
models: [
{ id: "z-ai/glm-5.3-flash", name: "GLM 5.3 Flash" },
{ id: "deepseek/deepseek-v4-flash", name: "DeepSeek V4 Flash" },
{ id: "deepseek/deepseek-v4-pro", name: "DeepSeek V4 Pro" },
{ id: "mimo/mimo-v2.5", name: "MiMo 2.5" },
{ id: "minimax/minimax-m3", name: "MiniMax M3" },
{ id: "openai/gpt-5.6-luna", name: "GPT-5.6 Luna" },
{ id: "mimo/mimo-v2.5", name: "MiMo 2.5" },
{ id: "upstage/solar-pro4", name: "Solar Pro 4" },
{ id: "meta/muse-spark-1.3-contributor", name: "Muse Spark 1.3" },
{ id: "anthropic/claude-fable-5", name: "Claude Fable 5 (limited offer)" },
],
// Login-flow host — the CLI in freebuff mode logs in via freebuff.com, and
// the server builds loginUrl from the host it was called on, so the link the
+95 -3
View File
@@ -16,6 +16,8 @@ const {
resetSessionCache,
rootAgentIdForModel,
injectFreebuffMarker,
fetchSessionOffers,
guardOfferClaim,
FREEBUFF_SYSTEM_MARKER,
} = __test__;
@@ -273,7 +275,7 @@ describe("freebuff session pre-flight", () => {
fetchMock.mockResolvedValue(
jsonResponse({ status: "active", instanceId: "inst-2", expiresAt: new Date(Date.now() + 3600000).toISOString() }),
);
await ensureSession("tok-1", "minimax/minimax-m3", null);
await ensureSession("tok-1", "z-ai/glm-5.3-flash", null);
expect(fetchMock.mock.calls.length).toBe(2);
});
@@ -305,6 +307,89 @@ describe("freebuff session pre-flight", () => {
});
});
describe("freebuff limited-offer (Claude Fable 5) claims", () => {
const FABLE = "anthropic/claude-fable-5";
const offerRow = (over = {}) => ({
model: FABLE,
remaining: 3,
total: 10,
userRemaining: 1,
userResetAt: new Date(Date.now() + 3600000).toISOString(),
...over,
});
it("GETs limitedModelOffers (never claims) and caches them per token", async () => {
fetchMock.mockResolvedValue(jsonResponse({ status: "none", limitedModelOffers: [offerRow()] }));
const offers = await fetchSessionOffers("tok-1", null);
expect(offers.map((o) => o.model)).toEqual([FABLE]);
const [url, opts] = fetchMock.mock.calls[0];
expect(url).toBe("https://www.codebuff.com/api/v1/freebuff/session");
expect(opts.method).toBe("GET");
expect(opts.headers.Authorization).toBe("Bearer tok-1");
expect(opts.headers.Accept).toBe("application/json");
// Second read within the cache TTL does not refetch.
await fetchSessionOffers("tok-1", null);
expect(fetchMock.mock.calls.length).toBe(1);
});
it("allows the claim while the offer is open", async () => {
fetchMock.mockResolvedValue(jsonResponse({ status: "none", limitedModelOffers: [offerRow()] }));
expect(await guardOfferClaim("tok-1", FABLE, null)).toMatchObject({ model: FABLE, remaining: 3 });
expect(fetchMock.mock.calls.length).toBe(1);
expect(fetchMock.mock.calls[0][1].method).toBe("GET");
});
it("lets non-offer models claim without any offer GET", async () => {
expect(await guardOfferClaim("tok-1", "deepseek/deepseek-v4-flash", null)).toBeNull();
expect(fetchMock.mock.calls.length).toBe(0);
});
it("refuses the claim when the wave pool is closed (no offer row)", async () => {
fetchMock.mockResolvedValue(jsonResponse({ status: "none", limitedModelOffers: [] }));
await expect(guardOfferClaim("tok-1", FABLE, null)).rejects.toThrow(/not being offered right now/i);
});
it("refuses the claim when the account's daily Fable sessions are used up", async () => {
fetchMock.mockResolvedValue(
jsonResponse({ status: "none", limitedModelOffers: [offerRow({ userRemaining: 0 })] }),
);
await expect(guardOfferClaim("tok-1", FABLE, null)).rejects.toThrow(/has used its Claude Fable 5 sessions/i);
});
it("claims a Fable session only after the offer passes: GET offers, then POST claim", async () => {
fetchMock.mockImplementation(async (url, opts = {}) => {
if (url.includes("/freebuff/session") && opts.method === "GET") {
return jsonResponse({ status: "none", limitedModelOffers: [offerRow()] });
}
if (url.includes("/freebuff/session")) {
return jsonResponse({ status: "active", instanceId: "inst-fable", expiresAt: new Date(Date.now() + 3600000).toISOString() });
}
return jsonResponse({ ok: false }, { status: 500, ok: false });
});
const res = await ensureSession("tok-1", FABLE, null);
expect(res).toEqual({ instanceId: "inst-fable", status: "active" });
const methods = fetchMock.mock.calls.map(([, o]) => o.method);
expect(methods).toEqual(["GET", "POST"]);
const [, postOpts] = fetchMock.mock.calls[1];
expect(postOpts.headers["x-freebuff-model"]).toBe(FABLE);
// Cached claim → no further requests.
await ensureSession("tok-1", FABLE, null);
expect(fetchMock.mock.calls.length).toBe(2);
});
it("never POSTs a claim when the Fable wave is closed", async () => {
fetchMock.mockResolvedValue(jsonResponse({ status: "none", limitedModelOffers: [] }));
await expect(ensureSession("tok-1", FABLE, null)).rejects.toThrow(/not being offered right now/i);
const methods = fetchMock.mock.calls.map(([, o]) => o.method);
expect(methods).toEqual(["GET"]);
});
});
describe("freebuff free-tier system marker", () => {
it("prepends the canonical marker when the first message is a system prompt", () => {
const out = injectFreebuffMarker({
@@ -339,10 +424,17 @@ describe("freebuff free-tier system marker", () => {
describe("freebuff run registration", () => {
it("maps freebuff models to their root free agent ids", () => {
expect(rootAgentIdForModel("deepseek/deepseek-v4-flash")).toBe("base3-free-deepseek-flash");
expect(rootAgentIdForModel("deepseek/deepseek-v4-pro")).toBe("base3-free-deepseek");
expect(rootAgentIdForModel("z-ai/glm-5.3-flash")).toBe("base3-free-glm-5-3-flash");
expect(rootAgentIdForModel("z-ai/glm-5.2")).toBe("base3-free-glm");
expect(rootAgentIdForModel("mimo/mimo-v2.5")).toBe("base3-free-mimo");
expect(rootAgentIdForModel("minimax/minimax-m3")).toBe("base3-free-minimax-m3");
expect(rootAgentIdForModel("openai/gpt-5.6-luna")).toBe("base3-free-luna");
expect(rootAgentIdForModel("upstage/solar-pro4")).toBe("base3-free-solar-pro4");
expect(rootAgentIdForModel("meta/muse-spark-1.3-contributor")).toBe("base3-free-muse-spark-1-3");
expect(rootAgentIdForModel("anthropic/claude-fable-5")).toBe("base3-free-fable");
// Withdrawn upstream models are unmapped — they fall back, and the backend
// refuses their sessions anyway.
expect(rootAgentIdForModel("deepseek/deepseek-v4-pro")).toBe("base2-free");
expect(rootAgentIdForModel("minimax/minimax-m3")).toBe("base2-free");
expect(rootAgentIdForModel("some/unknown-model")).toBe("base2-free");
});
+3 -3
View File
@@ -94,7 +94,7 @@ describe("getUsageForProvider(freebuff)", () => {
status: "active",
accessTier: "full",
instanceId: "inst-1",
model: "deepseek/deepseek-v4-pro",
model: "meta/muse-spark-1.3-contributor",
expiresAt: new Date(Date.now() + 3600000).toISOString(),
rateLimit: {
limit: 6,
@@ -112,10 +112,10 @@ describe("getUsageForProvider(freebuff)", () => {
});
expect(usage.plan).toBe("Freebuff");
expect(usage.quotas["deepseek/deepseek-v4-pro"]).toMatchObject({
expect(usage.quotas["meta/muse-spark-1.3-contributor"]).toMatchObject({
used: 2.4,
total: 6,
displayName: "DeepSeek V4 Pro",
displayName: "Muse Spark 1.3",
});
});