feat(freebuff): update model catalog + fable offer claims
This commit is contained in:
@@ -40,6 +40,18 @@ const SESSION_DEFAULT_TTL_MS = 60 * 60 * 1000; // active sessions live ~1h
|
||||
// before retrying (mirrors the CLI's FreebuffGateErrorKind statuses).
|
||||
const SESSION_STALE_CODES = new Set([428, 409, 410]);
|
||||
|
||||
// Models the backend runs as a CAPACITY-LIMITED OFFER rather than a standing
|
||||
// picker row. Claude Fable 5 is not in the client catalog at all: the server
|
||||
// advertises it per-session-response (`limitedModelOffers`) only while its
|
||||
// shared wave pool has sessions left, and a request without a live offer is
|
||||
// refused. A claim must therefore peek at the current offers first instead of
|
||||
// POSTing blind (mirrors the CLI: the "Claude Fable 5 · N of M left" row only
|
||||
// renders from that payload). Offer state is per-account and cached briefly —
|
||||
// the pool can reopen at any time, so a closed offer must NOT set a long
|
||||
// cooldown.
|
||||
const OFFER_GATED_MODELS = new Set(["anthropic/claude-fable-5"]);
|
||||
const OFFER_CACHE_TTL_MS = 45_000;
|
||||
|
||||
// The free tier rejects requests whose first system message doesn't open with
|
||||
// the canonical Freebuff CLI root prompt (server gate
|
||||
// requestHasFreebuffSystemMarker → 403 free_mode_cli_required). The check is a
|
||||
@@ -103,12 +115,21 @@ function injectEndTurnTool(body) {
|
||||
// FREEBUFF_CLI_BASE3_AGENT_ID_BY_MODEL — the CLI harness moved from base2 to
|
||||
// base3, and the backend can return 404 "No endpoints found" for the old
|
||||
// base2 roots during the transition).
|
||||
//
|
||||
// Withdrawn upstream models (deepseek-v4-pro, minimax-m3, stealth/ox-alpha,
|
||||
// google/gemini-3.8-flash) are deliberately absent: no new session can be
|
||||
// admitted on them, so mapping them would only hide a dead pick behind a
|
||||
// wrong root. z-ai/glm-5.2 stays mapped (referral-earned accounts can still
|
||||
// run it) even though it is not a standing picker row.
|
||||
const FREE_ROOT_AGENT_BY_MODEL = {
|
||||
"deepseek/deepseek-v4-flash": "base3-free-deepseek-flash",
|
||||
"deepseek/deepseek-v4-pro": "base3-free-deepseek",
|
||||
"z-ai/glm-5.2": "base3-free-glm",
|
||||
"z-ai/glm-5.3-flash": "base3-free-glm-5-3-flash",
|
||||
"mimo/mimo-v2.5": "base3-free-mimo",
|
||||
"minimax/minimax-m3": "base3-free-minimax-m3",
|
||||
"openai/gpt-5.6-luna": "base3-free-luna",
|
||||
"upstage/solar-pro4": "base3-free-solar-pro4",
|
||||
"meta/muse-spark-1.3-contributor": "base3-free-muse-spark-1-3",
|
||||
"anthropic/claude-fable-5": "base3-free-fable",
|
||||
};
|
||||
|
||||
// Per-token+model session cache (in-memory; keyed so multi-account setups
|
||||
@@ -122,11 +143,13 @@ const fbState = (globalThis[FB_STATE_KEY] ??= {
|
||||
inflight: new Map(), // dedupe concurrent claims for the same key
|
||||
modelLockCooldowns: new Map(), // `${token}::${model}` -> expiresAt (ms)
|
||||
poolLimitCooldowns: new Map(), // `${proxyKey}::${model}` -> expiresAt (ms)
|
||||
offerCache: new Map(), // `${token}` -> { fetchedAt, offers: [] } (limited-offer rows)
|
||||
});
|
||||
const sessionCache = fbState.sessionCache;
|
||||
const inflight = fbState.inflight;
|
||||
const modelLockCooldowns = fbState.modelLockCooldowns;
|
||||
const poolLimitCooldowns = fbState.poolLimitCooldowns;
|
||||
const offerCache = fbState.offerCache;
|
||||
|
||||
const MODEL_LOCK_COOLDOWN_MS = 10 * 60 * 1000; // session bound to another model (~1h) — re-check every 10 min
|
||||
const POOL_LIMITED_COOLDOWN_MS = 5 * 60 * 1000; // IP tier refuses this model — try a different pool/relay
|
||||
@@ -256,6 +279,10 @@ async function fetchWithNetworkRetry(url, options, proxyOptions, attempts = 3, t
|
||||
}
|
||||
|
||||
async function requestSession(token, model, proxyOptions) {
|
||||
// Offer-gated models (Fable) refuse claims while their wave pool is closed —
|
||||
// checked before the POST so a closed offer never burns a claim attempt.
|
||||
await guardOfferClaim(token, model, proxyOptions);
|
||||
|
||||
const response = await fetchWithNetworkRetry(`${sessionOrigin()}${SESSION_PATH}`, {
|
||||
method: "POST",
|
||||
headers: {
|
||||
@@ -313,6 +340,78 @@ async function requestSession(token, model, proxyOptions) {
|
||||
throw new Error(`Freebuff session rejected (${status || response.status}): ${JSON.stringify(data).slice(0, 200)}`);
|
||||
}
|
||||
|
||||
// Fetch the account's current limited-model offers (GET — never claims).
|
||||
// Cached per token for OFFER_CACHE_TTL_MS: the wave pool changes on server
|
||||
// time, not ours, and a claim only needs to know "is it open right now".
|
||||
async function fetchSessionOffers(token, proxyOptions) {
|
||||
const now = Date.now();
|
||||
const cached = offerCache.get(token);
|
||||
if (cached && now - cached.fetchedAt < OFFER_CACHE_TTL_MS) {
|
||||
return cached.offers;
|
||||
}
|
||||
|
||||
const response = await fetchWithNetworkRetry(`${sessionOrigin()}${SESSION_PATH}`, {
|
||||
method: "GET",
|
||||
headers: {
|
||||
Authorization: `Bearer ${token}`,
|
||||
"User-Agent": "codebuff-cli/0.0.138",
|
||||
Accept: "application/json",
|
||||
},
|
||||
}, proxyOptions);
|
||||
|
||||
let data = {};
|
||||
try { data = await response.json(); } catch { data = {}; }
|
||||
|
||||
if (response.status === 401) {
|
||||
const err = new Error("Freebuff session auth failed (401) — re-login in the dashboard");
|
||||
err.status = 401;
|
||||
throw err;
|
||||
}
|
||||
if (!response.ok) {
|
||||
const err = new Error(`Freebuff offer check failed: ${response.status} ${JSON.stringify(data).slice(0, 200)}`);
|
||||
err.status = response.status;
|
||||
throw err;
|
||||
}
|
||||
|
||||
const offers = Array.isArray(data?.limitedModelOffers)
|
||||
? data.limitedModelOffers.filter((o) => o && typeof o.model === "string")
|
||||
: [];
|
||||
offerCache.set(token, { fetchedAt: now, offers });
|
||||
return offers;
|
||||
}
|
||||
|
||||
// For an offer-gated model (Fable), refuse the claim BEFORE the POST when the
|
||||
// backend is not currently advertising it. Returns the matching offer when the
|
||||
// claim may proceed. Throws a plain Error (no JSON tail) so the executor's
|
||||
// sessionGateFromError stays null and the cooldown maps are never touched —
|
||||
// a closed offer is availability, not a lock, and the pool can reopen any time.
|
||||
async function guardOfferClaim(token, model, proxyOptions) {
|
||||
if (!OFFER_GATED_MODELS.has(model)) return null;
|
||||
|
||||
const offers = await fetchSessionOffers(token, proxyOptions);
|
||||
const offer = offers.find((o) => o.model === model);
|
||||
if (!offer || Number(offer.remaining) <= 0) {
|
||||
const err = new Error(
|
||||
`Claude Fable 5 is not being offered right now — it is a capacity-limited trial served in waves, and freebuff's shared Fable pool is currently empty. Watch the official freebuff CLI for the "Claude Fable 5 · N of M left" row, or retry later.`,
|
||||
);
|
||||
err.status = 409;
|
||||
err.code = "offer_closed";
|
||||
throw err;
|
||||
}
|
||||
const userLeft = Number(offer.userRemaining);
|
||||
if (Number.isFinite(userLeft) && userLeft <= 0) {
|
||||
const resetAt = Date.parse(offer.userResetAt || "");
|
||||
const err = new Error(
|
||||
`Your Freebuff account has used its Claude Fable 5 sessions for today (pool: ${offer.remaining} of ${offer.total} left)${Number.isFinite(resetAt) ? ` — next slot ${new Date(resetAt).toLocaleString()}` : ""}.`,
|
||||
);
|
||||
err.status = 409;
|
||||
err.code = "offer_user_capped";
|
||||
if (Number.isFinite(resetAt)) err.resetsAtMs = resetAt;
|
||||
throw err;
|
||||
}
|
||||
return offer;
|
||||
}
|
||||
|
||||
async function ensureSession(token, model, proxyOptions, force = false) {
|
||||
const key = sessionCacheKey(token, model);
|
||||
// Lazy prune: drop stale rows so the cache never accumulates expired entries.
|
||||
@@ -394,6 +493,7 @@ async function finishRun(token, runId, status, proxyOptions) {
|
||||
export function resetSessionCache() {
|
||||
sessionCache.clear();
|
||||
inflight.clear();
|
||||
offerCache.clear();
|
||||
}
|
||||
|
||||
// Snapshot sizes of in-memory freebuff state (for the dashboard memory panel).
|
||||
@@ -403,6 +503,7 @@ export function sessionStateSize() {
|
||||
inflight: inflight.size,
|
||||
modelLocks: modelLockCooldowns.size,
|
||||
poolLimits: poolLimitCooldowns.size,
|
||||
offerCaches: offerCache.size,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -417,6 +518,12 @@ export function pruneSessionState(now = Date.now()) {
|
||||
removed += 1;
|
||||
}
|
||||
}
|
||||
for (const [key, entry] of offerCache) {
|
||||
if (now - entry.fetchedAt >= OFFER_CACHE_TTL_MS) {
|
||||
offerCache.delete(key);
|
||||
removed += 1;
|
||||
}
|
||||
}
|
||||
for (const [key, until] of modelLockCooldowns) {
|
||||
if (until <= now) {
|
||||
modelLockCooldowns.delete(key);
|
||||
@@ -686,6 +793,9 @@ export const __test__ = {
|
||||
injectFreebuffMarker,
|
||||
injectEndTurnTool,
|
||||
fetchWithNetworkRetry,
|
||||
fetchSessionOffers,
|
||||
guardOfferClaim,
|
||||
OFFER_GATED_MODELS,
|
||||
FREEBUFF_SYSTEM_MARKER,
|
||||
SESSION_STALE_CODES,
|
||||
};
|
||||
|
||||
@@ -62,15 +62,24 @@ export default {
|
||||
features: {
|
||||
usage: true,
|
||||
},
|
||||
// Mirrors the CLI's free picker (FREEBUFF_ROOT_AGENT_ID_BY_MODEL).
|
||||
// mimo/mimo-v2.5-pro is intentionally absent — it is not a free-tier model
|
||||
// and would bill credits or be rejected under the base2-free agent.
|
||||
// Mirrors the Freebuff waiting-room picker (upstream FREEBUFF_MODELS) as of
|
||||
// 2026-09-05, plus the capacity-limited Fable trial. deepseek-v4-pro and
|
||||
// minimax-m3 were withdrawn upstream (2026-08-26 / 2026-08-20) and ox-alpha
|
||||
// (2026-08-27) + gemini-3.8-flash (2026-09-03) never stuck — none are
|
||||
// claimable anymore. z-ai/glm-5.2 is a referral reward (not a free pick),
|
||||
// luna-es / kimi-k3-eco are god-only rows, and the `-max` variants are
|
||||
// provisioned per-account — all intentionally omitted. Fable is a
|
||||
// capacity-limited WAVE trial: sessions only claim while the backend
|
||||
// advertises it via limitedModelOffers on the session status (the executor
|
||||
// auto-checks before claiming); the model is otherwise refused.
|
||||
models: [
|
||||
{ id: "z-ai/glm-5.3-flash", name: "GLM 5.3 Flash" },
|
||||
{ id: "deepseek/deepseek-v4-flash", name: "DeepSeek V4 Flash" },
|
||||
{ id: "deepseek/deepseek-v4-pro", name: "DeepSeek V4 Pro" },
|
||||
{ id: "mimo/mimo-v2.5", name: "MiMo 2.5" },
|
||||
{ id: "minimax/minimax-m3", name: "MiniMax M3" },
|
||||
{ id: "openai/gpt-5.6-luna", name: "GPT-5.6 Luna" },
|
||||
{ id: "mimo/mimo-v2.5", name: "MiMo 2.5" },
|
||||
{ id: "upstage/solar-pro4", name: "Solar Pro 4" },
|
||||
{ id: "meta/muse-spark-1.3-contributor", name: "Muse Spark 1.3" },
|
||||
{ id: "anthropic/claude-fable-5", name: "Claude Fable 5 (limited offer)" },
|
||||
],
|
||||
// Login-flow host — the CLI in freebuff mode logs in via freebuff.com, and
|
||||
// the server builds loginUrl from the host it was called on, so the link the
|
||||
|
||||
@@ -16,6 +16,8 @@ const {
|
||||
resetSessionCache,
|
||||
rootAgentIdForModel,
|
||||
injectFreebuffMarker,
|
||||
fetchSessionOffers,
|
||||
guardOfferClaim,
|
||||
FREEBUFF_SYSTEM_MARKER,
|
||||
} = __test__;
|
||||
|
||||
@@ -273,7 +275,7 @@ describe("freebuff session pre-flight", () => {
|
||||
fetchMock.mockResolvedValue(
|
||||
jsonResponse({ status: "active", instanceId: "inst-2", expiresAt: new Date(Date.now() + 3600000).toISOString() }),
|
||||
);
|
||||
await ensureSession("tok-1", "minimax/minimax-m3", null);
|
||||
await ensureSession("tok-1", "z-ai/glm-5.3-flash", null);
|
||||
expect(fetchMock.mock.calls.length).toBe(2);
|
||||
});
|
||||
|
||||
@@ -305,6 +307,89 @@ describe("freebuff session pre-flight", () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe("freebuff limited-offer (Claude Fable 5) claims", () => {
|
||||
const FABLE = "anthropic/claude-fable-5";
|
||||
const offerRow = (over = {}) => ({
|
||||
model: FABLE,
|
||||
remaining: 3,
|
||||
total: 10,
|
||||
userRemaining: 1,
|
||||
userResetAt: new Date(Date.now() + 3600000).toISOString(),
|
||||
...over,
|
||||
});
|
||||
|
||||
it("GETs limitedModelOffers (never claims) and caches them per token", async () => {
|
||||
fetchMock.mockResolvedValue(jsonResponse({ status: "none", limitedModelOffers: [offerRow()] }));
|
||||
const offers = await fetchSessionOffers("tok-1", null);
|
||||
expect(offers.map((o) => o.model)).toEqual([FABLE]);
|
||||
|
||||
const [url, opts] = fetchMock.mock.calls[0];
|
||||
expect(url).toBe("https://www.codebuff.com/api/v1/freebuff/session");
|
||||
expect(opts.method).toBe("GET");
|
||||
expect(opts.headers.Authorization).toBe("Bearer tok-1");
|
||||
expect(opts.headers.Accept).toBe("application/json");
|
||||
|
||||
// Second read within the cache TTL does not refetch.
|
||||
await fetchSessionOffers("tok-1", null);
|
||||
expect(fetchMock.mock.calls.length).toBe(1);
|
||||
});
|
||||
|
||||
it("allows the claim while the offer is open", async () => {
|
||||
fetchMock.mockResolvedValue(jsonResponse({ status: "none", limitedModelOffers: [offerRow()] }));
|
||||
expect(await guardOfferClaim("tok-1", FABLE, null)).toMatchObject({ model: FABLE, remaining: 3 });
|
||||
expect(fetchMock.mock.calls.length).toBe(1);
|
||||
expect(fetchMock.mock.calls[0][1].method).toBe("GET");
|
||||
});
|
||||
|
||||
it("lets non-offer models claim without any offer GET", async () => {
|
||||
expect(await guardOfferClaim("tok-1", "deepseek/deepseek-v4-flash", null)).toBeNull();
|
||||
expect(fetchMock.mock.calls.length).toBe(0);
|
||||
});
|
||||
|
||||
it("refuses the claim when the wave pool is closed (no offer row)", async () => {
|
||||
fetchMock.mockResolvedValue(jsonResponse({ status: "none", limitedModelOffers: [] }));
|
||||
await expect(guardOfferClaim("tok-1", FABLE, null)).rejects.toThrow(/not being offered right now/i);
|
||||
});
|
||||
|
||||
it("refuses the claim when the account's daily Fable sessions are used up", async () => {
|
||||
fetchMock.mockResolvedValue(
|
||||
jsonResponse({ status: "none", limitedModelOffers: [offerRow({ userRemaining: 0 })] }),
|
||||
);
|
||||
await expect(guardOfferClaim("tok-1", FABLE, null)).rejects.toThrow(/has used its Claude Fable 5 sessions/i);
|
||||
});
|
||||
|
||||
it("claims a Fable session only after the offer passes: GET offers, then POST claim", async () => {
|
||||
fetchMock.mockImplementation(async (url, opts = {}) => {
|
||||
if (url.includes("/freebuff/session") && opts.method === "GET") {
|
||||
return jsonResponse({ status: "none", limitedModelOffers: [offerRow()] });
|
||||
}
|
||||
if (url.includes("/freebuff/session")) {
|
||||
return jsonResponse({ status: "active", instanceId: "inst-fable", expiresAt: new Date(Date.now() + 3600000).toISOString() });
|
||||
}
|
||||
return jsonResponse({ ok: false }, { status: 500, ok: false });
|
||||
});
|
||||
|
||||
const res = await ensureSession("tok-1", FABLE, null);
|
||||
expect(res).toEqual({ instanceId: "inst-fable", status: "active" });
|
||||
|
||||
const methods = fetchMock.mock.calls.map(([, o]) => o.method);
|
||||
expect(methods).toEqual(["GET", "POST"]);
|
||||
const [, postOpts] = fetchMock.mock.calls[1];
|
||||
expect(postOpts.headers["x-freebuff-model"]).toBe(FABLE);
|
||||
|
||||
// Cached claim → no further requests.
|
||||
await ensureSession("tok-1", FABLE, null);
|
||||
expect(fetchMock.mock.calls.length).toBe(2);
|
||||
});
|
||||
|
||||
it("never POSTs a claim when the Fable wave is closed", async () => {
|
||||
fetchMock.mockResolvedValue(jsonResponse({ status: "none", limitedModelOffers: [] }));
|
||||
await expect(ensureSession("tok-1", FABLE, null)).rejects.toThrow(/not being offered right now/i);
|
||||
const methods = fetchMock.mock.calls.map(([, o]) => o.method);
|
||||
expect(methods).toEqual(["GET"]);
|
||||
});
|
||||
});
|
||||
|
||||
describe("freebuff free-tier system marker", () => {
|
||||
it("prepends the canonical marker when the first message is a system prompt", () => {
|
||||
const out = injectFreebuffMarker({
|
||||
@@ -339,10 +424,17 @@ describe("freebuff free-tier system marker", () => {
|
||||
describe("freebuff run registration", () => {
|
||||
it("maps freebuff models to their root free agent ids", () => {
|
||||
expect(rootAgentIdForModel("deepseek/deepseek-v4-flash")).toBe("base3-free-deepseek-flash");
|
||||
expect(rootAgentIdForModel("deepseek/deepseek-v4-pro")).toBe("base3-free-deepseek");
|
||||
expect(rootAgentIdForModel("z-ai/glm-5.3-flash")).toBe("base3-free-glm-5-3-flash");
|
||||
expect(rootAgentIdForModel("z-ai/glm-5.2")).toBe("base3-free-glm");
|
||||
expect(rootAgentIdForModel("mimo/mimo-v2.5")).toBe("base3-free-mimo");
|
||||
expect(rootAgentIdForModel("minimax/minimax-m3")).toBe("base3-free-minimax-m3");
|
||||
expect(rootAgentIdForModel("openai/gpt-5.6-luna")).toBe("base3-free-luna");
|
||||
expect(rootAgentIdForModel("upstage/solar-pro4")).toBe("base3-free-solar-pro4");
|
||||
expect(rootAgentIdForModel("meta/muse-spark-1.3-contributor")).toBe("base3-free-muse-spark-1-3");
|
||||
expect(rootAgentIdForModel("anthropic/claude-fable-5")).toBe("base3-free-fable");
|
||||
// Withdrawn upstream models are unmapped — they fall back, and the backend
|
||||
// refuses their sessions anyway.
|
||||
expect(rootAgentIdForModel("deepseek/deepseek-v4-pro")).toBe("base2-free");
|
||||
expect(rootAgentIdForModel("minimax/minimax-m3")).toBe("base2-free");
|
||||
expect(rootAgentIdForModel("some/unknown-model")).toBe("base2-free");
|
||||
});
|
||||
|
||||
|
||||
@@ -94,7 +94,7 @@ describe("getUsageForProvider(freebuff)", () => {
|
||||
status: "active",
|
||||
accessTier: "full",
|
||||
instanceId: "inst-1",
|
||||
model: "deepseek/deepseek-v4-pro",
|
||||
model: "meta/muse-spark-1.3-contributor",
|
||||
expiresAt: new Date(Date.now() + 3600000).toISOString(),
|
||||
rateLimit: {
|
||||
limit: 6,
|
||||
@@ -112,10 +112,10 @@ describe("getUsageForProvider(freebuff)", () => {
|
||||
});
|
||||
|
||||
expect(usage.plan).toBe("Freebuff");
|
||||
expect(usage.quotas["deepseek/deepseek-v4-pro"]).toMatchObject({
|
||||
expect(usage.quotas["meta/muse-spark-1.3-contributor"]).toMatchObject({
|
||||
used: 2.4,
|
||||
total: 6,
|
||||
displayName: "DeepSeek V4 Pro",
|
||||
displayName: "Muse Spark 1.3",
|
||||
});
|
||||
});
|
||||
|
||||
|
||||
Reference in New Issue
Block a user