From 9cd61d585c320a10dcf8f7e3efd98d30f15df828 Mon Sep 17 00:00:00 2001 From: "MUH. IQRAM BAHRING" Date: Sat, 5 Sep 2026 23:18:42 +0800 Subject: [PATCH] feat(freebuff): update model catalog + fable offer claims --- open-sse/executors/freebuff.js | 114 +++++++++++++++++++++++- open-sse/providers/registry/freebuff.js | 21 +++-- tests/unit/freebuff-provider.test.js | 98 +++++++++++++++++++- tests/unit/freebuff-usage.test.js | 6 +- 4 files changed, 225 insertions(+), 14 deletions(-) diff --git a/open-sse/executors/freebuff.js b/open-sse/executors/freebuff.js index 8efe719f..aa7079f3 100644 --- a/open-sse/executors/freebuff.js +++ b/open-sse/executors/freebuff.js @@ -40,6 +40,18 @@ const SESSION_DEFAULT_TTL_MS = 60 * 60 * 1000; // active sessions live ~1h // before retrying (mirrors the CLI's FreebuffGateErrorKind statuses). const SESSION_STALE_CODES = new Set([428, 409, 410]); +// Models the backend runs as a CAPACITY-LIMITED OFFER rather than a standing +// picker row. Claude Fable 5 is not in the client catalog at all: the server +// advertises it per-session-response (`limitedModelOffers`) only while its +// shared wave pool has sessions left, and a request without a live offer is +// refused. A claim must therefore peek at the current offers first instead of +// POSTing blind (mirrors the CLI: the "Claude Fable 5 · N of M left" row only +// renders from that payload). Offer state is per-account and cached briefly — +// the pool can reopen at any time, so a closed offer must NOT set a long +// cooldown. +const OFFER_GATED_MODELS = new Set(["anthropic/claude-fable-5"]); +const OFFER_CACHE_TTL_MS = 45_000; + // The free tier rejects requests whose first system message doesn't open with // the canonical Freebuff CLI root prompt (server gate // requestHasFreebuffSystemMarker → 403 free_mode_cli_required). The check is a @@ -103,12 +115,21 @@ function injectEndTurnTool(body) { // FREEBUFF_CLI_BASE3_AGENT_ID_BY_MODEL — the CLI harness moved from base2 to // base3, and the backend can return 404 "No endpoints found" for the old // base2 roots during the transition). +// +// Withdrawn upstream models (deepseek-v4-pro, minimax-m3, stealth/ox-alpha, +// google/gemini-3.8-flash) are deliberately absent: no new session can be +// admitted on them, so mapping them would only hide a dead pick behind a +// wrong root. z-ai/glm-5.2 stays mapped (referral-earned accounts can still +// run it) even though it is not a standing picker row. const FREE_ROOT_AGENT_BY_MODEL = { "deepseek/deepseek-v4-flash": "base3-free-deepseek-flash", - "deepseek/deepseek-v4-pro": "base3-free-deepseek", + "z-ai/glm-5.2": "base3-free-glm", + "z-ai/glm-5.3-flash": "base3-free-glm-5-3-flash", "mimo/mimo-v2.5": "base3-free-mimo", - "minimax/minimax-m3": "base3-free-minimax-m3", "openai/gpt-5.6-luna": "base3-free-luna", + "upstage/solar-pro4": "base3-free-solar-pro4", + "meta/muse-spark-1.3-contributor": "base3-free-muse-spark-1-3", + "anthropic/claude-fable-5": "base3-free-fable", }; // Per-token+model session cache (in-memory; keyed so multi-account setups @@ -122,11 +143,13 @@ const fbState = (globalThis[FB_STATE_KEY] ??= { inflight: new Map(), // dedupe concurrent claims for the same key modelLockCooldowns: new Map(), // `${token}::${model}` -> expiresAt (ms) poolLimitCooldowns: new Map(), // `${proxyKey}::${model}` -> expiresAt (ms) + offerCache: new Map(), // `${token}` -> { fetchedAt, offers: [] } (limited-offer rows) }); const sessionCache = fbState.sessionCache; const inflight = fbState.inflight; const modelLockCooldowns = fbState.modelLockCooldowns; const poolLimitCooldowns = fbState.poolLimitCooldowns; +const offerCache = fbState.offerCache; const MODEL_LOCK_COOLDOWN_MS = 10 * 60 * 1000; // session bound to another model (~1h) — re-check every 10 min const POOL_LIMITED_COOLDOWN_MS = 5 * 60 * 1000; // IP tier refuses this model — try a different pool/relay @@ -256,6 +279,10 @@ async function fetchWithNetworkRetry(url, options, proxyOptions, attempts = 3, t } async function requestSession(token, model, proxyOptions) { + // Offer-gated models (Fable) refuse claims while their wave pool is closed — + // checked before the POST so a closed offer never burns a claim attempt. + await guardOfferClaim(token, model, proxyOptions); + const response = await fetchWithNetworkRetry(`${sessionOrigin()}${SESSION_PATH}`, { method: "POST", headers: { @@ -313,6 +340,78 @@ async function requestSession(token, model, proxyOptions) { throw new Error(`Freebuff session rejected (${status || response.status}): ${JSON.stringify(data).slice(0, 200)}`); } +// Fetch the account's current limited-model offers (GET — never claims). +// Cached per token for OFFER_CACHE_TTL_MS: the wave pool changes on server +// time, not ours, and a claim only needs to know "is it open right now". +async function fetchSessionOffers(token, proxyOptions) { + const now = Date.now(); + const cached = offerCache.get(token); + if (cached && now - cached.fetchedAt < OFFER_CACHE_TTL_MS) { + return cached.offers; + } + + const response = await fetchWithNetworkRetry(`${sessionOrigin()}${SESSION_PATH}`, { + method: "GET", + headers: { + Authorization: `Bearer ${token}`, + "User-Agent": "codebuff-cli/0.0.138", + Accept: "application/json", + }, + }, proxyOptions); + + let data = {}; + try { data = await response.json(); } catch { data = {}; } + + if (response.status === 401) { + const err = new Error("Freebuff session auth failed (401) — re-login in the dashboard"); + err.status = 401; + throw err; + } + if (!response.ok) { + const err = new Error(`Freebuff offer check failed: ${response.status} ${JSON.stringify(data).slice(0, 200)}`); + err.status = response.status; + throw err; + } + + const offers = Array.isArray(data?.limitedModelOffers) + ? data.limitedModelOffers.filter((o) => o && typeof o.model === "string") + : []; + offerCache.set(token, { fetchedAt: now, offers }); + return offers; +} + +// For an offer-gated model (Fable), refuse the claim BEFORE the POST when the +// backend is not currently advertising it. Returns the matching offer when the +// claim may proceed. Throws a plain Error (no JSON tail) so the executor's +// sessionGateFromError stays null and the cooldown maps are never touched — +// a closed offer is availability, not a lock, and the pool can reopen any time. +async function guardOfferClaim(token, model, proxyOptions) { + if (!OFFER_GATED_MODELS.has(model)) return null; + + const offers = await fetchSessionOffers(token, proxyOptions); + const offer = offers.find((o) => o.model === model); + if (!offer || Number(offer.remaining) <= 0) { + const err = new Error( + `Claude Fable 5 is not being offered right now — it is a capacity-limited trial served in waves, and freebuff's shared Fable pool is currently empty. Watch the official freebuff CLI for the "Claude Fable 5 · N of M left" row, or retry later.`, + ); + err.status = 409; + err.code = "offer_closed"; + throw err; + } + const userLeft = Number(offer.userRemaining); + if (Number.isFinite(userLeft) && userLeft <= 0) { + const resetAt = Date.parse(offer.userResetAt || ""); + const err = new Error( + `Your Freebuff account has used its Claude Fable 5 sessions for today (pool: ${offer.remaining} of ${offer.total} left)${Number.isFinite(resetAt) ? ` — next slot ${new Date(resetAt).toLocaleString()}` : ""}.`, + ); + err.status = 409; + err.code = "offer_user_capped"; + if (Number.isFinite(resetAt)) err.resetsAtMs = resetAt; + throw err; + } + return offer; +} + async function ensureSession(token, model, proxyOptions, force = false) { const key = sessionCacheKey(token, model); // Lazy prune: drop stale rows so the cache never accumulates expired entries. @@ -394,6 +493,7 @@ async function finishRun(token, runId, status, proxyOptions) { export function resetSessionCache() { sessionCache.clear(); inflight.clear(); + offerCache.clear(); } // Snapshot sizes of in-memory freebuff state (for the dashboard memory panel). @@ -403,6 +503,7 @@ export function sessionStateSize() { inflight: inflight.size, modelLocks: modelLockCooldowns.size, poolLimits: poolLimitCooldowns.size, + offerCaches: offerCache.size, }; } @@ -417,6 +518,12 @@ export function pruneSessionState(now = Date.now()) { removed += 1; } } + for (const [key, entry] of offerCache) { + if (now - entry.fetchedAt >= OFFER_CACHE_TTL_MS) { + offerCache.delete(key); + removed += 1; + } + } for (const [key, until] of modelLockCooldowns) { if (until <= now) { modelLockCooldowns.delete(key); @@ -686,6 +793,9 @@ export const __test__ = { injectFreebuffMarker, injectEndTurnTool, fetchWithNetworkRetry, + fetchSessionOffers, + guardOfferClaim, + OFFER_GATED_MODELS, FREEBUFF_SYSTEM_MARKER, SESSION_STALE_CODES, }; diff --git a/open-sse/providers/registry/freebuff.js b/open-sse/providers/registry/freebuff.js index 78e8df3c..c79e3253 100644 --- a/open-sse/providers/registry/freebuff.js +++ b/open-sse/providers/registry/freebuff.js @@ -62,15 +62,24 @@ export default { features: { usage: true, }, - // Mirrors the CLI's free picker (FREEBUFF_ROOT_AGENT_ID_BY_MODEL). - // mimo/mimo-v2.5-pro is intentionally absent — it is not a free-tier model - // and would bill credits or be rejected under the base2-free agent. + // Mirrors the Freebuff waiting-room picker (upstream FREEBUFF_MODELS) as of + // 2026-09-05, plus the capacity-limited Fable trial. deepseek-v4-pro and + // minimax-m3 were withdrawn upstream (2026-08-26 / 2026-08-20) and ox-alpha + // (2026-08-27) + gemini-3.8-flash (2026-09-03) never stuck — none are + // claimable anymore. z-ai/glm-5.2 is a referral reward (not a free pick), + // luna-es / kimi-k3-eco are god-only rows, and the `-max` variants are + // provisioned per-account — all intentionally omitted. Fable is a + // capacity-limited WAVE trial: sessions only claim while the backend + // advertises it via limitedModelOffers on the session status (the executor + // auto-checks before claiming); the model is otherwise refused. models: [ + { id: "z-ai/glm-5.3-flash", name: "GLM 5.3 Flash" }, { id: "deepseek/deepseek-v4-flash", name: "DeepSeek V4 Flash" }, - { id: "deepseek/deepseek-v4-pro", name: "DeepSeek V4 Pro" }, - { id: "mimo/mimo-v2.5", name: "MiMo 2.5" }, - { id: "minimax/minimax-m3", name: "MiniMax M3" }, { id: "openai/gpt-5.6-luna", name: "GPT-5.6 Luna" }, + { id: "mimo/mimo-v2.5", name: "MiMo 2.5" }, + { id: "upstage/solar-pro4", name: "Solar Pro 4" }, + { id: "meta/muse-spark-1.3-contributor", name: "Muse Spark 1.3" }, + { id: "anthropic/claude-fable-5", name: "Claude Fable 5 (limited offer)" }, ], // Login-flow host — the CLI in freebuff mode logs in via freebuff.com, and // the server builds loginUrl from the host it was called on, so the link the diff --git a/tests/unit/freebuff-provider.test.js b/tests/unit/freebuff-provider.test.js index 7db8274f..51cbf2b9 100644 --- a/tests/unit/freebuff-provider.test.js +++ b/tests/unit/freebuff-provider.test.js @@ -16,6 +16,8 @@ const { resetSessionCache, rootAgentIdForModel, injectFreebuffMarker, + fetchSessionOffers, + guardOfferClaim, FREEBUFF_SYSTEM_MARKER, } = __test__; @@ -273,7 +275,7 @@ describe("freebuff session pre-flight", () => { fetchMock.mockResolvedValue( jsonResponse({ status: "active", instanceId: "inst-2", expiresAt: new Date(Date.now() + 3600000).toISOString() }), ); - await ensureSession("tok-1", "minimax/minimax-m3", null); + await ensureSession("tok-1", "z-ai/glm-5.3-flash", null); expect(fetchMock.mock.calls.length).toBe(2); }); @@ -305,6 +307,89 @@ describe("freebuff session pre-flight", () => { }); }); +describe("freebuff limited-offer (Claude Fable 5) claims", () => { + const FABLE = "anthropic/claude-fable-5"; + const offerRow = (over = {}) => ({ + model: FABLE, + remaining: 3, + total: 10, + userRemaining: 1, + userResetAt: new Date(Date.now() + 3600000).toISOString(), + ...over, + }); + + it("GETs limitedModelOffers (never claims) and caches them per token", async () => { + fetchMock.mockResolvedValue(jsonResponse({ status: "none", limitedModelOffers: [offerRow()] })); + const offers = await fetchSessionOffers("tok-1", null); + expect(offers.map((o) => o.model)).toEqual([FABLE]); + + const [url, opts] = fetchMock.mock.calls[0]; + expect(url).toBe("https://www.codebuff.com/api/v1/freebuff/session"); + expect(opts.method).toBe("GET"); + expect(opts.headers.Authorization).toBe("Bearer tok-1"); + expect(opts.headers.Accept).toBe("application/json"); + + // Second read within the cache TTL does not refetch. + await fetchSessionOffers("tok-1", null); + expect(fetchMock.mock.calls.length).toBe(1); + }); + + it("allows the claim while the offer is open", async () => { + fetchMock.mockResolvedValue(jsonResponse({ status: "none", limitedModelOffers: [offerRow()] })); + expect(await guardOfferClaim("tok-1", FABLE, null)).toMatchObject({ model: FABLE, remaining: 3 }); + expect(fetchMock.mock.calls.length).toBe(1); + expect(fetchMock.mock.calls[0][1].method).toBe("GET"); + }); + + it("lets non-offer models claim without any offer GET", async () => { + expect(await guardOfferClaim("tok-1", "deepseek/deepseek-v4-flash", null)).toBeNull(); + expect(fetchMock.mock.calls.length).toBe(0); + }); + + it("refuses the claim when the wave pool is closed (no offer row)", async () => { + fetchMock.mockResolvedValue(jsonResponse({ status: "none", limitedModelOffers: [] })); + await expect(guardOfferClaim("tok-1", FABLE, null)).rejects.toThrow(/not being offered right now/i); + }); + + it("refuses the claim when the account's daily Fable sessions are used up", async () => { + fetchMock.mockResolvedValue( + jsonResponse({ status: "none", limitedModelOffers: [offerRow({ userRemaining: 0 })] }), + ); + await expect(guardOfferClaim("tok-1", FABLE, null)).rejects.toThrow(/has used its Claude Fable 5 sessions/i); + }); + + it("claims a Fable session only after the offer passes: GET offers, then POST claim", async () => { + fetchMock.mockImplementation(async (url, opts = {}) => { + if (url.includes("/freebuff/session") && opts.method === "GET") { + return jsonResponse({ status: "none", limitedModelOffers: [offerRow()] }); + } + if (url.includes("/freebuff/session")) { + return jsonResponse({ status: "active", instanceId: "inst-fable", expiresAt: new Date(Date.now() + 3600000).toISOString() }); + } + return jsonResponse({ ok: false }, { status: 500, ok: false }); + }); + + const res = await ensureSession("tok-1", FABLE, null); + expect(res).toEqual({ instanceId: "inst-fable", status: "active" }); + + const methods = fetchMock.mock.calls.map(([, o]) => o.method); + expect(methods).toEqual(["GET", "POST"]); + const [, postOpts] = fetchMock.mock.calls[1]; + expect(postOpts.headers["x-freebuff-model"]).toBe(FABLE); + + // Cached claim → no further requests. + await ensureSession("tok-1", FABLE, null); + expect(fetchMock.mock.calls.length).toBe(2); + }); + + it("never POSTs a claim when the Fable wave is closed", async () => { + fetchMock.mockResolvedValue(jsonResponse({ status: "none", limitedModelOffers: [] })); + await expect(ensureSession("tok-1", FABLE, null)).rejects.toThrow(/not being offered right now/i); + const methods = fetchMock.mock.calls.map(([, o]) => o.method); + expect(methods).toEqual(["GET"]); + }); +}); + describe("freebuff free-tier system marker", () => { it("prepends the canonical marker when the first message is a system prompt", () => { const out = injectFreebuffMarker({ @@ -339,10 +424,17 @@ describe("freebuff free-tier system marker", () => { describe("freebuff run registration", () => { it("maps freebuff models to their root free agent ids", () => { expect(rootAgentIdForModel("deepseek/deepseek-v4-flash")).toBe("base3-free-deepseek-flash"); - expect(rootAgentIdForModel("deepseek/deepseek-v4-pro")).toBe("base3-free-deepseek"); + expect(rootAgentIdForModel("z-ai/glm-5.3-flash")).toBe("base3-free-glm-5-3-flash"); + expect(rootAgentIdForModel("z-ai/glm-5.2")).toBe("base3-free-glm"); expect(rootAgentIdForModel("mimo/mimo-v2.5")).toBe("base3-free-mimo"); - expect(rootAgentIdForModel("minimax/minimax-m3")).toBe("base3-free-minimax-m3"); expect(rootAgentIdForModel("openai/gpt-5.6-luna")).toBe("base3-free-luna"); + expect(rootAgentIdForModel("upstage/solar-pro4")).toBe("base3-free-solar-pro4"); + expect(rootAgentIdForModel("meta/muse-spark-1.3-contributor")).toBe("base3-free-muse-spark-1-3"); + expect(rootAgentIdForModel("anthropic/claude-fable-5")).toBe("base3-free-fable"); + // Withdrawn upstream models are unmapped — they fall back, and the backend + // refuses their sessions anyway. + expect(rootAgentIdForModel("deepseek/deepseek-v4-pro")).toBe("base2-free"); + expect(rootAgentIdForModel("minimax/minimax-m3")).toBe("base2-free"); expect(rootAgentIdForModel("some/unknown-model")).toBe("base2-free"); }); diff --git a/tests/unit/freebuff-usage.test.js b/tests/unit/freebuff-usage.test.js index 7f53a2e3..71ffb3db 100644 --- a/tests/unit/freebuff-usage.test.js +++ b/tests/unit/freebuff-usage.test.js @@ -94,7 +94,7 @@ describe("getUsageForProvider(freebuff)", () => { status: "active", accessTier: "full", instanceId: "inst-1", - model: "deepseek/deepseek-v4-pro", + model: "meta/muse-spark-1.3-contributor", expiresAt: new Date(Date.now() + 3600000).toISOString(), rateLimit: { limit: 6, @@ -112,10 +112,10 @@ describe("getUsageForProvider(freebuff)", () => { }); expect(usage.plan).toBe("Freebuff"); - expect(usage.quotas["deepseek/deepseek-v4-pro"]).toMatchObject({ + expect(usage.quotas["meta/muse-spark-1.3-contributor"]).toMatchObject({ used: 2.4, total: 6, - displayName: "DeepSeek V4 Pro", + displayName: "Muse Spark 1.3", }); });