feat(server): Bun/TypeScript re-implementation of PR review engine
Reimplements the pr_agent review pipeline (previously Python package) in
Bun/TypeScript, verified end-to-end against a real GitHub App + 9router:
- src/diff.ts: git patch processing (extend_patch, line-numbered hunks,
token-budget generate_full_patch, generated/invalid file filters)
- src/review.ts: orchestrator (fetch PR/diff -> render prompts -> LLM ->
YAML parse -> markdown render -> publish), fallback models
- src/github.ts: GitHub App auth via @octokit/auth-app (JWT -> installation
token, custom authStrategy for Bun), PR/diff/languages/comments REST
- src/prompts.ts: verbatim pr_reviewer_prompts.toml port (nunjucks templates)
- src/yaml.ts: resilient load_yaml with 9 fallback fixers
- src/markdown.ts: convert_to_markdown_v2 port ('PR Reviewer Guide' comment)
- src/llm.ts: 9router OpenAI-compatible chat completions (non-stream,
temp omitted for claude-opus-5), timeout 600s
- src/token.ts: js-tiktoken o200k counting
- src/index.ts: webhook server (HMAC verify, segment analytics, /health,
/api/metrics) + cli.ts one-shot review
- e2e.ts: real-world harness (bun e2e --repo o/r --pr N [--publish])
Verified: bun test 13/13, tsc clean, E2E published a real review comment
(## PR Reviewer Guide) on asepharyana/nextjs-template#19 via GitHub App
MythEclipseBotReview + claude-opus-5 through 9router.
Python run_server.py + pr_agent remain for the live queue worker; server/
is the replacement path.
This commit is contained in:
@@ -0,0 +1,50 @@
|
||||
// Token counting — port of pr_agent.algo.token_handler using js-tiktoken
|
||||
// (o200k_base, same as Python tiktoken). The Python version tried an accurate
|
||||
// Anthropic count_tokens API call when anthropic.key was set; with 9router that
|
||||
// call is not an Anthropic API, so we always use the local tokenizer estimate
|
||||
// (which is what the Python server effectively falls back to on failure).
|
||||
|
||||
import { getEncoding, type Tiktoken } from "js-tiktoken";
|
||||
|
||||
let _encoder: Tiktoken | null = null;
|
||||
|
||||
export function getTokenEncoder(): Tiktoken {
|
||||
if (!_encoder) {
|
||||
try {
|
||||
_encoder = getEncoding("o200k_base");
|
||||
} catch {
|
||||
// cl100k_base as a safe fallback if o200k unavailable on old bundles
|
||||
_encoder = getEncoding("cl100k_base");
|
||||
}
|
||||
}
|
||||
return _encoder;
|
||||
}
|
||||
|
||||
export function countTokens(text: string): number {
|
||||
if (!text) return 0;
|
||||
try {
|
||||
return getTokenEncoder().encode(text, "all").length;
|
||||
} catch {
|
||||
// approximate: ~4 chars/token
|
||||
return Math.ceil(text.length / 4);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Approximation of pr_agent's TokenHandler.prompt_tokens: renders the system
|
||||
* and user prompt strings with the given vars (nunjucks), then counts tokens.
|
||||
*/
|
||||
export function countPromptTokens(
|
||||
systemTemplate: string,
|
||||
userTemplate: string,
|
||||
vars: Record<string, unknown>,
|
||||
render: (tmpl: string, data: Record<string, unknown>) => string,
|
||||
): number {
|
||||
try {
|
||||
const system = render(systemTemplate, vars);
|
||||
const user = render(userTemplate, vars);
|
||||
return countTokens(system) + countTokens(user);
|
||||
} catch {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user