Reimplements the pr_agent review pipeline (previously Python package) in
Bun/TypeScript, verified end-to-end against a real GitHub App + 9router:
- src/diff.ts: git patch processing (extend_patch, line-numbered hunks,
token-budget generate_full_patch, generated/invalid file filters)
- src/review.ts: orchestrator (fetch PR/diff -> render prompts -> LLM ->
YAML parse -> markdown render -> publish), fallback models
- src/github.ts: GitHub App auth via @octokit/auth-app (JWT -> installation
token, custom authStrategy for Bun), PR/diff/languages/comments REST
- src/prompts.ts: verbatim pr_reviewer_prompts.toml port (nunjucks templates)
- src/yaml.ts: resilient load_yaml with 9 fallback fixers
- src/markdown.ts: convert_to_markdown_v2 port ('PR Reviewer Guide' comment)
- src/llm.ts: 9router OpenAI-compatible chat completions (non-stream,
temp omitted for claude-opus-5), timeout 600s
- src/token.ts: js-tiktoken o200k counting
- src/index.ts: webhook server (HMAC verify, segment analytics, /health,
/api/metrics) + cli.ts one-shot review
- e2e.ts: real-world harness (bun e2e --repo o/r --pr N [--publish])
Verified: bun test 13/13, tsc clean, E2E published a real review comment
(## PR Reviewer Guide) on asepharyana/nextjs-template#19 via GitHub App
MythEclipseBotReview + claude-opus-5 through 9router.
Python run_server.py + pr_agent remain for the live queue worker; server/
is the replacement path.
50 lines
1.5 KiB
TypeScript
50 lines
1.5 KiB
TypeScript
// Token counting — port of pr_agent.algo.token_handler using js-tiktoken
|
|
// (o200k_base, same as Python tiktoken). The Python version tried an accurate
|
|
// Anthropic count_tokens API call when anthropic.key was set; with 9router that
|
|
// call is not an Anthropic API, so we always use the local tokenizer estimate
|
|
// (which is what the Python server effectively falls back to on failure).
|
|
|
|
import { getEncoding, type Tiktoken } from "js-tiktoken";
|
|
|
|
let _encoder: Tiktoken | null = null;
|
|
|
|
export function getTokenEncoder(): Tiktoken {
|
|
if (!_encoder) {
|
|
try {
|
|
_encoder = getEncoding("o200k_base");
|
|
} catch {
|
|
// cl100k_base as a safe fallback if o200k unavailable on old bundles
|
|
_encoder = getEncoding("cl100k_base");
|
|
}
|
|
}
|
|
return _encoder;
|
|
}
|
|
|
|
export function countTokens(text: string): number {
|
|
if (!text) return 0;
|
|
try {
|
|
return getTokenEncoder().encode(text, "all").length;
|
|
} catch {
|
|
// approximate: ~4 chars/token
|
|
return Math.ceil(text.length / 4);
|
|
}
|
|
}
|
|
|
|
/**
|
|
* Approximation of pr_agent's TokenHandler.prompt_tokens: renders the system
|
|
* and user prompt strings with the given vars (nunjucks), then counts tokens.
|
|
*/
|
|
export function countPromptTokens(
|
|
systemTemplate: string,
|
|
userTemplate: string,
|
|
vars: Record<string, unknown>,
|
|
render: (tmpl: string, data: Record<string, unknown>) => string,
|
|
): number {
|
|
try {
|
|
const system = render(systemTemplate, vars);
|
|
const user = render(userTemplate, vars);
|
|
return countTokens(system) + countTokens(user);
|
|
} catch {
|
|
return 0;
|
|
}
|
|
} |