feat(server): Bun/TypeScript re-implementation of PR review engine

Reimplements the pr_agent review pipeline (previously Python package) in
Bun/TypeScript, verified end-to-end against a real GitHub App + 9router:

- src/diff.ts: git patch processing (extend_patch, line-numbered hunks,
  token-budget generate_full_patch, generated/invalid file filters)
- src/review.ts: orchestrator (fetch PR/diff -> render prompts -> LLM ->
  YAML parse -> markdown render -> publish), fallback models
- src/github.ts: GitHub App auth via @octokit/auth-app (JWT -> installation
  token, custom authStrategy for Bun), PR/diff/languages/comments REST
- src/prompts.ts: verbatim pr_reviewer_prompts.toml port (nunjucks templates)
- src/yaml.ts: resilient load_yaml with 9 fallback fixers
- src/markdown.ts: convert_to_markdown_v2 port ('PR Reviewer Guide' comment)
- src/llm.ts: 9router OpenAI-compatible chat completions (non-stream,
  temp omitted for claude-opus-5), timeout 600s
- src/token.ts: js-tiktoken o200k counting
- src/index.ts: webhook server (HMAC verify, segment analytics, /health,
  /api/metrics) + cli.ts one-shot review
- e2e.ts: real-world harness (bun e2e --repo o/r --pr N [--publish])

Verified: bun test 13/13, tsc clean, E2E published a real review comment
(## PR Reviewer Guide) on asepharyana/nextjs-template#19 via GitHub App
MythEclipseBotReview + claude-opus-5 through 9router.

Python run_server.py + pr_agent remain for the live queue worker; server/
is the replacement path.
This commit is contained in:
asepharyana
2026-09-21 12:50:10 +07:00
parent 835552c5ca
commit 055501aed1
19 changed files with 3119 additions and 1 deletions
+50
View File
@@ -0,0 +1,50 @@
// Token counting — port of pr_agent.algo.token_handler using js-tiktoken
// (o200k_base, same as Python tiktoken). The Python version tried an accurate
// Anthropic count_tokens API call when anthropic.key was set; with 9router that
// call is not an Anthropic API, so we always use the local tokenizer estimate
// (which is what the Python server effectively falls back to on failure).
import { getEncoding, type Tiktoken } from "js-tiktoken";
let _encoder: Tiktoken | null = null;
export function getTokenEncoder(): Tiktoken {
if (!_encoder) {
try {
_encoder = getEncoding("o200k_base");
} catch {
// cl100k_base as a safe fallback if o200k unavailable on old bundles
_encoder = getEncoding("cl100k_base");
}
}
return _encoder;
}
export function countTokens(text: string): number {
if (!text) return 0;
try {
return getTokenEncoder().encode(text, "all").length;
} catch {
// approximate: ~4 chars/token
return Math.ceil(text.length / 4);
}
}
/**
* Approximation of pr_agent's TokenHandler.prompt_tokens: renders the system
* and user prompt strings with the given vars (nunjucks), then counts tokens.
*/
export function countPromptTokens(
systemTemplate: string,
userTemplate: string,
vars: Record<string, unknown>,
render: (tmpl: string, data: Record<string, unknown>) => string,
): number {
try {
const system = render(systemTemplate, vars);
const user = render(userTemplate, vars);
return countTokens(system) + countTokens(user);
} catch {
return 0;
}
}