Route all chat through a local LiteLLM gateway, drop command prefixes

- New litellm service (pinned v1.98.0 — litellm 1.82.7/1.82.8 on PyPI were
  compromised with credential-stealing malware in March 2026; internal-only,
  no Traefik route, no reason to expose an LLM gateway with a master key
  publicly).
- Replaces !claude/!ai command prefixes with automatic routing: every plain
  message in the control room goes through a classifier (router.js) that
  decides chat vs code_task. Chat replies use OpenRouter's own auto-router
  (openrouter/auto) via LiteLLM; code_task requests go through the existing
  runChatTask() flow (Claude Code CLI, unchanged, still using the
  subscription token directly).
- Investigated routing Claude itself through LiteLLM via OAuth token
  forwarding (general_settings.forward_client_headers_to_llm_api) so the
  Pro/Max subscription could be one of the auto-routable options. Confirmed
  non-functional: Anthropic returns a generic rate_limit_error for any
  direct API call using this token type outside the real Claude Code CLI,
  reproduced with plain curl straight to api.anthropic.com. Not included.
- MATRIX_BOT_USER_ID now required and set explicitly (self-message filtering
  can no longer rely on a command-prefix mismatch once there isn't one).
This commit is contained in:
2026-08-23 15:32:55 +00:00
parent 4fc4433833
commit f93bfb25a2
7 changed files with 180 additions and 96 deletions
+29 -55
View File
@@ -1,37 +1,23 @@
import { MatrixClient, SimpleFsStorageProvider } from "matrix-bot-sdk";
import { runChatTask } from "./runner.js";
import { askOpenRouter, DEFAULT_MODEL } from "./openrouter.js";
import { routeMessage, chatReply } from "./router.js";
const HOMESERVER_URL = process.env.MATRIX_HOMESERVER_URL;
const ACCESS_TOKEN = process.env.MATRIX_BOT_TOKEN;
const CONTROL_ROOM_ID = process.env.MATRIX_CONTROL_ROOM_ID;
const GITEA_URL = process.env.GITEA_URL;
// Set explicitly rather than fetched via client.getUserId() — that call hits /whoami,
// which (like /joined_rooms before it) 404s against Continuwuity for reasons unrelated
// to the endpoint itself. This is also the only reliable way to filter the bot's own
// messages now that there's no command prefix to naturally exclude them by.
const BOT_USER_ID = process.env.MATRIX_BOT_USER_ID;
const KNOWN_REPOS = (process.env.KNOWN_REPOS || "")
.split(",")
.map((r) => r.trim())
.filter(Boolean);
const MAX_REPLY_LENGTH = 4000;
// "!claude owner/repo do the thing" — everything after the repo slug is the instruction.
function parseClaudeCommand(text) {
const match = text.match(/^!claude\s+([^\s/]+\/[^\s/]+)\s+(.+)$/s);
if (!match) return null;
const [, repoFullName, instruction] = match;
return { repoFullName, instruction: instruction.trim() };
}
// "!ai <prompt>" uses the default model. "!ai provider/model <prompt>" (first token
// contains a "/") picks a specific OpenRouter model, e.g. "!ai google/gemini-2.0-flash-001
// summarize this repo's README".
function parseAiCommand(text) {
const match = text.match(/^!ai\s+(.+)$/s);
if (!match) return null;
const rest = match[1].trim();
const firstSpace = rest.search(/\s/);
const firstToken = firstSpace === -1 ? rest : rest.slice(0, firstSpace);
if (firstToken.includes("/") && firstSpace !== -1) {
return { model: firstToken, prompt: rest.slice(firstSpace + 1).trim() };
}
return { model: DEFAULT_MODEL, prompt: rest };
}
function truncate(text) {
if (text.length <= MAX_REPLY_LENGTH) return text;
return `${text.slice(0, MAX_REPLY_LENGTH)}\n\n[truncated]`;
@@ -42,15 +28,14 @@ export async function startMatrixBot() {
console.warn("Matrix env vars not set — skipping bot startup");
return;
}
if (!BOT_USER_ID) {
console.warn("MATRIX_BOT_USER_ID not set — bot could reply to its own messages, skipping startup");
return;
}
const storage = new SimpleFsStorageProvider("/workspace/matrix-bot-storage.json");
const client = new MatrixClient(HOMESERVER_URL, ACCESS_TOKEN, storage);
// Not using AutojoinRoomsMixin: it calls /joined_rooms at startup to build its initial
// state, which — like the /whoami call removed earlier — 404s against Continuwuity for
// reasons unrelated to the endpoint itself (curling it directly works fine). This
// simpler handler does the one thing we actually need — auto-join on invite — without
// that startup scan.
client.on("room.invite", async (roomId) => {
try {
await client.joinRoom(roomId);
@@ -60,38 +45,27 @@ export async function startMatrixBot() {
});
client.on("room.message", async (roomId, event) => {
// Invite-only control room enforces who can reach the bot at all; this just scopes
// command handling to that one room. No need to fetch/compare the bot's own user ID
// to filter out its own messages — its replies never match either command pattern
// below, so they're ignored the same as any other non-command message.
if (roomId !== CONTROL_ROOM_ID) return;
if (event.sender === BOT_USER_ID) return;
const body = event.content?.body;
if (!body) return;
const aiCmd = parseAiCommand(body);
if (aiCmd) {
try {
const reply = await askOpenRouter(aiCmd.model, aiCmd.prompt);
await client.sendText(roomId, `[${aiCmd.model}] ${truncate(reply)}`);
} catch (err) {
console.error("openrouter query failed", err);
await client.sendText(roomId, `Failed: ${err.message}`);
}
return;
}
const cmd = parseClaudeCommand(body);
if (!cmd) return;
const [owner, repo] = cmd.repoFullName.split("/");
await client.sendText(roomId, `Working on it: ${cmd.repoFullName}${cmd.instruction}`);
try {
const cloneUrl = `${GITEA_URL}/${owner}/${repo}.git`;
const pr = await runChatTask({ owner, repo, cloneUrl, instruction: cmd.instruction });
await client.sendText(roomId, `Opened PR: ${pr.html_url}`);
const decision = await routeMessage(body, KNOWN_REPOS);
if (decision.type === "code_task") {
const [owner, repo] = decision.repo.split("/");
await client.sendText(roomId, `Working on it: ${decision.repo}${decision.instruction}`);
const cloneUrl = `${GITEA_URL}/${owner}/${repo}.git`;
const pr = await runChatTask({ owner, repo, cloneUrl, instruction: decision.instruction });
await client.sendText(roomId, `Opened PR: ${pr.html_url}`);
return;
}
const reply = await chatReply(body);
await client.sendText(roomId, truncate(reply));
} catch (err) {
console.error("chat task failed", err);
console.error("message handling failed", err);
await client.sendText(roomId, `Failed: ${err.message}`);
}
});