<?xml version="1.0" encoding="UTF-8"?><rss xmlns:dc="http://purl.org/dc/elements/1.1/" xmlns:content="http://purl.org/rss/1.0/modules/content/" xmlns:atom="http://www.w3.org/2005/Atom" version="2.0"><channel><title><![CDATA[Deepseek-Harness 单卡（7900XTX）运行Qwen3.8-27B 完整文档-2]]></title><description><![CDATA[<h2>4. 开发工具调用 Shim</h2>
<h3>4.1 核心代码 dsh-llm-shim.mjs</h3>
<p dir="auto">单文件、无第三方依赖，存放到 <code>~/llama-bin/dsh-llm-shim.mjs</code>：</p>
<pre><code class="language-javascript">#!/usr/bin/env node
/**
 * dsh-llm-shim —— 放在 DSH 与 llama.cpp 之间的 OpenAI 兼容工具调用 shim
 *
 * 背景：llama.cpp b11223（Vulkan + MTP）自带的 tool-call 解析器（chat_format:
 * peg-native）解析不了 Qwen3.x 的 XML 工具调用：第二个参数起的整段输出会被
 * 吞进第一个参数的值里，并且这种失败会诱发模型无限重复 &lt;tool_call&gt;，直到
 * max_tokens（实测 32768 token / 7.7 分钟、零可用输出）。
 *
 * 但模型本身的输出是正确的，所以这一层这样做：
 *   1. 把 tools 原样保留在请求里（模板仍会把工具定义写进 prompt），
 *      但把 tool_choice 改成 "none" —— 于是 llama.cpp 不做任何工具语法约束，
 *      模型的原始 XML 会完整出现在 message.content 里；
 *   2. 这一层自己把 XML 解析成标准 OpenAI tool_calls；
 *   3. 一旦拿到一个完整的 &lt;tool_call&gt; 且后面又开始重复，立刻掐断上游流，
 *      顺手解决“跑飞 32k token”的问题。
 *
 * 用法：
 *   node dsh-llm-shim.mjs                 # 监听 127.0.0.1:8090
 *                                           # qwen3.8-* → :8080，其余 → :8081（一一对应，不替换）
 *   SHIM_PORT=8091 node dsh-llm-shim.mjs
 *   SHIM_UPSTREAM=http://127.0.0.1:8080 node dsh-llm-shim.mjs   # 固定上游
 * 环境变量：SHIM_UPSTREAM / SHIM_UPSTREAM_36 / SHIM_UPSTREAM_38 / SHIM_PORT / SHIM_HOST /
 *          SHIM_VERBOSE / SHIM_ALLOW_FALLBACK / SHIM_TOOL_REGION_CAP / SHIM_MAX_TOOL_CALLS
 */

import http from "node:http";

const UPSTREAM = (process.env.SHIM_UPSTREAM ?? "").replace(/\/+$/, "");
const PORT = Number(process.env.SHIM_PORT ?? 8090);
const HOST = process.env.SHIM_HOST ?? "127.0.0.1";
const VERBOSE = process.env.SHIM_VERBOSE === "1";

/**
 * llama.sh 的 36/38 两个实例互斥，同一时间只有一个在跑，所以按请求里的模型名选上游。
 * 默认**不静默回退**：选 3.8 却在跑 3.6 时，宁可报错也不要拿另一个模型冒充
 * （两者 ctx 不同，静默替换会让人误判在测哪个模型）。要旧的「谁在跑就用谁」行为
 * 就设 `SHIM_ALLOW_FALLBACK=1`。
 */
const ROUTES = [
  { pattern: /3\.8/i, url: (process.env.SHIM_UPSTREAM_38 ?? "http://127.0.0.1:8080").replace(/\/+$/, ""),
    label: "Qwen3.8-27B", key: "38" },
  { pattern: /./, url: (process.env.SHIM_UPSTREAM_36 ?? "http://127.0.0.1:8081").replace(/\/+$/, ""),
    label: "Qwen3.6-27B", key: "36" },
];
const ALLOW_FALLBACK = process.env.SHIM_ALLOW_FALLBACK === "1";

/** @returns 该请求要试的上游列表（默认严格模式只有一个）。 */
function upstreamsFor(model) {
  if (UPSTREAM) return [UPSTREAM];
  const picked = (ROUTES.find((r) =&gt; r.pattern.test(String(model ?? ""))) ?? ROUTES[0]).url;
  if (!ALLOW_FALLBACK) return [picked];
  return [...new Set([picked, ...ROUTES.map((r) =&gt; r.url)])];
}

/** 某个上游是否健康；用来把「现在到底哪个实例在跑」写进报错里。 */
async function isHealthy(url, signal) {
  if (url === undefined) return false;
  try {
    return (await fetch(`${url}/health`, { signal })).ok;
  } catch {
    return false;
  }
}

const routeLabel = (url) =&gt; {
  const hit = ROUTES.find((r) =&gt; r.url === url);
  return hit === undefined ? url : `:${new URL(url).port}（${hit.label}）`;
};

/** 一段工具调用最多累计多少字符就认定上游在跑飞。 */
const TOOL_REGION_CAP = Number(process.env.SHIM_TOOL_REGION_CAP ?? 4000);
/** 最多接受几个工具调用（正常一轮并行调用不会太多）。 */
const MAX_TOOL_CALLS = Number(process.env.SHIM_MAX_TOOL_CALLS ?? 4);
/** 纯文本流水的保留量，避免把跨界出现的 "&lt;tool_call&gt;" 前半截发出去。 */
const HOLDBACK = "&lt;tool_call&gt;".length - 1;
/** 工具调用区解析失败时，整轮重试几次（模型下一轮通常就写对了）。 */
const MAX_PARSE_ATTEMPTS = Number(process.env.SHIM_PARSE_ATTEMPTS ?? 3);

const log = (...a) =&gt; console.log(new Date().toISOString(), ...a);

// ── XML 工具调用解析 ────────────────────────────────────────────────────────

/** 去掉值尾部残留的闭合标签（模型偶尔漏写某个闭合标签）。 */
function cutValue(v) {
  const i = v.search(/&lt;\/parameter&gt;|&lt;\/function&gt;|&lt;\/tool_call&gt;/);
  if (i &gt;= 0) v = v.slice(0, i);
  return v.trim();
}

/** 解析单个 &lt;tool_call&gt; 的内容。支持 &lt;function=..&gt;&lt;parameter=..&gt; 与 JSON 两种形态。 */
function parseBlock(inner, knownNames) {
  const trimmed = inner.trim();
  // 形态二：&lt;tool_call&gt;{"name":"bash","arguments":{...}}&lt;/tool_call&gt;
  if (trimmed.startsWith("{")) {
    try {
      const o = JSON.parse(trimmed);
      const name = o.name ?? o.tool ?? o.function?.name;
      let args = o.arguments ?? o.parameters ?? o.tool_input ?? {};
      if (typeof args === "string") args = JSON.parse(args);
      if (typeof name === "string" &amp;&amp; args &amp;&amp; typeof args === "object") return { name, args };
    } catch { /* 落到 XML 形态 */ }
  }
  // 形态一：&lt;function=bash&gt;…&lt;/function&gt;
  let name;
  let body;
  const fm = /&lt;function\s*=\s*([^&gt;\s&gt;]+)\s*&gt;/.exec(trimmed);
  if (fm) {
    name = fm[1];
    body = trimmed.slice(fm.index + fm[0].length);
  } else {
    // 容错：模型偶尔漏写 function=，直接写成 &lt;bash&gt; … &lt;/function&gt;
    for (const candidate of knownNames ?? []) {
      const m = new RegExp(`&lt;${candidate.replace(/[.*+?^${}()|[\]\\]/g, "\\$&amp;")}\\s*&gt;`).exec(trimmed);
      if (m) {
        name = candidate;
        body = trimmed.slice(m.index + m[0].length);
        break;
      }
    }
    if (name === undefined) return null;
  }
  const parts = body.split(/&lt;parameter\s*=\s*([^&gt;\s]+)\s*&gt;/);
  const args = {};
  for (let i = 1; i &lt; parts.length; i += 2) {
    const key = parts[i];
    const value = cutValue(parts[i + 1] ?? "");
    args[key] = value;
  }
  // 有名字但一个参数都没解析出来时，多半是格式坏掉了，交给上层决定是否重试。
  return Object.keys(args).length &gt; 0 ? { name, args } : null;
}

/** 从累计文本里挤出所有已闭合的 &lt;tool_call&gt; 块。 */
function scanToolCalls(text, knownNames) {
  const calls = [];
  const re = /&lt;tool_call&gt;([\s\S]*?)&lt;\/tool_call&gt;/g;
  let m;
  while ((m = re.exec(text))) {
    const parsed = parseBlock(m[1], knownNames);
    if (parsed) calls.push(parsed);
  }
  return calls;
}

const sameCall = (a, b) =&gt; a.name === b.name &amp;&amp; JSON.stringify(a.args) === JSON.stringify(b.args);

/** 去掉模型跑飞时重复生成的相同调用。 */
function dedupe(calls) {
  const out = [];
  for (const c of calls) if (!out.some((o) =&gt; sameCall(o, c))) out.push(c);
  return out;
}

// ── 按 JSON Schema 把字符串值还原成正确类型 ────────────────────────────────
// XML 里所有参数都是字符串，但 DSH 的工具参数有 integer/boolean/array 等。

function schemaByName(tools) {
  const map = new Map();
  for (const t of tools ?? []) {
    const fn = t?.function ?? t;
    if (fn?.name) map.set(fn.name, fn.parameters ?? fn.input_schema ?? null);
  }
  return map;
}

function typeOf(schema) {
  if (!schema || typeof schema !== "object") return undefined;
  if (typeof schema.type === "string") return schema.type;
  const cands = [];
  for (const key of ["anyOf", "oneOf", "allOf"]) {
    for (const s of schema[key] ?? []) if (s?.type &amp;&amp; s.type !== "null") cands.push(s.type);
  }
  return cands.length === 1 ? cands[0] : undefined;
}

function coerce(value, schema) {
  switch (typeOf(schema)) {
    case "integer":
    case "number": {
      const n = Number(value);
      return Number.isFinite(n) ? n : value;
    }
    case "boolean": {
      const s = value.toLowerCase();
      if (["true", "yes", "1"].includes(s)) return true;
      if (["false", "no", "0"].includes(s)) return false;
      return value;
    }
    case "array":
    case "object": {
      try { return JSON.parse(value); } catch { return value; }
    }
    default:
      return value;
  }
}

function coerceCall(call, schemas) {
  const params = schemas.get(call.name)?.properties;
  if (!params) return call;
  const args = {};
  for (const [k, v] of Object.entries(call.args)) {
    args[k] = typeof v === "string" ? coerce(v, params[k]) : v;
  }
  return { name: call.name, args };
}

const newCallId = () =&gt; `call_${Math.random().toString(36).slice(2, 10)}${Date.now().toString(36)}`;

// ── HTTP ────────────────────────────────────────────────────────────────────

function readBody(req) {
  return new Promise((resolve, reject) =&gt; {
    const chunks = [];
    req.on("data", (c) =&gt; chunks.push(c));
    req.on("end", () =&gt; resolve(Buffer.concat(chunks)));
    req.on("error", reject);
  });
}

const sse = (res, obj) =&gt; res.write(`data: ${JSON.stringify(obj)}\n\n`);

/**
 * 连上第一个能用的本地上游。一旦开始收流就不再换。
 *
 * 4xx 是「请求本身有问题」，换实例没有意义 —— 直接把上游的状态码和正文原样交给
 * 客户端（否则真实原因会被 "unreachable" 掩盖）。只有连不上或 5xx 才换下一个。
 *
 * @param body - 已经改写成 tool_choice:"none" 的上游请求体。
 * @param model - 客户端请求的模型名，决定先试哪个实例。
 * @param signal - 客户端断开信号。
 * @returns `{ response }` 成功；失败为 `{ status, payload, detail }`。
 */
async function connectUpstream(body, model, signal) {
  const init = {
    method: "POST",
    headers: { "content-type": "application/json", accept: "text/event-stream" },
    body: JSON.stringify(body),
    signal,
  };
  let failure = { status: 502, payload: { error: { message: "shim: no local llama.cpp upstream reachable",
    code: "NO_UPSTREAM" } }, detail: "none tried" };
  for (const base of upstreamsFor(model)) {
    let response;
    try {
      response = await fetch(`${base}/v1/chat/completions`, init);
    } catch (error) {
      if (signal?.aborted) throw error;
      // 严格模式：把这个模型对应的实例没起来、而另一个在跑这件事说清楚，
      // 而不是偷偷拿另一个模型回答。
      const others = ROUTES.filter((r) =&gt; r.url !== base);
      const wanted = ROUTES.find((r) =&gt; r.url === base);
      let hint = "";
      for (const other of others) {
        if (await isHealthy(other.url, signal)) {
          hint = `现在运行的是 ${routeLabel(other.url)}；请在 GUI 的 /model 里选它，` +
            `或用 ~/llama-bin/llama.sh ${wanted?.key ?? "36|38"} 启动 ${routeLabel(base)}。`;
          break;
        }
      }
      if (hint === "") hint = "本地模型没启动？先用 ~/llama-bin/llama.sh 36|38 启动。";
      failure = { status: 502, payload: { error: {
        message: `shim: ${routeLabel(base)}未启动（${base} 连不上）。${hint}`,
        code: "UPSTREAM_UNREACHABLE", detail: error?.message ?? String(error) } },
        detail: `${base} unreachable: ${error?.message ?? error}` };
      log(failure.detail);
      continue;
    }
    if (response.ok &amp;&amp; response.body) {
      if (VERBOSE) log(`→ ${base}`);
      return { response };
    }
    const text = await response.text().catch(() =&gt; "");
    let payload;
    try { payload = JSON.parse(text); } catch { payload = { error: { message: text.slice(0, 500) } }; }
    // 上下文超限是最常见的 4xx，给一句能直接照做的提示。
    const type = payload?.error?.type;
    if (type === "exceed_context_size_error" || /exceed.*context/i.test(payload?.error?.message ?? "")) {
      payload = { error: { message:
        `本地模型上下文不足：请求 ${payload.error.n_prompt_tokens ?? "?"} token，` +
        `而模型 ctx 只有 ${payload.error.n_ctx ?? "?"}。请在 DSH 里 /compact 压缩会话，或另开一个新会话。`,
        code: "CONTEXT_OVERFLOW", upstream: payload.error } };
    }
    failure = { status: response.status, payload, detail: `${base} answered ${response.status}: ${text.slice(0, 200)}` };
    log(failure.detail);
    if (response.status &lt; 500) break;   // 4xx：请求本身的问题，换实例也没用
  }
  log(`upstream failed (${failure.detail})`);
  return failure;
}

/**
 * 处理一次 /v1/chat/completions。
 *
 * @param body - 客户端请求体。
 * @param res - 客户端响应。
 * @param signal - 客户端断开时中止上游。
 */
async function handleCompletions(body, res, signal) {
  // 与 OpenAI 语义一致：只有显式 stream:true 才是流式（DSH 每次都会显式带上）。
  const wantsStream = body.stream === true;
  const schemas = schemaByName(body.tools);
  const knownNames = [...schemas.keys()];
  const hasTools = (body.tools?.length ?? 0) &gt; 0;

  // 关键：工具定义留在 prompt 里，但关掉 llama.cpp 的工具语法约束。
  const upstreamBody = {
    ...body,
    ...(hasTools ? { tool_choice: "none" } : {}),
    stream: true,
    stream_options: { include_usage: true },
  };

  const model = body.model ?? "local";
  const id = `chatcmpl-shim-${Math.random().toString(36).slice(2, 12)}`;
  const created = Math.floor(Date.now() / 1000);

  // SSE 头必须等上游真的接受请求之后再发：否则上游报错时状态码已经锁成 200，
  // 只能把错误正文塞进流里，客户端会看到 "Stream ended without finish_reason"，
  // 真实原因（例如上下文超限）被完全掩盖。
  let streamHeadersSent = false;
  const ensureStreamHeaders = () =&gt; {
    if (!wantsStream || streamHeadersSent) return;
    res.writeHead(200, {
      "content-type": "text/event-stream; charset=utf-8",
      "cache-control": "no-cache",
      connection: "keep-alive",
    });
    streamHeadersSent = true;
  };

  const emitContent = (chunk) =&gt; {
    if (wantsStream) sse(res, { id, object: "chat.completion.chunk", created, model,
      choices: [{ index: 0, delta: { content: chunk }, finish_reason: null }] });
  };
  const emitToolCalls = (calls, prefixText, usage) =&gt; {
    const payload = calls.map((c, index) =&gt; ({
      index, id: newCallId(), type: "function",
      function: { name: c.name, arguments: JSON.stringify(c.args) },
    }));
    if (VERBOSE) log(`→ tool_calls ${calls.map((c) =&gt; `${c.name}(${JSON.stringify(c.args).slice(0, 80)})`).join(", ")}`);
    if (wantsStream) {
      // 工具调用之前的正文也要发出去（它可能还压在保留尾巴里），别丢内容。
      if (prefixText) emitContent(prefixText);
      sse(res, { id, object: "chat.completion.chunk", created, model,
        choices: [{ index: 0, delta: { tool_calls: payload }, finish_reason: null }] });
      sse(res, { id, object: "chat.completion.chunk", created, model,
        choices: [{ index: 0, delta: {}, finish_reason: "tool_calls" }],
        ...(usage ? { usage } : {}) });
      res.write("data: [DONE]\n\n");
      res.end();
    } else {
      res.writeHead(200, { "content-type": "application/json" });
      res.end(JSON.stringify({
        id, object: "chat.completion", created, model,
        choices: [{ index: 0, message: { role: "assistant", content: prefixText || null,
          tool_calls: payload }, finish_reason: "tool_calls" }],
        ...(usage ? { usage } : {}),
      }));
    }
  };
  /** @param tail - 还没转发出去的正文字（流式时 sent 之后的剩余部分）。 */
  const emitPlain = (tail, finish, usage) =&gt; {
    if (wantsStream) {
      if (tail) emitContent(tail);
      sse(res, { id, object: "chat.completion.chunk", created, model,
        choices: [{ index: 0, delta: {}, finish_reason: finish ?? "stop" }],
        ...(usage ? { usage } : {}) });
      res.write("data: [DONE]\n\n");
      res.end();
    } else {
      res.writeHead(200, { "content-type": "application/json" });
      res.end(JSON.stringify({
        id, object: "chat.completion", created, model,
        choices: [{ index: 0, message: { role: "assistant", content: tail },
          finish_reason: finish ?? "stop" }],
        ...(usage ? { usage } : {}),
      }));
    }
  };

  /**
   * 收完一次上游流：边收边把普通正文转发给客户端，遇到 &lt;tool_call&gt; 就转入解析。
   * @returns 本次尝试的结果，供上层决定要不要重试。
   */
  const consume = async (upstream) =&gt; {
    const upstreamAbort = new AbortController();
    signal.addEventListener("abort", () =&gt; upstreamAbort.abort(), { once: true });
    const decoder = new TextDecoder();
    let buffer = "";
    let text = "";
    let sent = 0;
    let toolMode = false;
    let toolText = "";
    let usage = null;
    let upstreamFinish = null;
    let aborted = false;
    try {
      for await (const chunk of upstream.body) {
        buffer += decoder.decode(chunk, { stream: true });
        let nl;
        while ((nl = buffer.indexOf("\n")) &gt;= 0) {
          const line = buffer.slice(0, nl).trim();
          buffer = buffer.slice(nl + 1);
          if (!line.startsWith("data:")) continue;
          const payload = line.slice(5).trim();
          if (payload === "[DONE]") continue;
          let parsed;
          try { parsed = JSON.parse(payload); } catch { continue; }
          if (parsed.usage) usage = parsed.usage;
          const choice = parsed.choices?.[0];
          if (!choice) continue;
          if (choice.finish_reason) upstreamFinish = choice.finish_reason;
          const piece = typeof choice.delta?.content === "string" ? choice.delta.content : "";
          if (!piece) continue;

          if (!toolMode) {
            text += piece;
            const at = text.indexOf("&lt;tool_call&gt;");
            if (at &lt; 0) {
              // 普通正文：留一点尾巴防止 "&lt;tool_call&gt;" 跨 chunk，其余立即转发。
              // 保留只对真正的流式有意义 —— 非流式一次性发送时不能推进 sent，
              // 否则最后 emitPlain 只会吐出被"保留"的那几个字符。
              if (wantsStream) {
                const safe = Math.max(sent, text.length - HOLDBACK);
                if (safe &gt; sent) { emitContent(text.slice(sent, safe)); sent = safe; }
              }
              continue;
            }
            if (at &gt; sent) { emitContent(text.slice(sent, at)); sent = at; }
            toolMode = true;
            toolText = text.slice(at);
            text = text.slice(0, at);
          } else {
            toolText += piece;
          }

          if (toolMode) {
            const found = scanToolCalls(toolText, knownNames);
            if (found.length &gt;= MAX_TOOL_CALLS || toolText.length &gt;= TOOL_REGION_CAP) {
              aborted = true;
              upstreamAbort.abort();
              break;
            }
          }
        }
        if (aborted) break;
      }
    } catch (error) {
      if (!aborted &amp;&amp; error?.name !== "AbortError") log(`upstream stream error: ${error?.message ?? error}`);
    } finally {
      upstreamAbort.abort();
    }
    const calls = toolMode
      ? dedupe(scanToolCalls(toolText, knownNames)).map((c) =&gt; coerceCall(c, schemas))
      : [];
    return { text, sent, toolMode, toolText, calls, usage, upstreamFinish, aborted };
  };

  for (let attempt = 1; attempt &lt;= MAX_PARSE_ATTEMPTS; attempt++) {
    const connected = await connectUpstream(upstreamBody, body.model, signal);
    if (connected.response === undefined) {
      // 上游拒绝了这次请求（或全连不上）：此时还没发过 SSE 头，可以如实回状态码和原因。
      const { status, payload } = connected;
      if (!res.headersSent) {
        res.writeHead(status, { "content-type": "application/json" });
        res.end(JSON.stringify(payload));
      } else if (!res.writableEnded) {
        // 上一轮已经发过流（解析失败重试的情形）：只能把错误写进流里。
        sse(res, payload);
        res.write("data: [DONE]\n\n");
        res.end();
      }
      return;
    }

    ensureStreamHeaders();
    const r = await consume(connected.response);

    if (r.calls.length &gt; 0) {
      if (r.aborted &amp;&amp; VERBOSE) log(`upstream cut early after ${r.toolText.length} chars (runaway guard)`);
      emitToolCalls(r.calls, r.text, r.usage);
      return;
    }

    // 看到了 &lt;tool_call&gt; 却没解析出调用、而且还没往客户端发过正文 → 重试一次，
    // 模型下一轮通常就写对了（实测常见形态：&lt;bash&gt; 漏写 function=）。
    if (r.toolMode &amp;&amp; r.sent === 0 &amp;&amp; attempt &lt; MAX_PARSE_ATTEMPTS &amp;&amp; !signal.aborted) {
      log(`attempt ${attempt}: unparsable tool region (${r.toolText.length} chars), retrying`);
      continue;
    }

    if (r.toolMode) {
      log(`tool region had no parsable call; falling back to text (${r.toolText.length} chars): ` +
        r.toolText.replace(/\s+/g, " ").slice(0, 200));
      emitPlain(r.text.slice(r.sent) + r.toolText, r.upstreamFinish === "length" ? "length" : "stop", r.usage);
      return;
    }
    emitPlain(r.text.slice(r.sent), r.upstreamFinish === "length" ? "length" : "stop", r.usage);
    return;
  }
}

const server = http.createServer(async (req, res) =&gt; {
  const url = req.url ?? "/";
  if (req.method === "GET" &amp;&amp; (url.startsWith("/health") || url.startsWith("/v1/models"))) {
    // /health 只有本层活着就返回 ok（上游没起来时 DSH 不该因此认为 shim 挂了）。
    if (url.startsWith("/health")) {
      res.writeHead(200, { "content-type": "application/json" });
      res.end(JSON.stringify({ status: "ok", upstreams: upstreamsFor(""),
        fallback: ALLOW_FALLBACK ? "on" : "off (strict: 选哪个模型就只打哪个实例)" }));
      return;
    }
    for (const base of upstreamsFor("")) {
      const upstream = await fetch(`${base}${url}`).catch(() =&gt; null);
      if (upstream?.ok) {
        const text = await upstream.text();
        res.writeHead(200, { "content-type": "application/json" });
        res.end(text);
        return;
      }
    }
    res.writeHead(502, { "content-type": "application/json" });
    res.end(JSON.stringify({ error: { message: "shim: upstream unreachable" } }));
    return;
  }
  if (req.method !== "POST" || !url.startsWith("/v1/chat/completions")) {
    res.writeHead(404, { "content-type": "application/json" });
    res.end(JSON.stringify({ error: { message: `shim: no route for ${req.method} ${url}` } }));
    return;
  }
  const controller = new AbortController();
  res.on("close", () =&gt; controller.abort());
  try {
    const body = JSON.parse((await readBody(req)).toString("utf8") || "{}");
    await handleCompletions(body, res, controller.signal);
  } catch (error) {
    log(`request failed: ${error?.stack ?? error}`);
    if (!res.headersSent) res.writeHead(500, { "content-type": "application/json" });
    if (!res.writableEnded) res.end(JSON.stringify({ error: { message: `shim: ${error?.message ?? error}` } }));
  }
});

server.listen(PORT, HOST, () =&gt; {
  log(`dsh-llm-shim listening on http://${HOST}:${PORT}  →  upstream ${upstreamsFor("").join(" , ")}`);
  log(`tool-call parsing: client-side (tool_choice forced to "none"; cap ${TOOL_REGION_CAP} chars, max ${MAX_TOOL_CALLS} calls)`);
});

// 便于离线单测解析逻辑（直接运行时这些导出无副作用）。
export { parseBlock, scanToolCalls, dedupe, coerce };
</code></pre>
<h3>4.2 管理脚本 <a href="http://shim.sh" rel="nofollow ugc">shim.sh</a></h3>
<p dir="auto">存放到 <code>~/llama-bin/shim.sh</code>：</p>
<pre><code class="language-bash">#!/bin/bash
# 本地模型工具调用 shim 管理 —— 配合 dsh-llm-shim.mjs
#
#   ./shim.sh start     启动（已在跑则提示）
#   ./shim.sh stop      停止
#   ./shim.sh restart
#   ./shim.sh status    shim 与两个 llama 实例的状态
#
# 为什么需要它：llama.cpp b11223（Vulkan + MTP）自带的 tool-call 解析器解析不了
# Qwen3.x 的 XML 工具调用（第二个参数起的整段被吞进第一个参数），并会诱发模型
# 无限重复 &lt;tool_call&gt; 直到 max_tokens。模型本身输出是对的，所以这一层让
# llama.cpp 不做工具语法约束，由 shim 自己把 XML 解析成标准 OpenAI tool_calls。
# 实测：直连 0/10 成功、每次跑飞 ~30s；走 shim 10/10 成功、每次 ~1s。
set -uo pipefail

SHIM_DIR="${SHIM_DIR:-$HOME/llama-bin}"
SHIM_JS="$SHIM_DIR/dsh-llm-shim.mjs"
PORT="${SHIM_PORT:-8090}"
LOG="${SHIM_LOG:-/tmp/dsh-llm-shim.log}"
PIDFILE="${SHIM_PIDFILE:-$SHIM_DIR/.dsh-llm-shim.pid}"
NODE="${NODE:-$(command -v node || echo "$HOME/.nvm/versions/node/v24.21.0/bin/node")}"

# 只认 argv[0] 是 node、且命令行含脚本绝对路径的进程。
# 绝不裸 pgrep 文件名 —— 否则 `pkill -f dsh-llm-shim.mjs` 会连
# `vim dsh-llm-shim.mjs`、甚至调用者自己的 shell 一起杀掉（踩过一次）。
_running() {
  local pid argv0
  if [ -r "$PIDFILE" ]; then
    pid=$(cat "$PIDFILE" 2&gt;/dev/null)
    if [ -n "$pid" ] &amp;&amp; [ -r "/proc/$pid/cmdline" ] &amp;&amp; \
       tr '\0' ' ' &lt;"/proc/$pid/cmdline" 2&gt;/dev/null | grep -qF -- "$SHIM_JS"; then
      echo "$pid"; return 0
    fi
  fi
  local found=""
  for pid in $(pgrep -f -- "$SHIM_JS" 2&gt;/dev/null); do
    [ "$pid" = "$$" ] &amp;&amp; continue
    [ "$pid" = "${PPID:-0}" ] &amp;&amp; continue
    argv0=$(tr '\0' '\n' &lt;"/proc/$pid/cmdline" 2&gt;/dev/null | head -1)
    case "$argv0" in *node*) found="$found $pid" ;; esac
  done
  [ -n "$found" ] || return 1        # 没找到要明确失败，否则调用方会误判“已在运行”
  echo $found
}
_healthy() { curl -fsS --max-time 2 "http://127.0.0.1:$PORT/health" &gt;/dev/null 2&gt;&amp;1; }

case "${1:-status}" in
  start)
    if _running &gt;/dev/null; then
      echo "shim 已在运行 (pid $(_running | tr '\n' ' '))"; exit 0
    fi
    # 进程看不到、但端口已经有人在服务（例如由别的终端/会话拉起）时别再抢端口
    if _healthy; then
      echo "shim 端口 :$PORT 已有实例在服务，跳过启动"; exit 0
    fi
    [ -f "$SHIM_JS" ] || { echo "缺少 $SHIM_JS" &gt;&amp;2; exit 1; }
    [ -x "$NODE" ] || { echo "找不到 node: $NODE" &gt;&amp;2; exit 1; }
    SHIM_PORT="$PORT" SHIM_VERBOSE=1 setsid nohup "$NODE" "$SHIM_JS" &lt;/dev/null &gt;"$LOG" 2&gt;&amp;1 &amp;
    echo $! &gt;"$PIDFILE"
    disown
    printf '启动中'
    for _ in $(seq 1 20); do _healthy &amp;&amp; break; sleep 0.5; printf '.'; done
    echo
    if _healthy; then
      echo "shim 就绪: http://127.0.0.1:$PORT   log: $LOG"
    else
      echo "启动失败，检查 $LOG" &gt;&amp;2; exit 1
    fi ;;
  stop)
    if _running &gt;/dev/null; then
      _pids=$(_running)
      kill $_pids 2&gt;/dev/null
      for _ in $(seq 1 20); do _running &gt;/dev/null || break; sleep 0.5; done
      kill -9 $_pids 2&gt;/dev/null
      rm -f "$PIDFILE"
      echo "shim 已停止"
    elif _healthy; then
      echo "shim 进程在本终端不可见，但 :$PORT 仍在服务（换个终端 pkill -f dsh-llm-shim.mjs）" &gt;&amp;2
    else
      echo "shim 未在运行"
    fi ;;
  restart) "$0" stop; sleep 1; exec "$0" start ;;
  status)
    if _running &gt;/dev/null; then
      printf 'shim      : 运行中 (pid %s) ' "$(_running | tr '\n' ' ')"
      _healthy &amp;&amp; echo "健康" || echo "端口 $PORT 无响应"
    else
      echo "shim      : 停止"
    fi
    for spec in "llama 3.6 :8081" "llama 3.8 :8080"; do
      set -- $spec; p="${3#:}"
      printf '%-13s: ' "$1 $2"
      curl -fsS --max-time 3 "http://127.0.0.1:$p/health" 2&gt;/dev/null | grep -q ok &amp;&amp; echo 运行中 || echo 未运行
    done ;;
  *) sed -n '2,8p' "$0"; exit 1 ;;
esac
</code></pre>
<h3>4.3 Shim 工作原理详解</h3>
<p dir="auto"><strong>请求流程</strong>：</p>
<ol>
<li>DSH 发送 OpenAI 格式请求（含 tools 定义）到 :8090</li>
<li>shim 保留 tools 但把 tool_choice 改为 none</li>
<li>转发到 llama.cpp，模型输出原始 XML 工具调用</li>
<li>shim 边收流边检测 tool_call 标签</li>
<li>解析 XML 为 {name, args}，按 JSON Schema 还原类型</li>
<li>封装为标准 OpenAI tool_calls 返回给 DSH</li>
</ol>
<p dir="auto"><strong>防跑飞机制</strong>：</p>
<ul>
<li>TOOL_REGION_CAP = 4000 字符：超过则判定跑飞，掐断流</li>
<li>MAX_TOOL_CALLS = 4：最多接受 4 个并行工具调用</li>
<li>重复检测：相同 name+args 的调用自动去重</li>
<li>解析失败重试：最多 3 次（模型下一轮通常写对）</li>
</ul>
<p dir="auto"><strong>环境变量</strong>：</p>
<table class="table table-bordered table-striped">
<thead>
<tr>
<th>变量</th>
<th>默认值</th>
<th>说明</th>
</tr>
</thead>
<tbody>
<tr>
<td>SHIM_PORT</td>
<td>8090</td>
<td>监听端口</td>
</tr>
<tr>
<td>SHIM_UPSTREAM</td>
<td>自动路由</td>
<td>固定上游地址</td>
</tr>
<tr>
<td>SHIM_UPSTREAM_38</td>
<td><a href="http://127.0.0.1:8080" rel="nofollow ugc">http://127.0.0.1:8080</a></td>
<td>Qwen3.8 上游</td>
</tr>
<tr>
<td>SHIM_UPSTREAM_36</td>
<td><a href="http://127.0.0.1:8081" rel="nofollow ugc">http://127.0.0.1:8081</a></td>
<td>Qwen3.6 上游</td>
</tr>
<tr>
<td>SHIM_VERBOSE</td>
<td>0</td>
<td>设为 1 输出调试日志</td>
</tr>
<tr>
<td>SHIM_ALLOW_FALLBACK</td>
<td>0</td>
<td>设为 1 允许跨模型回退</td>
</tr>
<tr>
<td>SHIM_TOOL_REGION_CAP</td>
<td>4000</td>
<td>工具调用区最大字符数</td>
</tr>
<tr>
<td>SHIM_MAX_TOOL_CALLS</td>
<td>4</td>
<td>最大并行工具调用数</td>
</tr>
</tbody>
</table>
<hr />
<h2>5. 配置 DSH 接入</h2>
<h3>5.1 配置方式</h3>
<p dir="auto">真实文件有两份，内容一致、用途不同：</p>
<table class="table table-bordered table-striped">
<thead>
<tr>
<th>文件</th>
<th>用途</th>
</tr>
</thead>
<tbody>
<tr>
<td><code>~/.dsh/profiles/web/cordis.patch.yml</code></td>
<td><strong>实际生效的</strong> DSH Web profile 路由层</td>
</tr>
<tr>
<td><code>~/llama-bin/dsh-shim-overlay.yml</code></td>
<td>headless 端到端验证用的 <code>--patch</code> 覆盖层（下方代码块即此文件）</td>
</tr>
<tr>
<td><code>~/llama-bin/dsh-direct-overlay.yml</code></td>
<td>同上的直连版（<code>:8080</code>，无 shim），供 A/B 用</td>
</tr>
</tbody>
</table>
<pre><code class="language-yaml"># headless 端到端验证：DSH → shim(8090) → llama.cpp(8080)
# 默认模型 qwen3.8-27b（:8080，与 llama.sh 默认一致）。
# 3.6（:8081）默认停用，需先 ./llama.sh 36 才会起；否则 shim 会返回 502 UPSTREAM_UNREACHABLE。
- id: llm-pi-ai
  name: "@deepseek-ai/dsh-llm-pi-ai"
  config:
    providers:
      local:
        displayName: 本地 llama.cpp
        api: openai-completions
        baseURL: http://127.0.0.1:8090/v1
        headers:
          authorization: Bearer local-no-auth
        compat:
          supportsStore: false
          supportsDeveloperRole: false
          supportsReasoningEffort: false
          maxTokensField: max_tokens
          supportsStrictMode: false
          supportsLongCacheRetention: false
        models:
          - id: qwen3.8-27b
            name: Qwen3.8-27B (本地 :8080)
            contextWindow: 131072
            maxTokens: 32768
          - id: qwen3.6-27b
            name: Qwen3.6-27B (本地 :8081，需手动启动)
            contextWindow: 196608
            maxTokens: 32768
- id: agent-default-model
  name: "@deepseek-ai/dsh-agent-default-model"
  config:
    provider: local
    model: qwen3.8-27b
</code></pre>
<h3>5.2 配置说明</h3>
<table class="table table-bordered table-striped">
<thead>
<tr>
<th>字段</th>
<th>值</th>
<th>说明</th>
</tr>
</thead>
<tbody>
<tr>
<td>api</td>
<td>openai-completions</td>
<td>OpenAI 兼容 API</td>
</tr>
<tr>
<td>baseURL</td>
<td><a href="http://127.0.0.1:8090/v1" rel="nofollow ugc">http://127.0.0.1:8090/v1</a></td>
<td>指向 shim</td>
</tr>
<tr>
<td>headers.authorization</td>
<td>Bearer local-no-auth</td>
<td>占位密钥（服务端未开认证，但 pi-ai 缺了会拒发）</td>
</tr>
<tr>
<td>compat.maxTokensField</td>
<td>max_tokens</td>
<td>llama.cpp 用 max_tokens</td>
</tr>
<tr>
<td>contextWindow (3.8)</td>
<td>131072</td>
<td><strong>必须等于 <code>llama.sh</code> 的 <code>P38_CTX</code></strong>。声明大了 DSH 不会提前压缩，直到服务端报 <code>CONTEXT_OVERFLOW</code></td>
</tr>
<tr>
<td>maxTokens (3.8)</td>
<td>32768</td>
<td>输出上限须给 prompt 留余量：3.8 的窗口 128K，留 ~96K 给 prompt</td>
</tr>
<tr>
<td>contextWindow (3.6)</td>
<td>196608</td>
<td>对应 <code>P36_CTX</code>；换 3.6 前先确认 <code>llama.sh 36</code> 起来了</td>
</tr>
</tbody>
</table>
<blockquote>
<p dir="auto"><code>contextWindow</code> / <code>maxTokens</code> 三处必须对齐：<code>llama.sh</code> 的 <code>P38_CTX</code> → overlay → 运行中的<br />
<code>/v1/models</code> 返回的 <code>n_ctx</code>。对不上时以服务端为准，改配置后 <code>systemctl --user restart qwen38</code>。</p>
</blockquote>
<h3>5.3 设置默认模型</h3>
<pre><code class="language-yaml">- id: agent-default-model
  name: "@deepseek-ai/dsh-agent-default-model"
  config:
    provider: local
    model: qwen3.8-27b
</code></pre>
<hr />
]]></description><link>https://lcz.me/topic/2014</link><generator>RSS for Node</generator><lastBuildDate>Fri, 02 Oct 2026 17:10:44 GMT</lastBuildDate><atom:link href="https://lcz.me/topic/2014.rss" rel="self" type="application/rss+xml"/><pubDate>Wed, 30 Sep 2026 16:05:09 GMT</pubDate><ttl>60</ttl></channel></rss>