/**
 * warmup.ts — llama-server slot warm-up + slot stats + slot
 * prompt-cache save/restore/erase.
 *
 * Ported verbatim from pi-prefix-cache/warmup.ts: this is pure
 * llama-server HTTP and is client-agnostic. All llama-server details
 * live here; if the server API changes, only this file needs edits.
 *
 * Warm-up mechanics:
 *   - POST /v1/chat/completions with n_predict: 0 evaluates the prompt
 *     without generating; the slot retains the KV afterwards.
 *   - `"slot": N` request field selects the target slot.
 *   - GET /slots exposes per-slot cumulative counters:
 *       n_prompt_tokens          total prompt tokens seen
 *       n_prompt_tokens_processed  tokens actually evaluated
 *       n_prompt_tokens_cache    tokens served from KV cache
 *
 * Differences from the pi version: `cache_prompt: true` is sent
 * explicitly (server default, but a warm-up must not depend on a
 * default), and the probe text says "oc-prefix-cache" (cosmetic).
 */

export interface SlotStats {
  id: number;
  n_ctx: number;
  is_processing: boolean;
  n_prompt_tokens: number;
  n_prompt_tokens_processed: number;
  n_prompt_tokens_cache: number;
}

export interface WarmupTimings {
  cache_n?: number;
  prompt_n?: number;
  predict_n?: number;
}

export interface WarmupResult {
  status: number;
  timings?: WarmupTimings;
}

function stripTrailingSlash(url: string): string {
  return url.replace(/\/+$/, "");
}

/**
 * Server root from the API base URL: /slots lives at the server root,
 * not under /v1. `http://host:port/v1` → `http://host:port`.
 */
function serverRoot(baseUrl: string): string {
  return stripTrailingSlash(baseUrl).replace(/\/v1$/, "");
}

/**
 * Evaluate only the shared prefix in a dedicated slot (no generation).
 * The rendered token stream of this request must be a prefix of the
 * rendered token stream of the real opencode request (same chat
 * template, same role, same stable content, same tools).
 *
 * NOTE (Qwen3.8 template, verified 2026-09-23 on the pi side and
 * re-verified 2026-10-02 against a real opencode 1.18.34 body):
 *  - the template raises "No user query found in messages" when no user
 *    message is present, so the warm-up carries an empty user message.
 *    It renders a (near-empty) user block right after the system
 *    content, which is exactly where the real request diverges.
 *  - when `tools` is present, the template renders a tools system block
 *    ("# Tools … <tools>…</tools>") BEFORE the system message content,
 *    inside the same system block. The warm-up must therefore carry the
 *    SAME tools array as the real request (verbatim, same order), or
 *    the rendered token streams diverge at token ~3 and the warmed KV
 *    is never reused. For opencode this matters more than for pi: the
 *    18 tool schemas are 5,806 of the 8,619 rendered tokens.
 */
export async function warmSlot(
  baseUrl: string,
  model: string,
  sysRole: string,
  stableText: string,
  chatTemplateKwargs: unknown,
  slot: number,
  /** Non-empty user content makes the token stream diverge right after
   *  the stable system content — the same boundary as the normal Pi
   *  request. Used by the divergent restore probe. */
  userContent = "",
  /** Must be the same tools array as the normal request (the Qwen3.8
   *  template renders tool schemas before the system content). */
  tools: unknown = undefined,
): Promise<WarmupResult> {
  const body: Record<string, unknown> = {
    model,
    messages: [
      { role: sysRole, content: stableText },
      { role: "user", content: userContent },
    ],
    stream: false,
    n_predict: 0,
    cache_prompt: true,
    slot,
  };
  if (Array.isArray(tools) && tools.length > 0) {
    body.tools = tools;
  }
  if (chatTemplateKwargs !== null && typeof chatTemplateKwargs === "object") {
    body.chat_template_kwargs = chatTemplateKwargs;
  }

  const res = await fetch(`${stripTrailingSlash(baseUrl)}/chat/completions`, {
    method: "POST",
    headers: { "content-type": "application/json" },
    body: JSON.stringify(body),
    // A hung server must not block the normal request forever.
    signal: AbortSignal.timeout(120_000),
  });

  if (!res.ok) {
    const text = await res.text().catch(() => "");
    throw new Error(`warm-up HTTP ${res.status}: ${text.slice(0, 300)}`);
  }

  const json = (await res.json()) as { timings?: WarmupTimings };
  return { status: res.status, timings: json.timings };
}

/**
 * Divergent verification probe for a restored slot: identical prompt to
 * warmSlot except for a non-empty user message, so the request cannot
 * be served by a plain KV token match — it must resume from a context
 * checkpoint before the divergence point. A valid restore yields
 * cache_n > 0; cache_n = 0 means no usable checkpoint (usually a
 * server-flags issue, e.g. --checkpoint-min-step larger than the
 * prefix).
 */
export function probeSlot(
  baseUrl: string,
  model: string,
  sysRole: string,
  stableText: string,
  chatTemplateKwargs: unknown,
  slot: number,
  tools: unknown = undefined,
): Promise<WarmupResult> {
  return warmSlot(baseUrl, model, sysRole, stableText, chatTemplateKwargs, slot, "oc-prefix-cache divergent probe", tools);
}

/** Per-slot counters from GET /slots; null if the slot id is absent. */
export async function getSlotStats(baseUrl: string, slot: number): Promise<SlotStats | null> {
  const res = await fetch(`${serverRoot(baseUrl)}/slots`, {
    signal: AbortSignal.timeout(5_000),
  });
  if (!res.ok) {
    throw new Error(`GET /slots HTTP ${res.status}`);
  }
  const slots = (await res.json()) as SlotStats[];
  return slots.find((s) => s.id === slot) ?? null;
}

// ---------------------------------------------------------------------------
// Phase 5: slot prompt-cache save/restore (plan §14).
//
// Endpoints (llama-server, server root — not /v1):
//   POST /slots/{id}?action=save     {"filename": ...} → writes into the
//   POST /slots/{id}?action=restore  server's --slot-save-path
//   POST /slots/{id}?action=erase
// A restore of a missing file returns HTTP 400 with an error body.
// ---------------------------------------------------------------------------

export interface SlotFileResult {
  status: number;
  ok: boolean;
  n_saved?: number;
  n_restored?: number;
  n_erased?: number;
  timings?: { save_ms?: number; restore_ms?: number };
  error?: string;
}

async function slotFileAction(
  baseUrl: string,
  slot: number,
  action: "save" | "restore" | "erase",
  filename?: string,
): Promise<SlotFileResult> {
  const body: Record<string, unknown> = {};
  if (filename) body.filename = filename;

  const res = await fetch(`${serverRoot(baseUrl)}/slots/${slot}?action=${action}`, {
    method: "POST",
    headers: { "content-type": "application/json" },
    body: JSON.stringify(body),
    // Restore of a large slot (e.g. 27B) can take a while.
    signal: AbortSignal.timeout(120_000),
  });

  // Read the body as text so non-JSON 4xx responses keep their detail.
  const text = await res.text().catch(() => "");
  let json: {
    n_saved?: number;
    n_restored?: number;
    n_erased?: number;
    timings?: { save_ms?: number; restore_ms?: number };
    error?: { code?: number; message?: string };
  } = {};
  try {
    json = JSON.parse(text) as typeof json;
  } catch {
    /* non-JSON body; keep raw text as the error detail */
  }

  return {
    status: res.status,
    ok: res.ok,
    n_saved: json.n_saved,
    n_restored: json.n_restored,
    n_erased: json.n_erased,
    timings: json.timings,
    error: json.error?.message ?? (res.ok ? undefined : text.slice(0, 300)),
  };
}

/** Save the slot's prompt cache to `<slot-save-path>/<filename>`. */
export function saveSlotFile(baseUrl: string, slot: number, filename: string): Promise<SlotFileResult> {
  return slotFileAction(baseUrl, slot, "save", filename);
}

/** Restore the slot's prompt cache from `<slot-save-path>/<filename>`. */
export function restoreSlotFile(baseUrl: string, slot: number, filename: string): Promise<SlotFileResult> {
  return slotFileAction(baseUrl, slot, "restore", filename);
}

/** Erase the slot's prompt cache (KV). */
export function eraseSlotFile(baseUrl: string, slot: number): Promise<SlotFileResult> {
  return slotFileAction(baseUrl, slot, "erase");
}
