File size: 2,932 Bytes
0ef9612
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
// Streaming client for Inkling via the Hugging Face Inference Providers router.

export const MODEL_ID = "thinkingmachines/Inkling";
const ENDPOINT = "https://router.huggingface.co/together/v1/chat/completions";

/** Build the single multimodal user message. */
export function buildMessage(prompt, imageUri, audioB64) {
  const parts = [];
  if (imageUri) {
    parts.push({ type: "image_url", image_url: { url: imageUri } });
  }
  if (audioB64) {
    parts.push({ type: "input_audio", input_audio: { data: audioB64, format: "wav" } });
  }
  parts.push({ type: "text", text: prompt });
  return { role: "user", content: parts };
}

/** Pull a human-readable message out of a provider error body. */
function describeError(status, body) {
  try {
    const parsed = JSON.parse(body);
    const inner = parsed?.error?.message ?? parsed?.message;
    if (typeof inner === "string") return inner;
    if (inner?.message) return inner.message;
  } catch {
    /* fall through to the raw body */
  }
  return body?.slice(0, 400) || `Request failed with status ${status}.`;
}

/**
 * Stream a completion, invoking callbacks as tokens arrive.
 *
 * Inkling is a reasoning model: it emits `reasoning` deltas before committing
 * to `content`, so both are surfaced separately.
 */
export async function streamCompletion({
  token,
  prompt,
  imageUri = null,
  audioB64 = null,
  maxTokens = 2048,
  temperature = 0.7,
  signal,
  onReasoning = () => {},
  onContent = () => {},
}) {
  const response = await fetch(ENDPOINT, {
    method: "POST",
    signal,
    headers: {
      Authorization: `Bearer ${token}`,
      "Content-Type": "application/json",
    },
    body: JSON.stringify({
      model: MODEL_ID,
      messages: [buildMessage(prompt, imageUri, audioB64)],
      max_tokens: maxTokens,
      temperature,
      stream: true,
    }),
  });

  if (!response.ok) {
    throw new Error(describeError(response.status, await response.text()));
  }

  const reader = response.body.getReader();
  const decoder = new TextDecoder();
  let buffer = "";

  for (;;) {
    const { done, value } = await reader.read();
    if (done) break;

    buffer += decoder.decode(value, { stream: true });
    const lines = buffer.split("\n");
    // Keep the trailing fragment; it may be half an event.
    buffer = lines.pop() ?? "";

    for (const line of lines) {
      const trimmed = line.trim();
      if (!trimmed.startsWith("data:")) continue;

      const payload = trimmed.slice(5).trim();
      if (payload === "[DONE]") return;

      let parsed;
      try {
        parsed = JSON.parse(payload);
      } catch {
        continue; // Ignore keep-alives and partial frames.
      }

      const delta = parsed?.choices?.[0]?.delta;
      if (!delta) continue;

      const thought = delta.reasoning ?? delta.reasoning_content;
      if (thought) onReasoning(thought);
      if (delta.content) onContent(delta.content);
    }
  }
}