import os from "node:os"; import { parseArgs } from "node:util"; import { createOpenAICompatible } from "@ai-sdk/openai-compatible"; import { type ModelMessage, streamText } from "ai"; import StreamRenderer from "./src/render"; const { values: args, positionals } = parseArgs({ args: Bun.argv, options: { model: { type: "string", }, new: { type: "boolean", }, }, strict: true, allowPositionals: true, }); const home = os.homedir(); const convoFile = Bun.file(`${home}/.aicl_convo.json`); async function main() { const provider = createOpenAICompatible({ name: "llama.cpp", apiKey: "local", baseURL: "http://127.0.0.1:8080/v1", includeUsage: true, // Include usage information in streaming responses }); const messages: ModelMessage[] = []; if (!args.new) { try { const convo = await convoFile.json(); messages.push(...convo); } catch {} } const prompt = positionals[2] ?? "Introduce yourself."; messages.push({ role: "user", content: prompt }); const { fullStream, textStream, text, totalUsage } = streamText({ model: provider(args.model ?? "qwen3.5-35b-a3b"), messages, }); const start = Date.now(); const streamRender = new StreamRenderer().renderStream(fullStream); const [width] = process.stdout.getWindowSize(); const glow = Bun.spawn({ cmd: ["glow", "-w", `${width}`, "-"], stdin: textStream, stdout: "inherit", }); await streamRender; const stop = Date.now(); await glow.exited; messages.push({ role: "assistant", content: await text }); await convoFile.write(JSON.stringify(messages)); const { inputTokens, outputTokens } = await totalUsage; const context = (inputTokens ?? 0) / 1000; const output = (outputTokens ?? 0) / 1000; const time = (stop - start) / 1000; process.stderr.write( `\u001b[2mContext: ${context.toFixed(1)}k, Output: ${output.toFixed(1)}k, Time: ${time.toFixed(2)}s\u001b[22m\n`, ); } await main();