// The text↔multimodal agentic split. A strong text-only model fails outright
// when an agent hands it an image, and the available vision model reasons too
// weakly to be trusted with the conversation. So the text model stays in
// control and the vision model becomes a tool it calls: images are stripped to
// `[Image #N]` placeholders, a synthetic `view_image` tool is injected, and a
// bounded agentic loop services each `view_image` call against the vision model
// until the text model returns a final, image-aware answer.
//
// The client never learns any of this happened — `view_image` is not in its
// request tools and the final answer carries none.

import type { AgentId } from '../../core/agent.ts'
import { isVision, type ModelApi } from '../../core/model-registry.ts'
import type { NormalizedBlock, RiskTag } from '../../core/packet.ts'
import {
  type IRBlock,
  type IRMessage,
  type IRRequest,
  type IRResponse,
  type IRTool,
  parseRequestToIR,
  serializeRequestFromIR,
} from '../translate/index.ts'
import type { RoutedAttempt } from './capture.ts'
import { type AttemptResult, execAttempt } from './exec.ts'
import { dropRequest, getImage, newRequestId, putImage } from './image-cache.ts'
import type { ResolvedMember, ResolvedToolModel } from './resolve.ts'

/** Tool names Thomas synthesizes and injects — they are never the agent's own,
 *  so they must not be attributed to it in the tool-usage tables. */
export const SYNTHETIC_TOOLS: ReadonlySet<string> = new Set(['view_image'])

const VIEW_IMAGE = 'view_image'

const VIEW_IMAGE_TOOL: IRTool = {
  name: VIEW_IMAGE,
  description:
    'Inspect an image that appears in the conversation as an [Image #N] placeholder. ' +
    'Provide the image number and a specific question; you receive a textual answer. ' +
    'Call this whenever you need to know what an image contains.',
  inputSchema: {
    type: 'object',
    properties: {
      image_id: {
        type: 'integer',
        description: 'The number N from the [Image #N] placeholder.',
      },
      question: {
        type: 'string',
        description: 'A specific question to ask about the image.',
      },
    },
    required: ['image_id', 'question'],
  },
}

const VISION_SYSTEM_INSTRUCTION =
  'You cannot see images directly. Every image in this conversation has been ' +
  'replaced with an [Image #N] placeholder. Whenever you need visual information ' +
  'from an image, you MUST call the view_image tool with its image_id and a ' +
  'specific question. Do not guess what an image contains — always inspect it ' +
  'with view_image first.'

const VISION_CALL_SYSTEM =
  'Answer strictly about the provided image and only the specific question asked. ' +
  'Be concise, concrete, and factual. Do not speculate beyond what is visible.'

/** Token ceiling for one vision-model call — it answers a single focused
 *  question, so a small budget is ample. */
const VISION_CALL_MAX_TOKENS = 1024

const FALLBACK_TEXT = 'Unable to complete image analysis.'

// ── assessment ────────────────────────────────────────────────────

/** Whether a chain member needs the text↔multimodal split, and the two
 *  models it runs between when it does. */
export type VisionAssessment =
  | { kind: 'none' }
  | { kind: 'loop'; textMember: ResolvedMember; visionMember: ResolvedMember }

/** Decide whether a chain member must run the vision loop: the route
 *  carries a vision companion, the member is text-only, and the request
 *  carries at least one image. A member that can already see images, or a
 *  route with no vision companion, needs no split. */
export function assessVision(
  textMember: ResolvedMember,
  visionToolModel: ResolvedToolModel | undefined,
  clientApi: ModelApi,
  clientRequest: Record<string, unknown>,
): VisionAssessment {
  // No vision companion on the route — nothing to split to.
  if (!visionToolModel) return { kind: 'none' }
  // The member can already see images — no split needed.
  if (isVision(textMember.model)) return { kind: 'none' }

  const ir = parseRequestToIR(clientApi, clientRequest)
  if (countImages(ir) === 0) return { kind: 'none' }

  return {
    kind: 'loop',
    textMember,
    visionMember: {
      model: visionToolModel.model,
      provider: visionToolModel.provider,
      api: visionToolModel.api,
      switchOn: [],
    },
  }
}

/** Count image blocks anywhere in a request — including nested in
 *  `tool_result` content, where Claude Code delivers screenshots. */
export function countImages(ir: IRRequest): number {
  let n = 0
  const walk = (blocks: IRBlock[]): void => {
    for (const b of blocks) {
      if (b.type === 'image') n++
      else if (b.type === 'tool_result') walk(b.content)
    }
  }
  for (const m of ir.messages) walk(m.content)
  return n
}

// ── the loop ──────────────────────────────────────────────────────

export type VisionLoopParams = {
  clientApi: ModelApi
  clientRequest: Record<string, unknown>
  textMember: ResolvedMember
  visionMember: ResolvedMember
  ctx: { agent: AgentId; reqHeaders: Headers }
  maxIterations: number
}

export type VisionLoopOutcome = {
  /** Every upstream call the loop made — text-model turns and vision calls. */
  attempts: RoutedAttempt[]
  /** The final client-facing answer (`view_image` calls stripped out). */
  finalIR: IRResponse
  risk: RiskTag[]
  /** False when the loop could not produce an answer (a text turn failed). */
  ok: boolean
}

/** Run the bounded agentic loop. Never throws — an upstream failure ends the
 *  loop with `ok: false`; the cached images are always dropped on the way out. */
export async function executeVisionLoop(params: VisionLoopParams): Promise<VisionLoopOutcome> {
  const { clientApi, clientRequest, textMember, visionMember, ctx, maxIterations } = params
  const requestId = newRequestId()
  const attempts: RoutedAttempt[] = []
  const risk: RiskTag[] = []
  let step = 0

  try {
    const baseIR = parseRequestToIR(clientApi, clientRequest)
    prepareVisionRequest(baseIR, requestId)
    const conversation: IRMessage[] = baseIR.messages

    for (let iter = 1; iter <= maxIterations; iter++) {
      // ── text-model turn ──────────────────────────────────────────
      const textIR: IRRequest = {
        model: textMember.model.id,
        system: baseIR.system,
        messages: conversation,
        tools: baseIR.tools,
        maxTokens: baseIR.maxTokens,
        temperature: baseIR.temperature,
        stream: false,
      }
      const textBody = serializeRequestFromIR(textMember.api, textIR) as Record<string, unknown>
      const textResult = await execAttempt(textMember, textMember.api, textBody, ctx)
      attempts.push({
        member: textMember,
        result: textResult,
        role: 'text-turn',
        step: step++,
        request: { api: textMember.api, body: textBody },
      })

      if (!textResult.ok) {
        risk.push({
          tag: 'vision:text-failed',
          severity: 'high',
          detail: textResult.errorText,
        })
        return { attempts, finalIR: fallbackIR(textMember.model.id), risk, ok: false }
      }

      const blocks = textResult.ir.blocks
      const viewCalls = blocks.filter(
        (b): b is Extract<NormalizedBlock, { type: 'tool_use' }> =>
          b.type === 'tool_use' && b.name === VIEW_IMAGE,
      )

      // No view_image calls — the text model has its final, image-aware answer.
      if (viewCalls.length === 0) {
        return {
          attempts,
          finalIR: { ...textResult.ir, blocks: finalizeBlocks(blocks) },
          risk,
          ok: true,
        }
      }

      // Out of loop budget with image questions still pending.
      if (iter === maxIterations) {
        risk.push({
          tag: 'vision:loop-exhausted',
          severity: 'warn',
          detail: `reached ${maxIterations} iterations`,
        })
        return {
          attempts,
          finalIR: { ...textResult.ir, blocks: finalizeBlocks(blocks) },
          risk,
          ok: true,
        }
      }

      // ── service each view_image call against the vision model ────
      conversation.push(assistantTurn(blocks, viewCalls))
      const results: IRBlock[] = []
      for (const call of viewCalls) {
        const serviced = await serviceViewImage(call, requestId, visionMember, ctx)
        if (serviced.attempt) {
          attempts.push({
            member: visionMember,
            result: serviced.attempt.result,
            role: 'vision',
            step: step++,
            request: { api: visionMember.api, body: serviced.attempt.body },
          })
        }
        results.push(serviced.toolResult)
      }
      conversation.push({ role: 'user', content: results })
    }

    // Unreachable — the maxIterations branch always returns inside the loop.
    return { attempts, finalIR: fallbackIR(textMember.model.id), risk, ok: true }
  } finally {
    dropRequest(requestId)
  }
}

// ── request preparation ───────────────────────────────────────────

/** Mutate an IR request in place for the text model: replace every image with
 *  an `[Image #N]` placeholder (parking the bytes in the cache), drop stale
 *  `thinking` blocks (their signatures do not survive re-serialization), inject
 *  the `view_image` tool, and prepend the system instruction. */
function prepareVisionRequest(ir: IRRequest, requestId: string): void {
  let ordinal = 0
  const substitute = (blocks: IRBlock[]): IRBlock[] => {
    const out: IRBlock[] = []
    for (const b of blocks) {
      if (b.type === 'thinking') continue
      if (b.type === 'image') {
        ordinal++
        putImage(requestId, ordinal, b.source)
        out.push({ type: 'text', text: `[Image #${ordinal}]` })
      } else if (b.type === 'tool_result') {
        out.push({ ...b, content: substitute(b.content) })
      } else {
        out.push(b)
      }
    }
    return out
  }
  for (const m of ir.messages) m.content = substitute(m.content)
  ir.tools = [...(ir.tools ?? []), VIEW_IMAGE_TOOL]
  ir.system = ir.system
    ? `${ir.system}\n\n${VISION_SYSTEM_INSTRUCTION}`
    : VISION_SYSTEM_INSTRUCTION
}

// ── view_image servicing ──────────────────────────────────────────

type ServicedCall = {
  /** The tool_result block fed back to the text model. */
  toolResult: IRBlock
  /** The vision-model call, when one was made (absent for a bad image id). */
  attempt?: { result: AttemptResult; body: Record<string, unknown> }
}

/** Answer one `view_image` call: look up the cached image and ask the vision
 *  model the question. A missing/out-of-range id yields an error tool_result
 *  and no upstream call; a vision-model failure yields an error tool_result
 *  that still records the attempt. */
async function serviceViewImage(
  call: Extract<NormalizedBlock, { type: 'tool_use' }>,
  requestId: string,
  visionMember: ResolvedMember,
  ctx: VisionLoopParams['ctx'],
): Promise<ServicedCall> {
  const input = (call.input ?? {}) as Record<string, unknown>
  const rawId = input.image_id
  const imageId = Number(rawId)
  const question = typeof input.question === 'string' ? input.question : ''

  const source = Number.isFinite(imageId) ? getImage(requestId, imageId) : undefined
  if (!source) {
    return {
      toolResult: errorResult(
        call.id,
        `No image with id ${String(rawId)} is available. Image ids appear as ` +
          `[Image #N] in the conversation.`,
      ),
    }
  }

  const visionIR: IRRequest = {
    model: visionMember.model.id,
    system: VISION_CALL_SYSTEM,
    messages: [
      {
        role: 'user',
        content: [
          { type: 'image', source },
          { type: 'text', text: question || 'Describe this image in detail.' },
        ],
      },
    ],
    maxTokens: VISION_CALL_MAX_TOKENS,
    stream: false,
  }
  const body = serializeRequestFromIR(visionMember.api, visionIR) as Record<string, unknown>
  const result = await execAttempt(visionMember, visionMember.api, body, ctx)

  if (!result.ok) {
    return {
      toolResult: errorResult(call.id, `Image analysis failed: ${result.errorText}`),
      attempt: { result, body },
    }
  }

  const answer = textOf(result.ir.blocks)
  return {
    toolResult: {
      type: 'tool_result',
      toolUseId: call.id,
      content: [{ type: 'text', text: answer || '(the vision model returned no text)' }],
    },
    attempt: { result, body },
  }
}

// ── block helpers ─────────────────────────────────────────────────

/** The assistant turn fed back into the loop — its text plus the `view_image`
 *  calls only. Real tool calls are dropped (their results are the agent's, not
 *  ours to supply) and so is thinking (no signature survives the round-trip). */
function assistantTurn(
  blocks: NormalizedBlock[],
  viewCalls: Array<Extract<NormalizedBlock, { type: 'tool_use' }>>,
): IRMessage {
  const content: IRBlock[] = []
  for (const b of blocks) {
    if (b.type === 'text' && b.text.length > 0) content.push({ type: 'text', text: b.text })
  }
  for (const call of viewCalls) {
    content.push({ type: 'tool_use', id: call.id, name: call.name, input: call.input })
  }
  return { role: 'assistant', content }
}

/** Strip `view_image` calls from the final answer; if nothing the client can
 *  act on remains, synthesize a minimal text block so the response is valid. */
function finalizeBlocks(blocks: NormalizedBlock[]): NormalizedBlock[] {
  const kept = blocks.filter((b) => !(b.type === 'tool_use' && b.name === VIEW_IMAGE))
  const actionable = kept.some(
    (b) => (b.type === 'text' && b.text.trim().length > 0) || b.type === 'tool_use',
  )
  if (actionable) return kept
  return [{ type: 'text', text: FALLBACK_TEXT }, ...kept]
}

function errorResult(toolUseId: string, text: string): IRBlock {
  return {
    type: 'tool_result',
    toolUseId,
    content: [{ type: 'text', text }],
    isError: true,
  }
}

function textOf(blocks: NormalizedBlock[]): string {
  return blocks
    .filter((b): b is Extract<NormalizedBlock, { type: 'text' }> => b.type === 'text')
    .map((b) => b.text)
    .join('\n')
    .trim()
}

function fallbackIR(model: string): IRResponse {
  return {
    model,
    blocks: [{ type: 'text', text: FALLBACK_TEXT }],
    stopReason: 'end_turn',
    usage: { in: 0, out: 0 },
  }
}
