refactor(look-at): split tools.ts to comply with 200 LOC module rule

Extract input preparation and image conversion handling into look-at-input-preparer.ts.

Extract prompt construction and multimodal session execution into look-at-prompt.ts and look-at-session-runner.ts while keeping createLookAt stable.

Ultraworked with [Sisyphus](https://github.com/code-yeongyu/oh-my-openagent)

Co-authored-by: Sisyphus <clio-agent@sisyphuslabs.ai>
This commit is contained in:
YeonGyu-Kim
2026-04-18 01:45:30 +09:00
parent f94632ce64
commit 963355d241
4 changed files with 297 additions and 209 deletions
+154
View File
@@ -0,0 +1,154 @@
import { basename } from "node:path"
import { pathToFileURL } from "node:url"
import type { LookAtArgs } from "./types"
import {
extractBase64Data,
inferMimeTypeFromBase64,
inferMimeTypeFromFilePath,
} from "./mime-type-inference"
import {
needsConversion,
convertImageToJpeg,
convertBase64ImageToJpeg,
cleanupConvertedImage,
} from "./image-converter"
import { log } from "../../shared"
export interface LookAtFilePart {
type: "file"
mime: string
url: string
filename: string
}
export interface PreparedLookAtInput {
readonly filePart: LookAtFilePart
readonly isBase64Input: boolean
readonly sourceDescription: string
cleanup(): void
}
type PrepareLookAtInputResult =
| { ok: true; value: PreparedLookAtInput }
| { ok: false; error: string }
function getTemporaryConversionPath(error: unknown): string | null {
if (!(error instanceof Error)) {
return null
}
const temporaryOutputPath = Reflect.get(error, "temporaryOutputPath")
if (typeof temporaryOutputPath === "string" && temporaryOutputPath.length > 0) {
return temporaryOutputPath
}
const temporaryDirectory = Reflect.get(error, "temporaryDirectory")
if (typeof temporaryDirectory === "string" && temporaryDirectory.length > 0) {
return temporaryDirectory
}
return null
}
export function prepareLookAtInput(args: LookAtArgs): PrepareLookAtInputResult {
const imageData = args.image_data
const filePath = args.file_path
if (imageData) {
const mimeType = inferMimeTypeFromBase64(imageData)
let finalBase64Data = extractBase64Data(imageData)
let finalMimeType = mimeType
let tempFilesToCleanup: string[] = []
if (needsConversion(mimeType)) {
log(`[look_at] Detected unsupported Base64 format: ${mimeType}, converting to JPEG...`)
try {
const { base64, tempFiles } = convertBase64ImageToJpeg(finalBase64Data, mimeType)
finalBase64Data = base64
finalMimeType = "image/jpeg"
tempFilesToCleanup = tempFiles
log("[look_at] Base64 conversion successful")
} catch (conversionError) {
log(`[look_at] Base64 conversion failed: ${conversionError}`)
return {
ok: false,
error: `Error: Failed to convert Base64 image format. ${conversionError}`,
}
}
}
return {
ok: true,
value: {
isBase64Input: true,
sourceDescription: "clipboard/pasted image",
filePart: {
type: "file",
mime: finalMimeType,
url: `data:${finalMimeType};base64,${finalBase64Data}`,
filename: `clipboard-image.${finalMimeType.split("/")[1] || "png"}`,
},
cleanup() {
for (const temporaryFile of tempFilesToCleanup) {
cleanupConvertedImage(temporaryFile)
}
},
},
}
}
if (filePath) {
let mimeType = inferMimeTypeFromFilePath(filePath)
let actualFilePath = filePath
let tempFilePath: string | null = null
let tempConversionPath: string | null = null
if (needsConversion(mimeType)) {
log(`[look_at] Detected unsupported format: ${mimeType}, converting to JPEG...`)
try {
tempFilePath = convertImageToJpeg(filePath, mimeType)
tempConversionPath = tempFilePath
actualFilePath = tempFilePath
mimeType = "image/jpeg"
log(`[look_at] Conversion successful: ${tempFilePath}`)
} catch (conversionError) {
const failedConversionPath = getTemporaryConversionPath(conversionError)
if (failedConversionPath) {
tempConversionPath = failedConversionPath
}
log(`[look_at] Conversion failed: ${conversionError}`)
return {
ok: false,
error: `Error: Failed to convert image format. ${conversionError}`,
}
}
}
return {
ok: true,
value: {
isBase64Input: false,
sourceDescription: filePath,
filePart: {
type: "file",
mime: mimeType,
url: pathToFileURL(actualFilePath).href,
filename: basename(actualFilePath),
},
cleanup() {
if (tempConversionPath) {
cleanupConvertedImage(tempConversionPath)
} else if (tempFilePath) {
cleanupConvertedImage(tempFilePath)
}
},
},
}
}
return {
ok: false,
error: "Error: Must provide either 'file_path' or 'image_data'.",
}
}
+18
View File
@@ -0,0 +1,18 @@
export const READ_ENABLED = false
export function buildLookAtPrompt(goal: string, isBase64Input: boolean): string {
const subjectNoun = isBase64Input ? "image" : "file"
const sourceClause = READ_ENABLED
? "Use the Read tool on the provided file path to load its contents, then analyze it."
: `The ${subjectNoun} is already attached to this message. Analyze it directly from the attachment. Do NOT attempt to use the Read tool. The Read tool is disabled for this invocation and the ${subjectNoun} cannot be loaded by path.`
return `Analyze the attached ${subjectNoun} and extract the requested information.
${sourceClause}
Goal: ${goal}
Provide ONLY the extracted information that matches the goal.
Be thorough on what was requested, concise on everything else.
If the requested information is not found, clearly state what is missing.`
}
+107
View File
@@ -0,0 +1,107 @@
import type { PluginInput } from "@opencode-ai/plugin"
import type { ToolContext } from "@opencode-ai/plugin/tool"
import { log, promptSyncWithModelSuggestionRetry } from "../../shared"
import { extractLatestAssistantText } from "./assistant-message-extractor"
import { MULTIMODAL_LOOKER_AGENT } from "./constants"
import { READ_ENABLED, buildLookAtPrompt } from "./look-at-prompt"
import type { LookAtFilePart } from "./look-at-input-preparer"
import { resolveMultimodalLookerAgentMetadata } from "./multimodal-agent-metadata"
interface RunLookAtSessionInput {
ctx: PluginInput
toolContext: ToolContext
goal: string
filePart: LookAtFilePart
isBase64Input: boolean
}
export async function runLookAtSession({
ctx,
toolContext,
goal,
filePart,
isBase64Input,
}: RunLookAtSessionInput): Promise<string> {
const prompt = buildLookAtPrompt(goal, isBase64Input)
const { agentModel, agentVariant } = await resolveMultimodalLookerAgentMetadata(ctx)
log(`[look_at] Creating session with parent: ${toolContext.sessionID}`)
const parentSession = await ctx.client.session.get({
path: { id: toolContext.sessionID },
}).catch(() => null)
const parentDirectory = parentSession?.data?.directory ?? ctx.directory
const createResult = await ctx.client.session.create({
body: {
parentID: toolContext.sessionID,
title: `look_at: ${goal.substring(0, 50)}`,
},
query: { directory: parentDirectory },
})
if (createResult.error) {
log("[look_at] Session create error:", createResult.error)
const errorString = String(createResult.error)
if (errorString.toLowerCase().includes("unauthorized")) {
return `Error: Failed to create session (Unauthorized). This may be due to:
1. OAuth token restrictions (e.g., Claude Code credentials are restricted to Claude Code only)
2. Provider authentication issues
3. Session permission inheritance problems
Try using a different provider or API key authentication.
Original error: ${createResult.error}`
}
return `Error: Failed to create session: ${createResult.error}`
}
const sessionID = createResult.data.id
log(`[look_at] Created session: ${sessionID}`)
log(`[look_at] Sending prompt with ${isBase64Input ? "base64 image" : "file"} to session ${sessionID}`)
try {
await promptSyncWithModelSuggestionRetry(ctx.client, {
path: { id: sessionID },
body: {
agent: MULTIMODAL_LOOKER_AGENT,
tools: {
task: false,
call_omo_agent: false,
look_at: false,
read: READ_ENABLED,
},
parts: [
{ type: "text", text: prompt },
filePart,
],
...(agentModel ? { model: { providerID: agentModel.providerID, modelID: agentModel.modelID } } : {}),
...(agentVariant ? { variant: agentVariant } : {}),
},
})
} catch (promptError) {
log("[look_at] Prompt error (ignored, will still fetch messages):", promptError)
}
log(`[look_at] Fetching messages from session ${sessionID}...`)
const messagesResult = await ctx.client.session.messages({
path: { id: sessionID },
})
if (messagesResult.error) {
log("[look_at] Messages error:", messagesResult.error)
return `Error: Failed to get messages: ${messagesResult.error}`
}
const messages = messagesResult.data
log(`[look_at] Got ${messages.length} messages`)
const responseText = extractLatestAssistantText(messages)
if (!responseText) {
log("[look_at] No assistant message found")
return "Error: No response from multimodal-looker agent"
}
log(`[look_at] Got response, length: ${responseText.length}`)
return responseText
}
+18 -209
View File
@@ -1,43 +1,11 @@
import { basename } from "node:path"
import { pathToFileURL } from "node:url"
import { tool, type PluginInput, type ToolDefinition } from "@opencode-ai/plugin"
import { LOOK_AT_DESCRIPTION, MULTIMODAL_LOOKER_AGENT } from "./constants"
import { LOOK_AT_DESCRIPTION } from "./constants"
import type { LookAtArgs } from "./types"
import { log, promptSyncWithModelSuggestionRetry } from "../../shared"
import { extractLatestAssistantText } from "./assistant-message-extractor"
import { log } from "../../shared"
import type { LookAtArgsWithAlias } from "./look-at-arguments"
import { normalizeArgs, validateArgs } from "./look-at-arguments"
import {
extractBase64Data,
inferMimeTypeFromBase64,
inferMimeTypeFromFilePath,
} from "./mime-type-inference"
import { resolveMultimodalLookerAgentMetadata } from "./multimodal-agent-metadata"
import {
needsConversion,
convertImageToJpeg,
convertBase64ImageToJpeg,
cleanupConvertedImage,
} from "./image-converter"
function getTemporaryConversionPath(error: unknown): string | null {
if (!(error instanceof Error)) {
return null
}
const temporaryOutputPath = Reflect.get(error, "temporaryOutputPath")
if (typeof temporaryOutputPath === "string" && temporaryOutputPath.length > 0) {
return temporaryOutputPath
}
const temporaryDirectory = Reflect.get(error, "temporaryDirectory")
if (typeof temporaryDirectory === "string" && temporaryDirectory.length > 0) {
return temporaryDirectory
}
return null
}
import { prepareLookAtInput } from "./look-at-input-preparer"
import { runLookAtSession } from "./look-at-session-runner"
export { normalizeArgs, validateArgs } from "./look-at-arguments"
@@ -57,188 +25,29 @@ export function createLookAt(ctx: PluginInput): ToolDefinition {
return validationError
}
const isBase64Input = Boolean(args.image_data)
const sourceDescription = isBase64Input ? "clipboard/pasted image" : args.file_path
const preparedInputResult = prepareLookAtInput(args)
if (!preparedInputResult.ok) {
return preparedInputResult.error
}
const preparedInput = preparedInputResult.value
const { isBase64Input, sourceDescription } = preparedInput
log(`[look_at] Analyzing ${sourceDescription}, goal: ${args.goal}`)
const imageData = args.image_data
const filePath = args.file_path
let mimeType: string
let filePart: { type: "file"; mime: string; url: string; filename: string }
let tempFilePath: string | null = null
let tempConversionPath: string | null = null
let tempFilesToCleanup: string[] = []
try {
if (imageData) {
mimeType = inferMimeTypeFromBase64(imageData)
let finalBase64Data = extractBase64Data(imageData)
let finalMimeType = mimeType
if (needsConversion(mimeType)) {
log(`[look_at] Detected unsupported Base64 format: ${mimeType}, converting to JPEG...`)
try {
const { base64, tempFiles } = convertBase64ImageToJpeg(finalBase64Data, mimeType)
finalBase64Data = base64
finalMimeType = "image/jpeg"
tempFilesToCleanup = tempFiles
log(`[look_at] Base64 conversion successful`)
} catch (conversionError) {
log(`[look_at] Base64 conversion failed: ${conversionError}`)
return `Error: Failed to convert Base64 image format. ${conversionError}`
}
}
filePart = {
type: "file",
mime: finalMimeType,
url: `data:${finalMimeType};base64,${finalBase64Data}`,
filename: `clipboard-image.${finalMimeType.split("/")[1] || "png"}`,
}
} else if (filePath) {
mimeType = inferMimeTypeFromFilePath(filePath)
let actualFilePath = filePath
if (needsConversion(mimeType)) {
log(`[look_at] Detected unsupported format: ${mimeType}, converting to JPEG...`)
try {
tempFilePath = convertImageToJpeg(filePath, mimeType)
tempConversionPath = tempFilePath
actualFilePath = tempFilePath
mimeType = "image/jpeg"
log(`[look_at] Conversion successful: ${tempFilePath}`)
} catch (conversionError) {
const failedConversionPath = getTemporaryConversionPath(conversionError)
if (failedConversionPath) {
tempConversionPath = failedConversionPath
}
log(`[look_at] Conversion failed: ${conversionError}`)
return `Error: Failed to convert image format. ${conversionError}`
}
}
filePart = {
type: "file",
mime: mimeType,
url: pathToFileURL(actualFilePath).href,
filename: basename(actualFilePath),
}
} else {
return "Error: Must provide either 'file_path' or 'image_data'."
}
const readEnabled = false
const subjectNoun = isBase64Input ? "image" : "file"
const sourceClause = readEnabled
? `Use the Read tool on the provided file path to load its contents, then analyze it.`
: `The ${subjectNoun} is already attached to this message. Analyze it directly from the attachment. Do NOT attempt to use the Read tool. The Read tool is disabled for this invocation and the ${subjectNoun} cannot be loaded by path.`
const prompt = `Analyze the attached ${subjectNoun} and extract the requested information.
${sourceClause}
Goal: ${args.goal}
Provide ONLY the extracted information that matches the goal.
Be thorough on what was requested, concise on everything else.
If the requested information is not found, clearly state what is missing.`
const { agentModel, agentVariant } = await resolveMultimodalLookerAgentMetadata(ctx)
log(`[look_at] Creating session with parent: ${toolContext.sessionID}`)
const parentSession = await ctx.client.session.get({
path: { id: toolContext.sessionID },
}).catch(() => null)
const parentDirectory = parentSession?.data?.directory ?? ctx.directory
const createResult = await ctx.client.session.create({
body: {
parentID: toolContext.sessionID,
title: `look_at: ${args.goal.substring(0, 50)}`,
},
query: { directory: parentDirectory },
})
if (createResult.error) {
log(`[look_at] Session create error:`, createResult.error)
const errorStr = String(createResult.error)
if (errorStr.toLowerCase().includes("unauthorized")) {
return `Error: Failed to create session (Unauthorized). This may be due to:
1. OAuth token restrictions (e.g., Claude Code credentials are restricted to Claude Code only)
2. Provider authentication issues
3. Session permission inheritance problems
Try using a different provider or API key authentication.
Original error: ${createResult.error}`
}
return `Error: Failed to create session: ${createResult.error}`
}
const sessionID = createResult.data.id
log(`[look_at] Created session: ${sessionID}`)
log(`[look_at] Sending prompt with ${isBase64Input ? "base64 image" : "file"} to session ${sessionID}`)
try {
await promptSyncWithModelSuggestionRetry(ctx.client, {
path: { id: sessionID },
body: {
agent: MULTIMODAL_LOOKER_AGENT,
tools: {
task: false,
call_omo_agent: false,
look_at: false,
read: readEnabled,
},
parts: [
{ type: "text", text: prompt },
filePart,
],
...(agentModel ? { model: { providerID: agentModel.providerID, modelID: agentModel.modelID } } : {}),
...(agentVariant ? { variant: agentVariant } : {}),
},
return await runLookAtSession({
ctx,
toolContext,
goal: args.goal,
filePart: preparedInput.filePart,
isBase64Input,
})
} catch (promptError) {
log(`[look_at] Prompt error (ignored, will still fetch messages):`, promptError)
}
log(`[look_at] Fetching messages from session ${sessionID}...`)
const messagesResult = await ctx.client.session.messages({
path: { id: sessionID },
})
if (messagesResult.error) {
log(`[look_at] Messages error:`, messagesResult.error)
return `Error: Failed to get messages: ${messagesResult.error}`
}
const messages = messagesResult.data
log(`[look_at] Got ${messages.length} messages`)
const responseText = extractLatestAssistantText(messages)
if (!responseText) {
log("[look_at] No assistant message found")
return "Error: No response from multimodal-looker agent"
}
log(`[look_at] Got response, length: ${responseText.length}`)
return responseText
} catch (error) {
const errorMessage = error instanceof Error ? error.message : String(error)
log(`[look_at] Unexpected error analyzing ${sourceDescription}:`, error)
return `Error: Failed to analyze ${sourceDescription}: ${errorMessage}`
} finally {
if (tempConversionPath) {
cleanupConvertedImage(tempConversionPath)
} else if (tempFilePath) {
cleanupConvertedImage(tempFilePath)
}
tempFilesToCleanup.forEach(file => {
cleanupConvertedImage(file)
})
preparedInput.cleanup()
}
},
})