Merge pull request #1821 from AndyMik90/auto-claude/224-add-screenshot-paste-capability-to-chat

auto-claude: 224-add-screenshot-paste-capability-to-chat
This commit is contained in:
Andy
2026-02-20 11:54:26 +01:00
committed by GitHub
14 changed files with 598 additions and 94 deletions
+155 -7
View File
@@ -8,8 +8,10 @@ about a codebase. It can also suggest tasks based on the conversation.
import argparse
import asyncio
import base64
import json
import sys
import tempfile
from pathlib import Path
# Add auto-claude to path
@@ -111,6 +113,107 @@ def load_project_context(project_dir: str) -> str:
)
ALLOWED_MIME_TYPES = frozenset(
["image/png", "image/jpeg", "image/jpg", "image/gif", "image/webp"]
)
MAX_IMAGE_FILE_SIZE = 10 * 1024 * 1024 # 10 MB (aligned with frontend MAX_IMAGE_SIZE)
def load_images_from_manifest(manifest_path: str) -> list[dict]:
"""Load images from a manifest JSON file.
The manifest contains an array of objects with 'path' and 'mimeType' fields.
Each image file is read as binary and encoded to base64.
Returns a list of dicts with 'media_type' and 'data' (base64-encoded) fields.
"""
images = []
tmp_dir = Path(tempfile.gettempdir()).resolve()
try:
with open(manifest_path, encoding="utf-8") as f:
manifest = json.load(f)
for entry in manifest:
image_path = entry.get("path")
mime_type = entry.get("mimeType", "image/png")
if not image_path:
debug_error(
"insights_runner",
"Image entry missing path field",
)
continue
# Validate path is within temp directory before checking existence
try:
resolved = Path(image_path).resolve()
if not resolved.is_relative_to(tmp_dir):
debug_error(
"insights_runner",
f"Image path outside temp directory, skipping: {image_path}",
)
continue
except (ValueError, OSError):
debug_error(
"insights_runner",
f"Invalid image path, skipping: {image_path}",
)
continue
if not resolved.exists():
debug_error(
"insights_runner",
f"Image file not found: {image_path}",
)
continue
# Validate MIME type against allowlist
if mime_type not in ALLOWED_MIME_TYPES:
debug_error(
"insights_runner",
f"Invalid MIME type '{mime_type}', skipping: {image_path}",
)
continue
# Validate file size
file_size = resolved.stat().st_size
if file_size > MAX_IMAGE_FILE_SIZE:
debug_error(
"insights_runner",
f"Image too large ({file_size} bytes), skipping: {image_path}",
)
continue
try:
with open(resolved, "rb") as img_f:
image_data = base64.b64encode(img_f.read()).decode("utf-8")
images.append(
{
"media_type": mime_type,
"data": image_data,
}
)
debug(
"insights_runner",
"Loaded image",
path=image_path,
mime_type=mime_type,
size_bytes=file_size,
)
except Exception as e:
debug_error(
"insights_runner",
f"Failed to read image {image_path}: {e}",
)
except (json.JSONDecodeError, OSError) as e:
debug_error("insights_runner", f"Failed to load images manifest: {e}")
return images
def build_system_prompt(project_dir: str) -> str:
"""Build the system prompt for the insights agent."""
context = load_project_context(project_dir)
@@ -143,11 +246,12 @@ async def run_with_sdk(
history: list,
model: str = "sonnet", # Shorthand - resolved via API Profile if configured
thinking_level: str = "medium",
images: list[dict] | None = None,
) -> None:
"""Run the chat using Claude SDK with streaming."""
if not SDK_AVAILABLE:
print("Claude SDK not available, falling back to simple mode", file=sys.stderr)
run_simple(project_dir, message, history)
run_simple(project_dir, message, history, images)
return
if not get_auth_token():
@@ -155,7 +259,7 @@ async def run_with_sdk(
"No authentication token found, falling back to simple mode",
file=sys.stderr,
)
run_simple(project_dir, message, history)
run_simple(project_dir, message, history, images)
return
# Ensure SDK can find the token
@@ -205,8 +309,24 @@ Current question: {message}"""
# Use async context manager pattern
async with client:
# Send the query
await client.query(full_prompt)
# Build the query - images are stored for reference but SDK doesn't support multi-modal input yet
if images:
debug(
"insights_runner",
"Images attached but SDK does not support multi-modal input",
image_count=len(images),
)
# TODO: When the SDK adds support for multi-modal content blocks, update this.
image_note = f"\n\n[Note: The user attached {len(images)} image(s), but the current SDK version does not support multi-modal input. Please ask the user to describe the image content instead.]"
print(
"Warning: Image attachments cannot be sent to the model in SDK mode. Sending text-only query.",
file=sys.stderr,
)
await client.query(full_prompt + image_note)
else:
# Send the query as plain text
await client.query(full_prompt)
# Stream the response
response_text = ""
@@ -280,13 +400,21 @@ Current question: {message}"""
import traceback
traceback.print_exc(file=sys.stderr)
run_simple(project_dir, message, history)
run_simple(project_dir, message, history, images)
def run_simple(project_dir: str, message: str, history: list) -> None:
def run_simple(
project_dir: str, message: str, history: list, images: list[dict] | None = None
) -> None:
"""Simple fallback mode without SDK - uses subprocess to call claude CLI."""
import subprocess
if images:
print(
"Warning: Image attachments are not supported in simple mode and will be skipped.",
file=sys.stderr,
)
system_prompt = build_system_prompt(project_dir)
# Build conversation context
@@ -355,6 +483,10 @@ def main():
default="medium",
help="Thinking level for extended reasoning (low, medium, high)",
)
parser.add_argument(
"--images-file",
help="Path to JSON manifest file listing image file paths and MIME types",
)
args = parser.parse_args()
# Validate and sanitize thinking level (handles legacy values like 'ultrathink')
@@ -398,9 +530,25 @@ def main():
debug_error("insights_runner", f"Failed to load history: {e}")
history = []
# Load images from manifest file if provided
images = None
if args.images_file:
debug("insights_runner", "Loading images from manifest", file=args.images_file)
images = load_images_from_manifest(args.images_file)
if images:
debug(
"insights_runner",
"Loaded images for multi-modal query",
image_count=len(images),
)
else:
debug("insights_runner", "No valid images loaded from manifest")
# Run the async SDK function
debug("insights_runner", "Running SDK query")
asyncio.run(run_with_sdk(project_dir, user_message, history, model, thinking_level))
asyncio.run(
run_with_sdk(project_dir, user_message, history, model, thinking_level, images)
)
debug_success("insights_runner", "Query completed")
+35 -9
View File
@@ -3,8 +3,10 @@ import type {
InsightsSession,
InsightsSessionSummary,
InsightsChatMessage,
InsightsModelConfig
InsightsModelConfig,
ImageAttachment
} from '../shared/types';
import { MAX_IMAGES_PER_TASK } from '../shared/constants';
import { InsightsConfig } from './insights/config';
import { InsightsPaths } from './insights/paths';
import { SessionStorage } from './insights/session-storage';
@@ -116,7 +118,8 @@ export class InsightsService extends EventEmitter {
projectId: string,
projectPath: string,
message: string,
modelConfig?: InsightsModelConfig
modelConfig?: InsightsModelConfig,
images?: ImageAttachment[]
): Promise<void> {
// Cancel any existing session
this.executor.cancelSession(projectId);
@@ -139,22 +142,44 @@ export class InsightsService extends EventEmitter {
session.title = this.storage.generateTitle(message);
}
// Add user message
// Guard: cap images to MAX_IMAGES_PER_TASK
if (images && images.length > MAX_IMAGES_PER_TASK) {
images = images.slice(0, MAX_IMAGES_PER_TASK);
}
// Add user message (store thumbnails only for persistence, strip full data)
const persistImages = images?.map(img => ({
...img,
data: undefined
}));
const userMessage: InsightsChatMessage = {
id: `msg-${Date.now()}`,
role: 'user',
content: message,
timestamp: new Date()
timestamp: new Date(),
images: persistImages && persistImages.length > 0 ? persistImages : undefined
};
session.messages.push(userMessage);
session.updatedAt = new Date();
this.sessionManager.saveSession(projectPath, session);
// Build conversation history for context
const conversationHistory = session.messages.map(m => ({
role: m.role,
content: m.content
}));
// Add notation when images are present so the AI has context
// For historical messages (all but the last), use past tense to avoid confusion
const conversationHistory = session.messages.map((m, index) => {
const imageCount = m.images?.length ?? 0;
const isLastMessage = index === session.messages.length - 1;
let imageNotation = '';
if (imageCount > 0 && m.role === 'user') {
imageNotation = isLastMessage
? `\n[User attached ${imageCount} image(s)]`
: `\n[User previously attached ${imageCount} image(s) - not visible in this context]`;
}
return {
role: m.role,
content: imageNotation ? m.content + imageNotation : m.content
};
});
// Use provided modelConfig or fall back to session's config
const configToUse = modelConfig || session.modelConfig;
@@ -166,7 +191,8 @@ export class InsightsService extends EventEmitter {
projectPath,
message,
conversationHistory,
configToUse
configToUse,
images
);
// Add assistant message to session
@@ -1,5 +1,7 @@
import { spawn, ChildProcess } from 'child_process';
import { existsSync, writeFileSync, unlinkSync } from 'fs';
import { existsSync, unlinkSync } from 'fs';
import { writeFile } from 'fs/promises';
import { randomBytes } from 'crypto';
import path from 'path';
import os from 'os';
import { EventEmitter } from 'events';
@@ -8,12 +10,23 @@ import type {
InsightsChatStatus,
InsightsStreamChunk,
InsightsToolUsage,
InsightsModelConfig
InsightsModelConfig,
ImageAttachment
} from '../../shared/types';
import { MODEL_ID_MAP } from '../../shared/constants';
import { MODEL_ID_MAP, MAX_IMAGES_PER_TASK, MAX_IMAGE_SIZE } from '../../shared/constants';
import { InsightsConfig } from './config';
import { detectRateLimit, createSDKRateLimitInfo } from '../rate-limit-detector';
// Safe extension map for image MIME types — prevents path traversal via crafted mimeType
// SVG excluded: contains active script content and is unsupported by Claude Vision API
const SAFE_EXT_MAP: Record<string, string> = {
'image/png': 'png',
'image/jpeg': 'jpg',
'image/jpg': 'jpg',
'image/gif': 'gif',
'image/webp': 'webp'
};
/**
* Message processor result
*/
@@ -63,7 +76,8 @@ export class InsightsExecutor extends EventEmitter {
projectPath: string,
message: string,
conversationHistory: Array<{ role: string; content: string }>,
modelConfig?: InsightsModelConfig
modelConfig?: InsightsModelConfig,
images?: ImageAttachment[]
): Promise<ProcessorResult> {
// Cancel any existing session
this.cancelSession(projectId);
@@ -90,18 +104,80 @@ export class InsightsExecutor extends EventEmitter {
// Write conversation history to temp file to avoid Windows command-line length limit
const historyFile = path.join(
os.tmpdir(),
`insights-history-${projectId}-${Date.now()}.json`
`insights-history-${projectId}-${Date.now()}-${randomBytes(8).toString('hex')}.json`
);
let historyFileCreated = false;
try {
writeFileSync(historyFile, JSON.stringify(conversationHistory), 'utf-8');
await writeFile(historyFile, JSON.stringify(conversationHistory), { encoding: 'utf-8', mode: 0o600 });
historyFileCreated = true;
} catch (err) {
console.error('[Insights] Failed to write history file:', err);
throw new Error('Failed to write conversation history to temp file');
}
// Write image files and manifest if images are provided
const imagesTempFiles: string[] = [];
let imagesManifestFile: string | undefined;
// Defense-in-depth: cap image count and filter oversized images in the executor
if (images && images.length > MAX_IMAGES_PER_TASK) {
images = images.slice(0, MAX_IMAGES_PER_TASK);
}
if (images) {
images = images.filter(img => !img.data || Buffer.byteLength(img.data, 'base64') <= MAX_IMAGE_SIZE);
}
if (images && images.length > 0) {
try {
const manifest: Array<{ path: string; mimeType: string }> = [];
const timestamp = Date.now();
for (let i = 0; i < images.length; i++) {
const image = images[i];
if (!image.data) continue;
// Validate mimeType against allowlist (defense-in-depth for main process)
const ext = SAFE_EXT_MAP[image.mimeType];
if (!ext) {
console.warn(`[Insights] Skipping image with invalid mimeType: ${image.mimeType}`);
continue;
}
const imagePath = path.join(
os.tmpdir(),
`insights-image-${projectId}-${timestamp}-${i}-${randomBytes(8).toString('hex')}.${ext}`
);
await writeFile(imagePath, Buffer.from(image.data, 'base64'), { mode: 0o600 });
imagesTempFiles.push(imagePath);
manifest.push({ path: imagePath, mimeType: image.mimeType });
}
// Only write manifest file if we actually wrote any images
if (manifest.length > 0) {
imagesManifestFile = path.join(
os.tmpdir(),
`insights-images-manifest-${projectId}-${timestamp}-${randomBytes(8).toString('hex')}.json`
);
imagesTempFiles.push(imagesManifestFile); // Push before writeFile for cleanup on failure
await writeFile(imagesManifestFile, JSON.stringify(manifest), { encoding: 'utf-8', mode: 0o600 });
}
} catch (err) {
// Clean up any already-written image files
for (const tmpFile of imagesTempFiles) {
try {
if (existsSync(tmpFile)) unlinkSync(tmpFile);
} catch { /* ignore cleanup errors */ }
}
// Also clean up the history file (cleanupTempFiles isn't defined yet at this point)
if (existsSync(historyFile)) {
try { unlinkSync(historyFile); } catch { /* ignore */ }
}
console.error('[Insights] Failed to write image files:', err);
throw new Error('Failed to write image files to temp directory');
}
}
// Build command arguments
const args = [
runnerPath,
@@ -110,6 +186,11 @@ export class InsightsExecutor extends EventEmitter {
'--history-file', historyFile
];
// Add images manifest file if images were provided
if (imagesManifestFile) {
args.push('--images-file', imagesManifestFile);
}
// Add model config if provided
if (modelConfig) {
const modelId = MODEL_ID_MAP[modelConfig.model] || MODEL_ID_MAP['sonnet'];
@@ -125,6 +206,27 @@ export class InsightsExecutor extends EventEmitter {
this.activeSessions.set(projectId, proc);
// Shared cleanup for temp files used across close/error handlers
let cleanedUp = false;
const cleanupTempFiles = () => {
if (cleanedUp) return;
cleanedUp = true;
if (historyFileCreated && existsSync(historyFile)) {
try {
unlinkSync(historyFile);
} catch (cleanupErr) {
console.error('[Insights] Failed to cleanup history file:', cleanupErr);
}
}
for (const tmpFile of imagesTempFiles) {
try {
if (existsSync(tmpFile)) unlinkSync(tmpFile);
} catch (cleanupErr) {
console.error('[Insights] Failed to cleanup image temp file:', cleanupErr);
}
}
};
return new Promise((resolve, reject) => {
let fullResponse = '';
const suggestedTasks: InsightsChatMessage['suggestedTasks'] = [];
@@ -170,15 +272,7 @@ export class InsightsExecutor extends EventEmitter {
proc.on('close', (code) => {
this.activeSessions.delete(projectId);
// Cleanup temp file
if (historyFileCreated && existsSync(historyFile)) {
try {
unlinkSync(historyFile);
} catch (cleanupErr) {
console.error('[Insights] Failed to cleanup history file:', cleanupErr);
}
}
cleanupTempFiles();
// Check for rate limit if process failed
if (code !== 0) {
@@ -217,15 +311,7 @@ export class InsightsExecutor extends EventEmitter {
proc.on('error', (err) => {
this.activeSessions.delete(projectId);
// Cleanup temp file
if (historyFileCreated && existsSync(historyFile)) {
try {
unlinkSync(historyFile);
} catch (cleanupErr) {
console.error('[Insights] Failed to cleanup history file:', cleanupErr);
}
}
cleanupTempFiles();
this.emit('error', projectId, err.message);
reject(err);
@@ -1,6 +1,6 @@
import { existsSync, readFileSync, writeFileSync, mkdirSync, readdirSync, unlinkSync } from 'fs';
import path from 'path';
import type { InsightsSession, InsightsSessionSummary } from '../../shared/types';
import type { InsightsSession, InsightsSessionSummary, ImageAttachment } from '../../shared/types';
import { InsightsPaths } from './paths';
/**
@@ -61,7 +61,8 @@ export class SessionStorage {
}
const sessionPath = this.paths.getSessionPath(projectPath, session.id);
writeFileSync(sessionPath, JSON.stringify(session, null, 2), 'utf-8');
const strippedSession = this.stripImageDataForPersistence(session);
writeFileSync(sessionPath, JSON.stringify(strippedSession, null, 2), 'utf-8');
}
/**
@@ -165,6 +166,23 @@ export class SessionStorage {
}
}
/**
* Strip full-resolution image data from a session for persistence.
* Keeps only thumbnail, id, filename, mimeType, and size to prevent bloated JSON files.
*/
private stripImageDataForPersistence(session: InsightsSession): InsightsSession {
return {
...session,
messages: session.messages.map(m => {
if (!m.images || m.images.length === 0) return m;
return {
...m,
images: m.images.map(({ data, path: _path, ...rest }: ImageAttachment) => rest)
};
})
};
}
/**
* Migrate old session format to new multi-session format
*/
@@ -16,6 +16,7 @@ import type {
InsightsSession,
InsightsSessionSummary,
InsightsModelConfig,
ImageAttachment,
Task,
TaskMetadata,
AppSettings,
@@ -81,7 +82,7 @@ export function registerInsightsHandlers(getMainWindow: () => BrowserWindow | nu
ipcMain.on(
IPC_CHANNELS.INSIGHTS_SEND_MESSAGE,
async (_, projectId: string, message: string, modelConfig?: InsightsModelConfig) => {
async (_, projectId: string, message: string, modelConfig?: InsightsModelConfig, images?: ImageAttachment[]) => {
const project = projectStore.getProject(projectId);
if (!project) {
safeSendToRenderer(
@@ -112,7 +113,7 @@ export function registerInsightsHandlers(getMainWindow: () => BrowserWindow | nu
// the handler returns. This fixes race conditions on Windows where
// environment setup wouldn't complete before process spawn.
try {
await insightsService.sendMessage(projectId, project.path, message, configWithSettings);
await insightsService.sendMessage(projectId, project.path, message, configWithSettings, images);
} catch (error) {
// Errors during sendMessage (executor errors) are already emitted via
// the 'error' event, but we catch here to prevent unhandled rejection
@@ -5,6 +5,7 @@ import type {
InsightsChatStatus,
InsightsStreamChunk,
InsightsModelConfig,
ImageAttachment,
Task,
TaskMetadata,
IPCResult
@@ -17,7 +18,7 @@ import { createIpcListener, invokeIpc, sendIpc, IpcListenerCleanup } from './ipc
export interface InsightsAPI {
// Operations
getInsightsSession: (projectId: string) => Promise<IPCResult<InsightsSession | null>>;
sendInsightsMessage: (projectId: string, message: string, modelConfig?: InsightsModelConfig) => void;
sendInsightsMessage: (projectId: string, message: string, modelConfig?: InsightsModelConfig, images?: ImageAttachment[]) => void;
clearInsightsSession: (projectId: string) => Promise<IPCResult>;
createTaskFromInsights: (
projectId: string,
@@ -55,8 +56,8 @@ export const createInsightsAPI = (): InsightsAPI => ({
getInsightsSession: (projectId: string): Promise<IPCResult<InsightsSession | null>> =>
invokeIpc(IPC_CHANNELS.INSIGHTS_GET_SESSION, projectId),
sendInsightsMessage: (projectId: string, message: string, modelConfig?: InsightsModelConfig): void =>
sendIpc(IPC_CHANNELS.INSIGHTS_SEND_MESSAGE, projectId, message, modelConfig),
sendInsightsMessage: (projectId: string, message: string, modelConfig?: InsightsModelConfig, images?: ImageAttachment[]): void =>
sendIpc(IPC_CHANNELS.INSIGHTS_SEND_MESSAGE, projectId, message, modelConfig, images),
clearInsightsSession: (projectId: string): Promise<IPCResult> =>
invokeIpc(IPC_CHANNELS.INSIGHTS_CLEAR_SESSION, projectId),
@@ -14,7 +14,9 @@ import {
FileText,
FolderSearch,
PanelLeftClose,
PanelLeft
PanelLeft,
Camera,
X
} from 'lucide-react';
import ReactMarkdown, { type Components } from 'react-markdown';
import remarkGfm from 'remark-gfm';
@@ -23,6 +25,7 @@ import { Textarea } from './ui/textarea';
import { ScrollArea } from './ui/scroll-area';
import { Card, CardContent } from './ui/card';
import { Badge } from './ui/badge';
import { ScreenshotCapture } from './ScreenshotCapture';
import { cn } from '../lib/utils';
import {
useInsightsStore,
@@ -36,15 +39,19 @@ import {
createTaskFromSuggestion,
setupInsightsListeners
} from '../stores/insights-store';
import { useImageUpload } from './task-form/useImageUpload';
import { createThumbnail, generateImageId } from './ImageUpload';
import { loadTasks } from '../stores/task-store';
import { ChatHistorySidebar } from './ChatHistorySidebar';
import { InsightsModelSelector } from './InsightsModelSelector';
import type { InsightsChatMessage, InsightsModelConfig, TaskMetadata } from '../../shared/types';
import type { InsightsChatMessage, InsightsModelConfig, TaskMetadata, ImageAttachment } from '../../shared/types';
import {
TASK_CATEGORY_LABELS,
TASK_CATEGORY_COLORS,
TASK_COMPLEXITY_LABELS,
TASK_COMPLEXITY_COLORS
TASK_COMPLEXITY_COLORS,
MAX_IMAGE_SIZE,
MAX_IMAGES_PER_TASK
} from '../../shared/constants';
// createSafeLink - factory function that creates a SafeLink component with i18n support
@@ -107,9 +114,38 @@ export function Insights({ projectId }: InsightsProps) {
const [showSidebar, setShowSidebar] = useState(true);
const [isUserAtBottom, setIsUserAtBottom] = useState(true);
const [viewportEl, setViewportEl] = useState<HTMLElement | null>(null);
const [screenshotOpen, setScreenshotOpen] = useState(false);
const [imageError, setImageError] = useState<string | null>(null);
const pendingImages = useInsightsStore((state) => state.pendingImages);
const setPendingImages = useInsightsStore((state) => state.setPendingImages);
const textareaRef = useRef<HTMLTextAreaElement | null>(null);
const isLoading = status.phase === 'thinking' || status.phase === 'streaming';
// Image upload hook
const {
isDragOver,
handlePaste,
handleDragOver,
handleDragLeave,
handleDrop,
removeImage,
canAddMore
} = useImageUpload({
images: pendingImages,
onImagesChange: setPendingImages,
disabled: isLoading,
onError: setImageError,
errorMessages: {
maxImagesReached: t('insights.images.maxImagesReached'),
invalidImageType: t('insights.images.invalidType'),
processPasteFailed: t('insights.images.processFailed'),
processDropFailed: t('insights.images.processFailed')
}
});
// Scroll threshold in pixels - user is considered "at bottom" if within this distance
const SCROLL_BOTTOM_THRESHOLD = 100;
@@ -158,6 +194,7 @@ export function Insights({ projectId }: InsightsProps) {
}, []);
// Reset task creation state when switching sessions
// biome-ignore lint/correctness/useExhaustiveDependencies: session?.id is intentionally used as a trigger
useEffect(() => {
setTaskCreated(new Set());
setCreatingTask(new Set());
@@ -165,13 +202,46 @@ export function Insights({ projectId }: InsightsProps) {
const handleSend = () => {
const message = inputValue.trim();
if (!message || status.phase === 'thinking' || status.phase === 'streaming') return;
const hasImages = pendingImages.length > 0;
if ((!message && !hasImages) || isLoading) return;
setInputValue('');
sendMessage(projectId, message);
sendMessage(projectId, message, session?.modelConfig, hasImages ? pendingImages : undefined);
setPendingImages([]);
setImageError(null);
setIsUserAtBottom(true); // Resume auto-scroll when user sends a message
};
const handleScreenshotCapture = useCallback(async (imageData: string) => {
// Check image count limit before processing
if (pendingImages.length >= MAX_IMAGES_PER_TASK) {
setImageError(t('insights.images.maxImagesReached'));
return;
}
// imageData is base64 PNG from ScreenshotCapture
const approximateSize = Math.ceil(imageData.length * 0.75); // approximate base64 size
// Validate size - match the validation used for regular image uploads
if (approximateSize > MAX_IMAGE_SIZE) {
setImageError(t('insights.images.screenshotTooLarge', { size: Math.round(approximateSize / 1024 / 1024), max: Math.round(MAX_IMAGE_SIZE / 1024 / 1024) }));
return;
}
const dataUrl = `data:image/png;base64,${imageData}`;
const thumbnail = await createThumbnail(dataUrl);
const newImage: ImageAttachment = {
id: generateImageId(),
filename: `screenshot-${Date.now()}.png`,
mimeType: 'image/png',
size: approximateSize,
data: imageData,
thumbnail
};
setPendingImages([...pendingImages, newImage]);
setImageError(null);
}, [pendingImages, setPendingImages, setImageError, t]);
const handleKeyDown = (e: React.KeyboardEvent) => {
if (e.key === 'Enter' && !e.shiftKey) {
e.preventDefault();
@@ -235,7 +305,6 @@ export function Insights({ projectId }: InsightsProps) {
}
};
const isLoading = status.phase === 'thinking' || status.phase === 'streaming';
const messages = session?.messages || [];
return (
@@ -402,32 +471,112 @@ export function Insights({ projectId }: InsightsProps) {
{/* Input */}
<div className="flex-shrink-0 border-t border-border p-4">
<div className="flex gap-2">
<Textarea
ref={textareaRef}
value={inputValue}
onChange={(e) => setInputValue(e.target.value)}
onKeyDown={handleKeyDown}
placeholder="Ask about your codebase..."
className="min-h-[80px] resize-none"
disabled={isLoading}
/>
<Button
onClick={handleSend}
disabled={!inputValue.trim() || isLoading}
className="self-end"
>
{isLoading ? (
<Loader2 className="h-4 w-4 animate-spin" />
) : (
<Send className="h-4 w-4" />
<div className="relative flex gap-2">
<div className="relative flex-1">
<Textarea
ref={textareaRef}
value={inputValue}
onChange={(e) => setInputValue(e.target.value)}
onKeyDown={handleKeyDown}
onPaste={handlePaste}
onDragOver={handleDragOver}
onDragLeave={handleDragLeave}
onDrop={handleDrop}
placeholder="Ask about your codebase..."
className={cn(
'min-h-[80px] resize-none',
isDragOver && 'border-primary ring-2 ring-primary/20'
)}
disabled={isLoading}
/>
{/* Drag-over overlay */}
{isDragOver && (
<div className="absolute inset-0 flex items-center justify-center rounded-md bg-primary/5 border-2 border-dashed border-primary pointer-events-none">
<span className="text-sm font-medium text-primary">
{t('insights.images.dragOver')}
</span>
</div>
)}
</Button>
</div>
<div className="flex flex-col gap-1 self-end">
<Button
variant="outline"
size="icon"
className="h-9 w-9"
onClick={() => setScreenshotOpen(true)}
disabled={isLoading || !canAddMore}
title={t('insights.images.screenshotButton')}
>
<Camera className="h-4 w-4" />
</Button>
<Button
onClick={handleSend}
disabled={(!inputValue.trim() && pendingImages.length === 0) || isLoading}
className="h-9 w-9"
size="icon"
>
{isLoading ? (
<Loader2 className="h-4 w-4 animate-spin" />
) : (
<Send className="h-4 w-4" />
)}
</Button>
</div>
</div>
{/* Image analysis warning */}
{pendingImages.length > 0 && (
<div className="mt-1 flex items-center gap-1.5 rounded-md bg-amber-500/10 px-2 py-1 text-xs text-amber-500">
<AlertCircle className="h-3 w-3 shrink-0" />
<span>{t('insights.images.analysisUnsupported')}</span>
</div>
)}
{/* Image error */}
{imageError && (
<p className="mt-1 text-xs text-destructive">{imageError}</p>
)}
{/* Image preview strip */}
{pendingImages.length > 0 && (
<div className="mt-2 flex flex-wrap items-center gap-2">
{pendingImages.map((image) => (
<div
key={image.id}
className="group relative h-16 w-16 rounded-md border border-border overflow-hidden"
>
<img
src={image.thumbnail || `data:${image.mimeType};base64,${image.data}`}
alt={image.filename}
className="h-full w-full object-cover"
/>
<button
type="button"
onClick={() => removeImage(image.id)}
className="absolute -right-1 -top-1 flex h-5 w-5 items-center justify-center rounded-full bg-destructive text-destructive-foreground opacity-0 transition-opacity group-hover:opacity-100"
title={t('insights.images.removeImage')}
>
<X className="h-3 w-3" />
</button>
</div>
))}
<span className="text-xs text-muted-foreground">
{t('insights.images.imageCount', { count: pendingImages.length })}
</span>
</div>
)}
<p className="mt-2 text-xs text-muted-foreground">
Press Enter to send, Shift+Enter for new line
{t('insights.images.pasteHint')} · Press Enter to send, Shift+Enter for new line
</p>
</div>
{/* Screenshot capture dialog */}
<ScreenshotCapture
open={screenshotOpen}
onOpenChange={setScreenshotOpen}
onCapture={handleScreenshotCapture}
/>
</div>
</div>
);
@@ -469,11 +618,32 @@ function MessageBubble({
<div className="text-sm font-medium text-foreground">
{isUser ? 'You' : 'Assistant'}
</div>
<div className="prose prose-sm dark:prose-invert max-w-none">
<ReactMarkdown remarkPlugins={[remarkGfm]} components={markdownComponents}>
{message.content}
</ReactMarkdown>
</div>
{message.content && (
<div className="prose prose-sm dark:prose-invert max-w-none">
<ReactMarkdown remarkPlugins={[remarkGfm]} components={markdownComponents}>
{message.content}
</ReactMarkdown>
</div>
)}
{/* Image attachments for user messages */}
{isUser && message.images && message.images.length > 0 && (
<div className="space-y-1.5">
<div className="flex flex-wrap gap-2">
{message.images
.filter(img => img.thumbnail || img.data)
.map((image) => (
<img
key={image.id}
src={image.thumbnail || `data:${image.mimeType};base64,${image.data}`}
alt={image.filename}
className="max-w-[200px] max-h-[200px] rounded-md border border-border object-contain"
/>
))}
</div>
<p className="text-xs text-muted-foreground italic">{t('insights.images.notAnalyzed')}</p>
</div>
)}
{/* Tool usage history for assistant messages */}
{!isUser && message.toolsUsed && message.toolsUsed.length > 0 && (
@@ -613,6 +783,7 @@ function ToolUsageHistory({ tools }: ToolUsageHistoryProps) {
return (
<div className="mt-2">
<button
type="button"
onClick={() => setExpanded(!expanded)}
className="flex items-center gap-2 text-xs text-muted-foreground hover:text-foreground transition-colors"
>
@@ -15,6 +15,7 @@ import {
import type { ImageAttachment } from '../../../shared/types';
import {
MAX_IMAGES_PER_TASK,
MAX_IMAGE_SIZE,
ALLOWED_IMAGE_TYPES_DISPLAY
} from '../../../shared/constants';
@@ -30,6 +31,7 @@ export interface FileReferenceData {
interface ImageUploadErrorMessages {
maxImagesReached?: string;
invalidImageType?: string;
imageTooLarge?: string;
processPasteFailed?: string;
processDropFailed?: string;
}
@@ -74,6 +76,7 @@ interface UseImageUploadReturn {
const DEFAULT_ERROR_MESSAGES: Required<ImageUploadErrorMessages> = {
maxImagesReached: `Maximum of ${MAX_IMAGES_PER_TASK} images allowed`,
invalidImageType: `Invalid image type. Allowed: ${ALLOWED_IMAGE_TYPES_DISPLAY}`,
imageTooLarge: `Image exceeds maximum size of ${Math.round(MAX_IMAGE_SIZE / 1024 / 1024)}MB`,
processPasteFailed: 'Failed to process pasted image',
processDropFailed: 'Failed to process dropped image'
};
@@ -148,6 +151,12 @@ export function useImageUpload({
continue;
}
// Validate file size
if (file.size > MAX_IMAGE_SIZE) {
onError?.(errors.imageTooLarge);
continue;
}
try {
const dataUrl = await blobToBase64(file);
const thumbnail = await createThumbnail(dataUrl);
@@ -8,7 +8,8 @@ import type {
InsightsToolUsage,
InsightsModelConfig,
TaskMetadata,
Task
Task,
ImageAttachment
} from '../../shared/types';
interface ToolUsage {
@@ -27,6 +28,7 @@ interface InsightsState {
currentTool: ToolUsage | null; // Currently executing tool
toolsUsed: InsightsToolUsage[]; // Tools used during current response
isLoadingSessions: boolean;
pendingImages: ImageAttachment[]; // Images pending attachment to next message
// Actions
setSession: (session: InsightsSession | null) => void;
@@ -44,6 +46,7 @@ interface InsightsState {
finalizeStreamingMessage: () => void;
clearSession: () => void;
setLoadingSessions: (loading: boolean) => void;
setPendingImages: (images: ImageAttachment[]) => void;
}
const initialStatus: InsightsChatStatus = {
@@ -62,6 +65,7 @@ export const useInsightsStore = create<InsightsState>((set, _get) => ({
currentTool: null,
toolsUsed: [],
isLoadingSessions: false,
pendingImages: [],
// Actions
setSession: (session) => set({ session }),
@@ -201,8 +205,11 @@ export const useInsightsStore = create<InsightsState>((set, _get) => ({
streamingContent: '',
streamingTasks: [],
currentTool: null,
toolsUsed: []
})
toolsUsed: [],
pendingImages: []
}),
setPendingImages: (images) => set({ pendingImages: images })
}));
// Helper functions
@@ -234,21 +241,27 @@ export async function loadInsightsSession(projectId: string): Promise<void> {
await loadInsightsSessions(projectId);
}
export function sendMessage(projectId: string, message: string, modelConfig?: InsightsModelConfig): void {
export function sendMessage(projectId: string, message: string, modelConfig?: InsightsModelConfig, images?: ImageAttachment[]): void {
const store = useInsightsStore.getState();
const session = store.session;
// Add user message to session
// Add user message to session (strip data to keep memory usage low)
const displayImages = images?.map(img => ({
...img,
data: undefined // Strip base64 data, keep thumbnails for display
}));
const userMessage: InsightsChatMessage = {
id: `msg-${Date.now()}`,
role: 'user',
content: message,
timestamp: new Date()
timestamp: new Date(),
...(displayImages && displayImages.length > 0 ? { images: displayImages } : {})
};
store.addMessage(userMessage);
// Clear pending and set status
store.setPendingMessage('');
store.setPendingImages([]);
store.clearStreamingContent();
store.clearToolsUsed(); // Clear tools from previous response
store.setStatus({
@@ -260,7 +273,7 @@ export function sendMessage(projectId: string, message: string, modelConfig?: In
const configToUse = modelConfig || session?.modelConfig;
// Send to main process
window.electronAPI.sendInsightsMessage(projectId, message, configToUse);
window.electronAPI.sendInsightsMessage(projectId, message, configToUse, images);
}
export async function clearSession(projectId: string): Promise<void> {
+3 -4
View File
@@ -233,15 +233,14 @@ export const ALLOWED_IMAGE_TYPES = [
'image/jpeg',
'image/jpg',
'image/gif',
'image/webp',
'image/svg+xml'
'image/webp'
] as const;
// Allowed image file extensions (for display)
export const ALLOWED_IMAGE_EXTENSIONS = ['.png', '.jpg', '.jpeg', '.gif', '.webp', '.svg'] as const;
export const ALLOWED_IMAGE_EXTENSIONS = ['.png', '.jpg', '.jpeg', '.gif', '.webp'] as const;
// Human-readable allowed types for error messages
export const ALLOWED_IMAGE_TYPES_DISPLAY = 'PNG, JPEG, GIF, WebP, SVG';
export const ALLOWED_IMAGE_TYPES_DISPLAY = 'PNG, JPEG, GIF, WebP';
// Attachments directory name within spec folder
export const ATTACHMENTS_DIR = 'attachments';
@@ -449,7 +449,22 @@
"suggestedTask": "Suggested Task",
"creating": "Creating...",
"taskCreated": "Task Created",
"createTask": "Create Task"
"createTask": "Create Task",
"images": {
"pasteHint": "Paste an image or screenshot",
"dropHint": "Drop image here",
"screenshotButton": "Attach screenshot",
"removeImage": "Remove image",
"imageCount": "{{count}} image attached",
"imageCount_plural": "{{count}} images attached",
"maxImagesReached": "Maximum number of images reached",
"invalidType": "Invalid file type. Please use PNG, JPEG, GIF, or WebP.",
"processFailed": "Failed to process image",
"dragOver": "Drop image to attach",
"analysisUnsupported": "Note: Image analysis is not yet supported. Images are stored for reference but cannot be analyzed by the model.",
"screenshotTooLarge": "Screenshot is too large ({{size}}MB). Maximum size is {{max}}MB. Consider capturing a smaller area.",
"notAnalyzed": "Images were stored for reference but not analyzed by the model."
}
},
"ideation": {
"converting": "Converting...",
@@ -449,7 +449,22 @@
"suggestedTask": "Tâche suggérée",
"creating": "Création...",
"taskCreated": "Tâche créée",
"createTask": "Créer une tâche"
"createTask": "Créer une tâche",
"images": {
"pasteHint": "Coller une image ou une capture d'écran",
"dropHint": "Déposer l'image ici",
"screenshotButton": "Joindre une capture d'écran",
"removeImage": "Supprimer l'image",
"imageCount": "{{count}} image jointe",
"imageCount_plural": "{{count}} images jointes",
"maxImagesReached": "Nombre maximum d'images atteint",
"invalidType": "Type de fichier invalide. Veuillez utiliser PNG, JPEG, GIF ou WebP.",
"processFailed": "Échec du traitement de l'image",
"dragOver": "Déposer l'image pour joindre",
"analysisUnsupported": "Note : L'analyse d'images n'est pas encore prise en charge. Les images sont stockées pour référence mais ne peuvent pas être analysées par le modèle.",
"screenshotTooLarge": "La capture d'écran est trop volumineuse ({{size}}Mo). La taille maximale est de {{max}}Mo. Essayez de capturer une zone plus petite.",
"notAnalyzed": "Les images ont été stockées pour référence mais n'ont pas été analysées par le modèle."
}
},
"ideation": {
"converting": "Conversion...",
+3 -1
View File
@@ -2,7 +2,7 @@
* Insights and ideation types
*/
import type { TaskMetadata } from './task';
import type { TaskMetadata, ImageAttachment } from './task';
// ============================================
// Ideation Types
@@ -186,6 +186,8 @@ export interface InsightsChatMessage {
description: string;
metadata?: TaskMetadata;
}>;
// Image attachments (screenshots, pasted images)
images?: ImageAttachment[];
// Tools used during this response (assistant messages only)
toolsUsed?: InsightsToolUsage[];
}
+1 -1
View File
@@ -777,7 +777,7 @@ export interface ElectronAPI {
// Insights operations
getInsightsSession: (projectId: string) => Promise<IPCResult<InsightsSession | null>>;
sendInsightsMessage: (projectId: string, message: string, modelConfig?: InsightsModelConfig) => void;
sendInsightsMessage: (projectId: string, message: string, modelConfig?: InsightsModelConfig, images?: ImageAttachment[]) => void;
clearInsightsSession: (projectId: string) => Promise<IPCResult>;
createTaskFromInsights: (
projectId: string,