Refactor Content Processing and Enhance Search Functionality

- Updated the BuildContent method to utilize content parsing with improved error handling and context injection, enhancing the processing of user input.
- Refactored the loadMap function to support multiple search configuration types, improving flexibility in handling search settings.
- Enhanced the shouldAutoSearch method to include a check for search disabling via context metadata, allowing for more granular control over search behavior.
- Updated the getMergedSearchUses method to prioritize options.Uses, ensuring that search configurations can be dynamically adjusted based on provided options.
- Removed deprecated audio and excel handling code, streamlining the content processing package and improving maintainability.
This commit is contained in:
Max 2025-12-29 14:26:48 +08:00
parent 5f67ab2eef
commit 983dbd3cee
28 changed files with 1187 additions and 3103 deletions

View file

@ -4,6 +4,7 @@ import (
"fmt"
"github.com/yaoapp/yao/agent/content"
contentTypes "github.com/yaoapp/yao/agent/content/types"
"github.com/yaoapp/yao/agent/context"
)
@ -23,19 +24,24 @@ func (ast *Assistant) BuildContent(ctx *context.Context, messages []context.Mess
if err != nil {
return nil, fmt.Errorf("failed to get connector: %w", err)
}
_ = connector // unused but needed for GetConnector call
// Get Uses configuration from options (already merged in BuildRequest)
uses := options.Uses
// Get ForceUses configuration from options
forceUses := options.ForceUses
// Process content through Vision function
processedMessages, err := content.Vision(ctx, capabilities, messages, uses, forceUses)
if err != nil {
return nil, fmt.Errorf("failed to process content: %w", err)
// Build parse options
parseOptions := &contentTypes.Options{
Capabilities: capabilities,
CompletionOptions: options,
Connector: connector,
StreamOptions: options.StreamOptions,
}
return processedMessages, nil
contentMessages, referenceContext, err := content.ParseUserInput(ctx, messages, parseOptions)
if err != nil {
return nil, fmt.Errorf("failed to parse content: %w", err)
}
// Inject reference context into messages
if referenceContext != nil {
contentMessages = ast.injectSearchContext(contentMessages, referenceContext)
}
return contentMessages, nil
}

View file

@ -609,7 +609,15 @@ func loadMap(data map[string]interface{}) (*Assistant, error) {
// Search configuration (from package.yao search block)
// This contains search options like web.max_results, kb.threshold, citation.format, etc.
// Merge hierarchy: global config < assistant config
if v, ok := data["search"].(map[string]interface{}); ok {
switch v := data["search"].(type) {
case *searchTypes.Config:
assistant.Search = v
case searchTypes.Config:
assistant.Search = &v
case map[string]interface{}:
var assistantSearch searchTypes.Config
raw, err := jsoniter.Marshal(v)
if err != nil {
@ -621,8 +629,8 @@ func loadMap(data map[string]interface{}) (*Assistant, error) {
}
// Merge with global search config
assistant.Search = mergeSearchConfig(globalSearchConfig, &assistantSearch)
} else if globalSearchConfig != nil {
// No assistant-specific config, use global
default:
assistant.Search = globalSearchConfig
}

View file

@ -31,6 +31,15 @@ func (ast *Assistant) shouldAutoSearch(ctx *context.Context, messages []context.
return nil
}
// Check if search is skipped via ctx.Metadata["__disable_search"]
if ctx != nil && ctx.Metadata != nil {
disableSearch := getBool(ctx.Metadata, "__disable_search")
if disableSearch {
ctx.Logger.Debug("Auto search skipped by ctx.Metadata['__disable_search']")
return nil
}
}
// Check createResponse.Search field (highest priority from Create hook)
// Supports: bool | SearchIntent | nil
if createResponse != nil && createResponse.Search != nil {
@ -46,7 +55,7 @@ func (ast *Assistant) shouldAutoSearch(ctx *context.Context, messages []context.
}
// Get merged uses configuration
uses := ast.getMergedSearchUses(createResponse)
uses := ast.getMergedSearchUses(createResponse, opts)
// Check if search is explicitly disabled
if uses != nil && uses.Search == "disabled" {
@ -357,8 +366,9 @@ func (ast *Assistant) sendIntentDone(ctx *context.Context, loadingID string, nee
}
// getMergedSearchUses returns the merged uses configuration for search
// Priority: createResponse > assistant
func (ast *Assistant) getMergedSearchUses(createResponse *context.HookCreateResponse) *context.Uses {
// Priority: createResponse > options.Uses > assistant
func (ast *Assistant) getMergedSearchUses(createResponse *context.HookCreateResponse, opts ...*context.Options) *context.Uses {
// Start with assistant uses
var uses *context.Uses
if ast.Uses != nil {
@ -371,6 +381,31 @@ func (ast *Assistant) getMergedSearchUses(createResponse *context.HookCreateResp
}
}
// Override with options.Uses if provided (highest priority)
if len(opts) > 0 && opts[0] != nil && opts[0].Uses != nil {
if uses == nil {
uses = &context.Uses{}
}
if opts[0].Uses.Search != "" {
uses.Search = opts[0].Uses.Search
}
if opts[0].Uses.Web != "" {
uses.Web = opts[0].Uses.Web
}
if opts[0].Uses.Keyword != "" {
uses.Keyword = opts[0].Uses.Keyword
}
if opts[0].Uses.QueryDSL != "" {
uses.QueryDSL = opts[0].Uses.QueryDSL
}
if opts[0].Uses.Rerank != "" {
uses.Rerank = opts[0].Uses.Rerank
}
}
// Override with createResponse.Uses if provided (highest priority)
if createResponse != nil && createResponse.Uses != nil {
if uses == nil {
@ -405,7 +440,7 @@ func (ast *Assistant) executeAutoSearch(ctx *context.Context, messages []context
defer ctx.Logger.PhaseComplete("Search")
// Get merged uses configuration
uses := ast.getMergedSearchUses(createResponse)
uses := ast.getMergedSearchUses(createResponse, opts...)
// Convert to search.Uses
searchUses := &search.Uses{}

View file

@ -43,6 +43,25 @@ func getTimestamp(v interface{}) (int64, error) {
return 0, fmt.Errorf("invalid timestamp type %T", v)
}
// getBool gets bool from data map[string]interface{}, key string
func getBool(data map[string]interface{}, key string) bool {
switch v := data[key].(type) {
case bool:
return v
case int64:
return v != 0
case int:
return v != 0
case float64:
return v != 0
case string:
return v == "true" || v == "1" || v == "enabled" || v == "yes" || v == "on"
case nil:
return false
}
return false
}
// stringHash returns the sha256 hash of the string
func stringHash(v string) string {
h := sha256.New()

View file

@ -1,326 +0,0 @@
# Content Processing Package
This package handles content transformation for multimodal messages in agent conversations. It is called **BEFORE** sending messages to the LLM and converts extended content types into standard LLM-compatible formats.
## ⚠️ Critical Design Principle
**Input**: Messages with extended content types (`file`, `data`, etc.)
**Output**: Messages with ONLY standard LLM-compatible types (`text`, `image_url`, `input_audio`)
The LLM should NEVER receive `type="file"` or `type="data"` content parts. These MUST be converted to `text` (or `image_url` for images if model supports vision).
## Architecture
```
Vision (main entry)
Initialize processedFiles cache (map[fileID]text)
processMessage (for each message)
processContentPart (for each content part)
Is uploader wrapper?
├── Yes → Check cache
│ ├── In cache? → Use cached text ✅
│ └── Not in cache → Try GetText(fileID) preview
│ ├── Has preview? → Use preview + cache ✅
│ └── No preview → Proceed to full processing ↓
└── No (HTTP/other) → Proceed to full processing ↓
├── Fetch content (if needed)
│ ├── HTTP URL
│ └── Uploader Wrapper (__uploader://fileid)
├── Determine Processing Strategy
│ ├── Model supports? → Format for model
│ └── Model doesn't support? → Use agent/MCP
ProcessorRegistry
├── ImageProcessor
├── AudioProcessor
├── PDFProcessor
├── WordProcessor
├── ExcelProcessor
└── TextProcessor
Cache result (if uploader wrapper)
```
## Content Type Transformation
### Input → Output Mapping
| Input Type | Model Supports? | Output Type | Processing |
|------------|-----------------|-------------|------------|
| `text` | - | `text` | Pass through |
| `image_url` | ✅ Yes | `image_url` | Convert format if needed (base64/URL) |
| `image_url` | ❌ No | `text` | Use vision agent/MCP to describe |
| `input_audio` | ✅ Yes | `input_audio` | Keep as audio |
| `input_audio` | ❌ No | `text` | Transcribe using audio agent/MCP |
| `file` (image) | ✅ Yes | `image_url` | Same as image_url processing |
| `file` (image) | ❌ No | `text` | Use vision tool to describe |
| `file` (document) | - | `text` | Extract text from PDF/Word/Excel/etc |
| `data` | - | `text` | Fetch and format data sources |
### 1. Images and Audio
**If model supports (vision/audio capability):**
- Keep as multimodal content:
- `image_url`: Convert to appropriate format (OpenAI URL vs Claude base64)
- `input_audio`: Convert to base64 format
**If model doesn't support:**
- Convert to text:
- Use agent/MCP specified in `uses.Vision` or `uses.Audio`
- Extract text description or transcription
- Return as `type="text"` content
**HTTP URLs:**
- Fetch content first
- Then process the same way as above
### 2. Files (type="file")
**Critical**: All `type="file"` content MUST be converted to `text` or `image_url` (if image and model supports).
**Processing Steps:**
1. **Fetch file content**:
- Uploader wrapper: `__uploader://fileid` → Parse and fetch from attachment manager
- HTTP URL: Download from URL
2. **Detect file type** from content-type and magic bytes
3. **Process based on file type**:
| File Type | Output Type | Processing Method |
| ------------ | ----------- | -------------------------------------------------------------------------------------------------------- |
| **Image** | `image_url` or `text` | If model supports vision → `image_url`<br>If not → use vision tool → `text` |
| **PDF** | `text` | If `uses.Vision` supports PDF → use vision tool<br>Otherwise → extract text directly |
| **Word** | `text` | Extract text using Word document parser |
| **Excel** | `text` | Extract and format as readable table/CSV |
| **PPT** | `text` | Extract text and slide content |
| **CSV** | `text` | Format as readable table |
| **Text** | `text` | Read directly (with encoding detection) |
| **JSON/XML** | `text` | Pretty print for readability |
### 3. Data Sources (type="data")
**Critical**: All `type="data"` content MUST be converted to `text`.
**Processing Steps:**
1. **Parse DataContent.Sources** array
2. **Fetch data** from each source:
- `model`: Query data model
- `kb_collection`: Search knowledge base collection
- `kb_document`: Get document content
- `table`: Query database table
- `api`: Call API endpoint
- `mcp_resource`: Fetch MCP resource
3. **Format as readable text**:
- Tables: Format as markdown tables or CSV
- Documents: Include title and content
- JSON: Pretty print
4. **Return as** `type="text"` content
## Components
### Core Files
- **content.go** - Main entry point (`Vision` function)
- **types.go** - Type definitions and constants
- **interfaces.go** - Interface definitions
### Fetching
- **fetch.go** - Fetch content from HTTP or uploader
### Processors
- **processor.go** - Processor registry and routing
- **image.go** - Image processing
- **audio.go** - Audio processing
- **pdf.go** - PDF document processing
- **word.go** - Word document processing
- **excel.go** - Excel spreadsheet processing
- **text.go** - Plain text and CSV processing
## Frontend Message Format
The frontend (InputArea) sends messages in the following format:
### Image Attachments
```json
{
"type": "image_url",
"image_url": {
"url": "__yao.attachment://file_id",
"detail": "auto"
}
}
```
### File Attachments
```json
{
"type": "file",
"file": {
"url": "__yao.attachment://file_id",
"filename": "document.pdf"
}
}
```
The `url` field contains an uploader wrapper in the format `__uploader://fileid`.
## Data Structures
### ContentInfo
Holds information about content to be processed:
```go
type ContentInfo struct {
Source ContentSource // http, uploader, base64, local
FileType FileType // image, audio, pdf, word, excel, etc.
ContentType string // MIME type
URL string // Original URL or file ID
Data []byte // File data
// For uploader wrapper
UploaderName string
FileID string
}
```
### ProcessedContent
Result of content processing:
```go
type ProcessedContent struct {
Text string // Extracted text
ContentPart *context.ContentPart // For model input
Metadata map[string]interface{}
Error error
}
```
## Usage Example
```go
import (
"github.com/yaoapp/yao/agent/content"
"github.com/yaoapp/yao/agent/context"
)
// Process messages before sending to LLM
processedMessages, err := content.Vision(
ctx,
capabilities, // Model capabilities
messages, // Original messages
uses, // Tool specifications (vision, audio, etc.)
)
```
## Performance Optimization
### File Processing Cache
**Problem**: Same file (uploader wrapper) might appear in multiple messages or be referenced multiple times.
**Solution**: Three-level caching strategy:
1. **In-memory cache** (`processedFiles` map):
- Caches processed text for the duration of the Vision() call
- Key: file ID from uploader wrapper
- Value: extracted text content
2. **Attachment preview** (attachment.GetText with preview):
- Tries to get preview (first 2000 chars) from attachment manager
- If file was previously processed and saved, preview is available immediately
- Much faster than full file processing
3. **Full processing** (only if needed):
- Falls back to complete file processing if no cache/preview available
- Result is cached in memory and optionally saved to attachment manager
### Cache Flow
```go
// For uploader://file_id
1. Check processedFiles[file_id]
└── Found? → Return cached text ⚡ (fastest)
2. Not in cache → Call attachment.GetText(file_id, false) // preview only
└── Has preview? → Cache and return ⚡ (fast)
3. No preview → Process file fully 🔄 (slower)
└── Cache result in processedFiles
└── Optional: Save to attachment using SaveText for future use
```
### Benefits
- **Avoid duplicate processing**: Same file processed only once per Vision() call
- **Fast preview access**: Leverage pre-processed content from attachment manager
- **Reduced latency**: Especially important for large documents (PDFs, Word, Excel)
- **Resource efficient**: Less CPU/memory usage for repeated file references
## Implementation Status
### ✅ Completed
- [x] Package structure
- [x] Type definitions
- [x] Interface definitions
- [x] Skeleton functions with TODO comments
- [x] File processing cache infrastructure
- [x] Cache helper functions (tryGetCachedText, cacheProcessedText)
### 🚧 To Implement
- [ ] tryGetCachedText implementation (attachment.GetText integration)
- [ ] cacheProcessedText implementation (attachment.SaveText integration)
- [ ] HTTP fetching logic
- [ ] Uploader wrapper parsing and fetching
- [ ] Image processing (base64, vision API)
- [ ] Audio processing (transcription)
- [ ] PDF text extraction
- [ ] Word document parsing
- [ ] Excel spreadsheet parsing
- [ ] Text/CSV formatting
- [ ] Content part processing logic
- [ ] Model capability detection
- [ ] Agent/MCP tool invocation
## Configuration
Content processing behavior is controlled by:
1. **Model Capabilities** (`openai.Capabilities`)
- Determines if model can handle images/audio directly
- Specifies vision format (OpenAI vs Claude)
2. **Uses** (`context.Uses`)
```go
type Uses struct {
Vision string // "agent" or "mcp:server_id"
Audio string // "agent" or "mcp:server_id"
Search string
Fetch string
}
```
## Error Handling
- Errors during processing are logged but don't stop the entire pipeline
- Original content is kept if processing fails
- Graceful degradation: if advanced processing fails, fall back to simpler methods

View file

@ -1,61 +0,0 @@
package content
import (
"fmt"
"strings"
"github.com/yaoapp/gou/connector/openai"
agentContext "github.com/yaoapp/yao/agent/context"
)
// AudioHandler handles audio content
type AudioHandler struct{}
// CanHandle checks if this handler can handle the content type
func (h *AudioHandler) CanHandle(contentType string, fileType FileType) bool {
return fileType == FileTypeAudio || strings.HasPrefix(contentType, "audio/")
}
// Handle processes audio content
// Logic similar to image:
// 1. If model supports audio input -> convert to base64 format
// 2. If model doesn't support audio -> use agent/MCP specified in uses.Audio
func (h *AudioHandler) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses, forceUses bool) (*Result, error) {
// TODO: Implement audio handling
// 1. Check model audio capabilities
// 2. If supported:
// - Encode audio as base64 with proper format
// 3. If not supported:
// - Call audio agent/MCP to transcribe audio to text
// 4. Return Result with text or ContentPart
return nil, fmt.Errorf("not implemented")
}
// handleWithAudioModel processes audio using model's audio capability
func (h *AudioHandler) handleWithAudioModel(ctx *agentContext.Context, info *Info) (*Result, error) {
// TODO: Implement audio model processing
// Format audio according to model's audio input format
return nil, fmt.Errorf("not implemented")
}
// handleWithAudioAgent processes audio using audio agent or MCP
func (h *AudioHandler) handleWithAudioAgent(ctx *agentContext.Context, info *Info, audioTool string) (string, error) {
// TODO: Implement audio agent/MCP processing
// 1. Parse audioTool (format: "agent" or "mcp:server_id")
// 2. Call appropriate tool to transcribe audio
// 3. Return transcribed text
return "", fmt.Errorf("not implemented")
}
// encodeAudioBase64 encodes audio data to base64 with proper format
func encodeAudioBase64(data []byte, contentType string) string {
// TODO: Implement audio base64 encoding
return ""
}
// detectAudioFormat detects audio format from content type or data
func detectAudioFormat(contentType string, data []byte) string {
// TODO: Implement audio format detection
// Return format like "wav", "mp3", "flac", etc.
return ""
}

View file

@ -2,528 +2,118 @@ package content
import (
"fmt"
"strings"
"github.com/yaoapp/gou/connector/openai"
"github.com/yaoapp/kun/log"
"github.com/yaoapp/yao/agent/content/image"
"github.com/yaoapp/yao/agent/content/types"
agentContext "github.com/yaoapp/yao/agent/context"
"github.com/yaoapp/yao/attachment"
searchTypes "github.com/yaoapp/yao/agent/search/types"
)
// Vision transforms extended content types to LLM-compatible formats
// This is the main entry point for content preprocessing before sending to LLM
//
// IMPORTANT: This function is called BEFORE sending messages to LLM (in agent.executeLLMStream)
// It must convert all extended content types to standard LLM-compatible types.
//
// Input Content Types (Extended):
// - type="text" -> Pass through (already standard)
// - type="image_url" -> Process based on model capability (may need base64 conversion or vision tool)
// - type="input_audio" -> Process based on model capability (may need transcription)
// - type="file" -> Convert to text or image_url (MUST be converted)
// - type="data" -> Convert to text (MUST be converted)
//
// Output Content Types (LLM-compatible only):
// - type="text" -> Text content
// - type="image_url" -> Image (only if model supports vision)
// - type="input_audio" -> Audio (only if model supports audio)
//
// Processing Logic:
// 1. For images (image_url):
// - If model supports vision -> keep as image_url (may convert URL to base64)
// - If model doesn't support -> use vision agent/MCP to extract text -> convert to type="text"
//
// 2. For audio (input_audio):
// - If model supports audio -> keep as input_audio
// - If model doesn't support -> use audio agent/MCP to transcribe -> convert to type="text"
//
// 3. For files (type="file"):
// - Parse uploader wrapper (__uploader://fileid) or fetch HTTP URL
// - Detect file type (PDF, Word, Excel, Image, etc.)
// - Process based on file type:
// - Images: same as image processing above
// - PDF: use vision tool if available, otherwise extract text -> type="text"
// - Word/Excel/PPT/CSV: extract text -> type="text"
// - MUST convert to type="text" or type="image_url" (if image and model supports)
//
// 4. For data (type="data"):
// - Fetch data from sources (models, KB, MCP resources, etc.)
// - Format as readable text
// - MUST convert to type="text"
//
// Return: Messages with only standard LLM-compatible content types (text, image_url, input_audio)
func Vision(ctx *agentContext.Context, capabilities *openai.Capabilities, messages []agentContext.Message, uses *agentContext.Uses, forceUses ...bool) ([]agentContext.Message, error) {
// Determine if we should force using Uses tools even when model has native capabilities
shouldForceUses := false
if len(forceUses) > 0 {
shouldForceUses = forceUses[0]
}
// Initialize handlers and fetcher
registry := NewRegistry()
fetcher := NewFetcher()
// Cache for processed files (uploader wrapper -> extracted text)
// Ensures each file is only processed once
processedFiles := make(map[string]string)
// Process each message
processedMessages := make([]agentContext.Message, 0, len(messages))
for _, msg := range messages {
processedMsg, err := processMessage(ctx, &msg, capabilities, uses, shouldForceUses, registry, fetcher, processedFiles)
if err != nil {
// Log error but continue processing other messages
// TODO: Add proper logging
fmt.Printf("Warning: failed to process message: %v\n", err)
processedMessages = append(processedMessages, msg) // Keep original on error
// ParseUserInput ParseUserInput
func ParseUserInput(ctx *agentContext.Context, messages []agentContext.Message, options *types.Options) ([]agentContext.Message, *searchTypes.ReferenceContext, error) {
var referenceContext *searchTypes.ReferenceContext = nil
var parsedMessages []agentContext.Message = make([]agentContext.Message, 0)
for _, message := range messages {
// Only process user messages (current or from history)
if message.Role != agentContext.RoleUser {
parsedMessages = append(parsedMessages, message)
continue
}
processedMessages = append(processedMessages, processedMsg)
}
return processedMessages, nil
}
// processMessage processes a single message and its content parts
func processMessage(
ctx *agentContext.Context,
msg *agentContext.Message,
capabilities *openai.Capabilities,
uses *agentContext.Uses,
forceUses bool,
registry *Registry,
fetcher Fetcher,
processedFiles map[string]string,
) (agentContext.Message, error) {
// If content is simple string, no processing needed
if _, ok := msg.GetContentAsString(); ok {
return *msg, nil
}
// Get content parts - try typed first, then convert from interface{}
parts, ok := msg.GetContentAsParts()
if !ok {
// Try to convert from []interface{} (common when loaded from history/JSON)
parts, ok = convertToContentParts(msg.Content)
if !ok {
return *msg, nil
}
}
// Note: File information will be collected and stored in Space by CallAgentWithFileInfo
// when calling vision agents, using the agent ID as namespace prefix
// Process each content part
processedParts := make([]agentContext.ContentPart, 0, len(parts))
for _, part := range parts {
processedPart, err := processContentPart(ctx, &part, capabilities, uses, forceUses, registry, fetcher, processedFiles)
// Parse user input message (Ignore errors)
parsedMessage, refs, err := parseUserInputMessage(ctx, message, options)
if err != nil {
// Log error and handle gracefully
fmt.Printf("Warning: failed to process content part: %v\n", err)
parsedMessages = append(parsedMessages, message)
log.Error("Failed to parse user input message: %v, %v", message.Content, err)
continue
}
parsedMessages = append(parsedMessages, parsedMessage)
// For image_url that failed to process, convert to text description
// This prevents sending unsupported multimodal content to non-vision models
if part.Type == agentContext.ContentImageURL {
processedParts = append(processedParts, agentContext.ContentPart{
Type: agentContext.ContentText,
Text: fmt.Sprintf("[Image processing failed: %s]", part.ImageURL.URL),
})
} else {
// For other types, keep original
processedParts = append(processedParts, part)
// Add reference to reference context
if refs != nil {
if referenceContext == nil {
referenceContext = &searchTypes.ReferenceContext{}
}
continue
}
// If handling returned text, convert to text part
if processedPart.Text != "" {
processedParts = append(processedParts, agentContext.ContentPart{
Type: agentContext.ContentText,
Text: processedPart.Text,
})
} else if processedPart.ContentPart != nil {
// Use the processed content part (e.g., base64 image)
processedParts = append(processedParts, *processedPart.ContentPart)
} else {
// Keep original if no handling result
processedParts = append(processedParts, part)
referenceContext.References = append(referenceContext.References, refs...)
}
}
// Return new message with processed content
return agentContext.Message{
Role: msg.Role,
Content: processedParts,
Name: msg.Name,
ToolCallID: msg.ToolCallID,
ToolCalls: msg.ToolCalls,
Refusal: msg.Refusal,
}, nil
return parsedMessages, referenceContext, nil
}
// processContentPart processes a single content part
// IMPORTANT: Must convert extended types (file, data) to standard types (text, image_url, input_audio)
func processContentPart(
ctx *agentContext.Context,
part *agentContext.ContentPart,
capabilities *openai.Capabilities,
uses *agentContext.Uses,
forceUses bool,
registry *Registry,
fetcher Fetcher,
processedFiles map[string]string,
) (*Result, error) {
// 1. Handle standard types - pass through
switch part.Type {
// parseUserInputMessage parse a user input message
func parseUserInputMessage(ctx *agentContext.Context, message agentContext.Message, options *types.Options) (agentContext.Message, []*searchTypes.Reference, error) {
// Context content type
switch content := message.Content.(type) {
case string:
return message, nil, nil
case []agentContext.ContentPart:
return parseContentParts(ctx, message, content, options)
case []interface{}:
// Handle content loaded from history/JSON ([]interface{} instead of []ContentPart)
parts, ok := convertToContentParts(content)
if !ok {
return message, nil, nil
}
return parseContentParts(ctx, message, parts, options)
}
return message, nil, fmt.Errorf("unsupported content type: %T", message.Content)
}
// parseContentParts parses content parts and returns the parsed message
func parseContentParts(ctx *agentContext.Context, message agentContext.Message, content []agentContext.ContentPart, options *types.Options) (agentContext.Message, []*searchTypes.Reference, error) {
allRefs := []*searchTypes.Reference{}
parts := make([]agentContext.ContentPart, 0, len(content))
for _, part := range content {
parsedPart, refs, err := parseContentPart(ctx, part, options)
if err != nil {
parts = append(parts, part)
continue
}
parts = append(parts, parsedPart)
if refs != nil {
allRefs = append(allRefs, refs...)
}
}
parsedMessage := message
parsedMessage.Content = parts
return parsedMessage, allRefs, nil
}
// parseContentPart parse a content part
func parseContentPart(ctx *agentContext.Context, content agentContext.ContentPart, options *types.Options) (agentContext.ContentPart, []*searchTypes.Reference, error) {
switch content.Type {
case agentContext.ContentText:
// Text is already standard, pass through
return &Result{
ContentPart: part,
}, nil
return content, nil, nil
case agentContext.ContentImageURL:
// Image URL - check if it needs processing
return processImageURLContent(ctx, part, capabilities, uses, forceUses, registry, fetcher, processedFiles)
return image.New(options).Parse(ctx, content)
case agentContext.ContentInputAudio:
// Audio - check if it needs processing
return processAudioContent(ctx, part, capabilities, uses, registry, fetcher, processedFiles)
}
return content, nil, nil
// 2. Handle extended types - MUST convert to standard types
switch part.Type {
case agentContext.ContentFile:
return processFileContent(ctx, part, capabilities, uses, forceUses, registry, fetcher, processedFiles)
return content, nil, nil
case agentContext.ContentData:
return processDataContent(ctx, part)
return content, nil, nil
default:
// Unknown type, return error
return nil, fmt.Errorf("unsupported content type: %s", part.Type)
return content, nil, fmt.Errorf("unsupported content part type: %s", content.Type)
}
}
// processFileContent processes file content with caching
func processFileContent(
ctx *agentContext.Context,
part *agentContext.ContentPart,
capabilities *openai.Capabilities,
uses *agentContext.Uses,
forceUses bool,
registry *Registry,
fetcher Fetcher,
processedFiles map[string]string,
) (*Result, error) {
if part.File == nil || part.File.URL == "" {
return nil, fmt.Errorf("file content part missing URL")
}
url := part.File.URL
// Step 1: Try to get cached text (three-tier cache)
cachedText, found, err := tryGetCachedText(ctx, url, processedFiles)
if err != nil {
return nil, fmt.Errorf("failed to check cache: %w", err)
}
if found {
// Cache hit! Return as text
return &Result{
Text: cachedText,
}, nil
}
// Step 2: No cache, need to process the file
// Determine content source
source, sourceURL, err := determineContentSource(part)
if err != nil {
return nil, fmt.Errorf("failed to determine content source: %w", err)
}
// Fetch content
info, err := fetcher.Fetch(ctx, source, sourceURL)
if err != nil {
return nil, fmt.Errorf("failed to fetch content: %w", err)
}
// Set filename from part if not already set
if info.Filename == "" && part.File.Filename != "" {
info.Filename = part.File.Filename
}
// Detect file type if not already set
if info.FileType == FileTypeUnknown {
info.FileType = DetectFileType(info.ContentType, part.File.Filename)
}
// Process with appropriate handler
result, err := registry.Handle(ctx, info, capabilities, uses, forceUses)
if err != nil {
return nil, fmt.Errorf("failed to handle content: %w", err)
}
// Step 3: Cache the result if it's text
if result.Text != "" {
if cacheErr := cacheProcessedText(ctx, url, result.Text, processedFiles); cacheErr != nil {
// Log error but don't fail the request
fmt.Printf("Warning: failed to cache processed text: %v\n", cacheErr)
}
}
return result, nil
}
// processImageURLContent processes image_url content
// If URL is uploader wrapper or HTTP, fetch and process it
func processImageURLContent(
ctx *agentContext.Context,
part *agentContext.ContentPart,
capabilities *openai.Capabilities,
uses *agentContext.Uses,
forceUses bool,
registry *Registry,
fetcher Fetcher,
processedFiles map[string]string,
) (*Result, error) {
if part.ImageURL == nil || part.ImageURL.URL == "" {
return nil, fmt.Errorf("image_url content missing URL")
}
url := part.ImageURL.URL
// If it's a data URI (base64), pass through
if strings.HasPrefix(url, "data:") {
return &Result{
ContentPart: part,
}, nil
}
// Check model capabilities
supportsVision := false
if capabilities != nil && capabilities.Vision != nil {
// Vision can be bool or string (format)
switch v := capabilities.Vision.(type) {
case bool:
supportsVision = v
case string:
supportsVision = v != "" && v != "false" && v != "none"
}
}
// If model supports vision AND we're not forcing uses, pass through
if supportsVision && !forceUses {
return &Result{
ContentPart: part,
}, nil
}
// Model doesn't support vision OR forceUses is true
// Need to convert image to text
// If it's uploader wrapper or HTTP URL, need to process
// Check cache first
cachedText, found, err := tryGetCachedText(ctx, url, processedFiles)
if err != nil {
return nil, fmt.Errorf("failed to check cache: %w", err)
}
if found {
// Cache hit! Return as text
return &Result{
Text: cachedText,
}, nil
}
// Determine source
source, sourceURL, err := determineContentSource(part)
if err != nil {
return nil, fmt.Errorf("failed to determine content source: %w", err)
}
// Fetch content
info, err := fetcher.Fetch(ctx, source, sourceURL)
if err != nil {
return nil, fmt.Errorf("failed to fetch image: %w", err)
}
// Set file type as image
info.FileType = FileTypeImage
// Process with image handler
result, err := registry.Handle(ctx, info, capabilities, uses, forceUses)
if err != nil {
return nil, fmt.Errorf("failed to handle image: %w", err)
}
// Cache if result is text
if result.Text != "" {
if cacheErr := cacheProcessedText(ctx, url, result.Text, processedFiles); cacheErr != nil {
fmt.Printf("Warning: failed to cache processed text: %v\n", cacheErr)
}
}
return result, nil
}
// processAudioContent processes input_audio content
func processAudioContent(
ctx *agentContext.Context,
part *agentContext.ContentPart,
capabilities *openai.Capabilities,
uses *agentContext.Uses,
registry *Registry,
fetcher Fetcher,
processedFiles map[string]string,
) (*Result, error) {
if part.InputAudio == nil || part.InputAudio.Data == "" {
return nil, fmt.Errorf("input_audio content missing data")
}
// For now, pass through audio as-is
// TODO: Implement audio processing (transcription, etc.)
return &Result{
ContentPart: part,
}, nil
}
// processDataContent processes data content (converts to text)
func processDataContent(ctx *agentContext.Context, part *agentContext.ContentPart) (*Result, error) {
if part.Data == nil {
return nil, fmt.Errorf("data content part missing data")
}
// TODO: Implement data processing
// For now, just return error
return nil, fmt.Errorf("data content processing not implemented yet")
}
// determineContentSource determines where the content comes from
func determineContentSource(part *agentContext.ContentPart) (Source, string, error) {
var url string
// Extract URL based on content type
switch part.Type {
case agentContext.ContentFile:
if part.File == nil || part.File.URL == "" {
return "", "", fmt.Errorf("file content missing URL")
}
url = part.File.URL
case agentContext.ContentImageURL:
if part.ImageURL == nil || part.ImageURL.URL == "" {
return "", "", fmt.Errorf("image_url content missing URL")
}
url = part.ImageURL.URL
case agentContext.ContentInputAudio:
if part.InputAudio == nil || part.InputAudio.Data == "" {
return "", "", fmt.Errorf("input_audio content missing data")
}
// Audio data is base64, treat as base64 source
return SourceBase64, part.InputAudio.Data, nil
default:
return "", "", fmt.Errorf("unsupported content type for source detection: %s", part.Type)
}
// Determine source type based on URL format
if strings.HasPrefix(url, "http://") || strings.HasPrefix(url, "https://") {
return SourceHTTP, url, nil
}
if strings.HasPrefix(url, "__") {
// Uploader wrapper format: __uploader://fileid
return SourceUploader, url, nil
}
if strings.HasPrefix(url, "data:") {
// Data URI (base64)
return SourceBase64, url, nil
}
// Default to treating as uploader if no prefix matches
return SourceUploader, url, nil
}
// shouldProcessWithModel checks if content should be processed by the model directly
func shouldProcessWithModel(capabilities *openai.Capabilities, fileType FileType) (bool, agentContext.VisionFormat) {
// TODO: Implement model capability check
// For images: check if model supports vision
// For audio: check if model supports audio input
// Return whether to use model and the format to use
return false, agentContext.VisionFormatNone
}
// getToolForProcessing gets the agent/MCP tool to use for processing
func getToolForProcessing(uses *agentContext.Uses, fileType FileType) string {
// TODO: Implement tool selection
// Based on file type, return the appropriate tool from uses
// - Images -> uses.Vision
// - Audio -> uses.Audio
// - PDF (if vision available) -> uses.Vision
return ""
}
// tryGetCachedText checks if the URL is an uploader wrapper and tries to get cached text
// Returns (text, found, error)
func tryGetCachedText(ctx *agentContext.Context, url string, processedFiles map[string]string) (string, bool, error) {
// Parse URL to check if it's an uploader wrapper
uploaderName, fileID, isWrapper := attachment.Parse(url)
if !isWrapper {
return "", false, nil // Not an uploader wrapper, no cache
}
// 1. Check in-memory cache for this Vision call
if text, ok := processedFiles[fileID]; ok {
return text, true, nil
}
// 2. Try attachment manager's content_preview (cross-call cache)
manager, exists := attachment.Managers[uploaderName]
if exists {
// GetText with fullContent=false to get preview (default)
text, err := manager.GetText(ctx.Context, fileID, false)
if err == nil && text != "" {
// Cache in-memory for this Vision call
processedFiles[fileID] = text
return text, true, nil
}
}
// No cache found
return "", false, nil
}
// cacheProcessedText caches the processed text for an uploader wrapper
func cacheProcessedText(ctx *agentContext.Context, url string, text string, processedFiles map[string]string) error {
// Parse URL to get uploader name and file ID
uploaderName, fileID, isWrapper := attachment.Parse(url)
if !isWrapper {
return nil // Not an uploader wrapper, nothing to cache
}
// 1. Cache in-memory for this Vision call
processedFiles[fileID] = text
// 2. Save to attachment manager for future Vision calls
manager, exists := attachment.Managers[uploaderName]
if exists {
return manager.SaveText(ctx.Context, fileID, text)
}
return nil
}
// convertToContentParts converts []interface{} to []ContentPart
// This is needed when content is loaded from JSON/history and is []interface{} instead of []ContentPart
func convertToContentParts(content interface{}) ([]agentContext.ContentPart, bool) {
// Check if it's []interface{}
arr, ok := content.([]interface{})
if !ok {
return nil, false
}
parts := make([]agentContext.ContentPart, 0, len(arr))
for _, item := range arr {
func convertToContentParts(content []interface{}) ([]agentContext.ContentPart, bool) {
parts := make([]agentContext.ContentPart, 0, len(content))
for _, item := range content {
// Each item should be a map
m, ok := item.(map[string]interface{})
if !ok {
@ -578,6 +168,35 @@ func convertToContentParts(content interface{}) ([]agentContext.ContentPart, boo
part.InputAudio.Format = format
}
}
case "data":
if dataContent, ok := m["data"].(map[string]interface{}); ok {
part.Data = &agentContext.DataContent{}
if sources, ok := dataContent["sources"].([]interface{}); ok {
part.Data.Sources = make([]agentContext.DataSource, 0, len(sources))
for _, src := range sources {
if srcMap, ok := src.(map[string]interface{}); ok {
source := agentContext.DataSource{}
if t, ok := srcMap["type"].(string); ok {
source.Type = agentContext.DataSourceType(t)
}
if name, ok := srcMap["name"].(string); ok {
source.Name = name
}
if id, ok := srcMap["id"].(string); ok {
source.ID = id
}
if filters, ok := srcMap["filters"].(map[string]interface{}); ok {
source.Filters = filters
}
if metadata, ok := srcMap["metadata"].(map[string]interface{}); ok {
source.Metadata = metadata
}
part.Data.Sources = append(part.Data.Sources, source)
}
}
}
}
}
parts = append(parts, part)

View file

@ -1,675 +0,0 @@
package content_test
import (
"bytes"
"context"
"image"
"image/color"
"image/png"
"mime/multipart"
"strings"
"testing"
"github.com/yaoapp/gou/connector/openai"
"github.com/yaoapp/yao/agent/content"
agentContext "github.com/yaoapp/yao/agent/context"
"github.com/yaoapp/yao/agent/testutils"
"github.com/yaoapp/yao/attachment"
)
// setupTestUploader creates and registers a test uploader manager
// The manager will be registered with "__" prefix as required by attachment.Parse
func setupTestUploader(t *testing.T, name string) attachment.FileManager {
// Register with __ prefix to match Parse behavior
managerName := "__" + name
manager, err := attachment.Register(managerName, "local", attachment.ManagerOption{
Driver: "local",
MaxSize: "10M",
AllowedTypes: []string{"text/*", "image/*", "application/*"},
Options: map[string]interface{}{
"path": "/tmp/test_vision_attachments_" + name,
},
})
if err != nil {
t.Fatalf("Failed to register attachment manager '%s': %v", managerName, err)
}
return manager
}
// cleanupTestUploader removes the test uploader from registry
func cleanupTestUploader(name string) {
delete(attachment.Managers, "__"+name)
}
// generateTestImage creates a valid PNG image (100x100 red square)
func generateTestImage(t *testing.T) []byte {
img := image.NewRGBA(image.Rect(0, 0, 100, 100))
red := color.RGBA{255, 0, 0, 255}
for y := 0; y < 100; y++ {
for x := 0; x < 100; x++ {
img.Set(x, y, red)
}
}
var buf bytes.Buffer
if err := png.Encode(&buf, img); err != nil {
t.Fatalf("Failed to encode test image: %v", err)
}
return buf.Bytes()
}
// TestVision_TextFile tests Vision function with text/code file parsing
func TestVision_TextFile(t *testing.T) {
testutils.Prepare(t)
defer testutils.Clean(t)
// Setup test uploader
uploaderName := "test-vision-text"
manager := setupTestUploader(t, uploaderName)
defer cleanupTestUploader(uploaderName)
// 1. Create and upload a Go source file
testContent := `package main
import "fmt"
func main() {
fmt.Println("Hello, Vision Test!")
}
`
// Upload file
reader := strings.NewReader(testContent)
fileHeader := &attachment.FileHeader{
FileHeader: &multipart.FileHeader{
Filename: "main.go",
Size: int64(len(testContent)),
Header: make(map[string][]string),
},
}
fileHeader.Header.Set("Content-Type", "text/x-go")
uploadedFile, err := manager.Upload(context.Background(), fileHeader, reader, attachment.UploadOption{
Groups: []string{"vision", "test"},
})
if err != nil {
t.Fatalf("Failed to upload file: %v", err)
}
t.Logf("Uploaded file ID: %s", uploadedFile.ID)
// 2. Prepare Vision context (text files don't need special capabilities)
ctx := agentContext.New(context.Background(), nil, "test")
capabilities := &openai.Capabilities{}
messages := []agentContext.Message{
{
Role: "user",
Content: []agentContext.ContentPart{
{
Type: agentContext.ContentFile,
File: &agentContext.FileAttachment{
URL: "__" + uploaderName + "://" + uploadedFile.ID,
Filename: "main.go",
},
},
},
},
}
// 3. Call Vision function
result, err := content.Vision(ctx, capabilities, messages, nil, false)
if err != nil {
t.Fatalf("Vision function failed: %v", err)
}
if len(result) != 1 {
t.Fatalf("Expected 1 message, got %d", len(result))
}
// 4. Verify result
contentParts, ok := result[0].Content.([]agentContext.ContentPart)
if !ok {
t.Fatalf("Expected content to be []ContentPart, got %T", result[0].Content)
}
if len(contentParts) != 1 {
t.Fatalf("Expected 1 content part, got %d", len(contentParts))
}
// Should be converted to text
if contentParts[0].Type != agentContext.ContentText {
t.Errorf("Expected ContentText type, got %s", contentParts[0].Type)
}
if !strings.Contains(contentParts[0].Text, "package main") {
t.Errorf("Expected text to contain 'package main', got: %s", contentParts[0].Text)
}
if !strings.Contains(contentParts[0].Text, "Hello, Vision Test!") {
t.Errorf("Expected text to contain 'Hello, Vision Test!', got: %s", contentParts[0].Text)
}
t.Logf("✓ Text file successfully parsed: %d characters", len(contentParts[0].Text))
}
// TestVision_ImageWithVisionSupport tests image processing with vision-capable model
func TestVision_ImageWithVisionSupport(t *testing.T) {
testutils.Prepare(t)
defer testutils.Clean(t)
// Setup test uploader
uploaderName := "test-vision-image"
manager := setupTestUploader(t, uploaderName)
defer cleanupTestUploader(uploaderName)
// 1. Create and upload a test image (1x1 red PNG)
imageData := []byte{
0x89, 0x50, 0x4E, 0x47, 0x0D, 0x0A, 0x1A, 0x0A,
0x00, 0x00, 0x00, 0x0D, 0x49, 0x48, 0x44, 0x52,
0x00, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01,
0x08, 0x02, 0x00, 0x00, 0x00, 0x90, 0x77, 0x53,
0xDE, 0x00, 0x00, 0x00, 0x0C, 0x49, 0x44, 0x41,
0x54, 0x08, 0xD7, 0x63, 0xF8, 0xCF, 0xC0, 0x00,
0x00, 0x03, 0x01, 0x01, 0x00, 0x18, 0xDD, 0x8D,
0xB4, 0x00, 0x00, 0x00, 0x00, 0x49, 0x45, 0x4E,
0x44, 0xAE, 0x42, 0x60, 0x82,
}
// Upload image
reader := strings.NewReader(string(imageData))
fileHeader := &attachment.FileHeader{
FileHeader: &multipart.FileHeader{
Filename: "test.png",
Size: int64(len(imageData)),
Header: make(map[string][]string),
},
}
fileHeader.Header.Set("Content-Type", "image/png")
uploadedFile, err := manager.Upload(context.Background(), fileHeader, reader, attachment.UploadOption{
Groups: []string{"vision", "test"},
})
if err != nil {
t.Fatalf("Failed to upload image: %v", err)
}
// 2. Prepare Vision context with vision-capable model
ctx := agentContext.New(context.Background(), nil, "test")
// Construct capabilities with vision support (OpenAI format)
capabilities := &openai.Capabilities{
Vision: agentContext.VisionFormatOpenAI, // OpenAI vision format
}
messages := []agentContext.Message{
{
Role: "user",
Content: []agentContext.ContentPart{
{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: "__" + uploaderName + "://" + uploadedFile.ID,
},
},
},
},
}
// 3. Call Vision function (no uses needed for direct vision support)
result, err := content.Vision(ctx, capabilities, messages, nil, false)
if err != nil {
t.Fatalf("Vision function failed: %v", err)
}
if len(result) != 1 {
t.Fatalf("Expected 1 message, got %d", len(result))
}
// 4. Verify result
contentParts, ok := result[0].Content.([]agentContext.ContentPart)
if !ok {
t.Fatalf("Expected content to be []ContentPart, got %T", result[0].Content)
}
if len(contentParts) != 1 {
t.Fatalf("Expected 1 content part, got %d", len(contentParts))
}
// If model supports vision, should be image_url with base64
if capabilities.Vision != nil {
if contentParts[0].Type != agentContext.ContentImageURL {
t.Errorf("Expected ContentImageURL type, got %s", contentParts[0].Type)
}
if contentParts[0].ImageURL == nil {
t.Fatal("Expected ImageURL to be set")
}
if !strings.Contains(contentParts[0].ImageURL.URL, "data:image/png;base64,") {
t.Errorf("Expected base64 data URI, got: %s", contentParts[0].ImageURL.URL)
}
t.Logf("✓ Image processed with vision support: %d bytes (base64)", len(contentParts[0].ImageURL.URL))
} else {
// If no vision support, should fall back to text (via agent/MCP)
t.Logf(" Model doesn't support vision, result type: %s", contentParts[0].Type)
}
}
// TestVision_ImageWithAgent tests image processing with vision agent when model doesn't support vision
// Note: This test demonstrates the agent fallback mechanism when the model doesn't support vision
func TestVision_ImageWithAgent(t *testing.T) {
testutils.Prepare(t)
defer testutils.Clean(t)
// Setup test uploader
uploaderName := "test-vision-agent"
manager := setupTestUploader(t, uploaderName)
defer cleanupTestUploader(uploaderName)
// 1. Generate and upload a valid test image (100x100 red PNG)
imageData := generateTestImage(t)
reader := strings.NewReader(string(imageData))
fileHeader := &attachment.FileHeader{
FileHeader: &multipart.FileHeader{
Filename: "test.png",
Size: int64(len(imageData)),
Header: make(map[string][]string),
},
}
fileHeader.Header.Set("Content-Type", "image/png")
uploadedFile, err := manager.Upload(context.Background(), fileHeader, reader, attachment.UploadOption{
Groups: []string{"vision", "test"},
})
if err != nil {
t.Fatalf("Failed to upload image: %v", err)
}
// 2. Prepare Vision context with proper setup
// Model does NOT support vision, but uses.Vision specifies a vision agent
ctx := agentContext.New(context.Background(), nil, "test")
// Capabilities without vision support
capabilities := &openai.Capabilities{
Vision: nil, // No vision support
}
// Uses configuration with vision agent
uses := &agentContext.Uses{
Vision: "tests.vision-helper", // Use vision-helper agent
}
messages := []agentContext.Message{
{
Role: "user",
Content: []agentContext.ContentPart{
{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: "__" + uploaderName + "://" + uploadedFile.ID,
},
},
},
},
}
// 3. Call Vision - should use agent since model doesn't support vision
result, err := content.Vision(ctx, capabilities, messages, uses, false)
if err != nil {
t.Fatalf("Vision function failed: %v", err)
}
if len(result) != 1 {
t.Fatalf("Expected 1 message, got %d", len(result))
}
// 4. Verify result is text (processed by vision agent)
contentParts, ok := result[0].Content.([]agentContext.ContentPart)
if !ok {
t.Fatalf("Expected content to be []ContentPart, got %T", result[0].Content)
}
if len(contentParts) != 1 {
t.Fatalf("Expected 1 content part, got %d", len(contentParts))
}
if contentParts[0].Type != agentContext.ContentText {
t.Errorf("Expected ContentText (from agent), got: %s", contentParts[0].Type)
}
if contentParts[0].Text == "" {
t.Error("Expected non-empty text from vision agent processing")
}
t.Logf("✓ Vision agent processed image to text: %d characters", len(contentParts[0].Text))
t.Logf("Agent response text:\n%s", contentParts[0].Text)
}
// TestVision_CachedContent tests that file content is cached and reused
func TestVision_CachedContent(t *testing.T) {
testutils.Prepare(t)
defer testutils.Clean(t)
// Setup test uploader
uploaderName := "test-vision-cache"
manager := setupTestUploader(t, uploaderName)
defer cleanupTestUploader(uploaderName)
// 1. Upload a text file
testContent := "Test content for caching verification"
reader := strings.NewReader(testContent)
fileHeader := &attachment.FileHeader{
FileHeader: &multipart.FileHeader{
Filename: "cache-test.txt",
Size: int64(len(testContent)),
Header: make(map[string][]string),
},
}
fileHeader.Header.Set("Content-Type", "text/plain")
uploadedFile, err := manager.Upload(context.Background(), fileHeader, reader, attachment.UploadOption{
Groups: []string{"vision", "test"},
})
if err != nil {
t.Fatalf("Failed to upload file: %v", err)
}
// 2. Prepare Vision context with same file referenced twice
ctx := agentContext.New(context.Background(), nil, "test")
// Construct simple capabilities (text files don't need vision)
capabilities := &openai.Capabilities{}
messages := []agentContext.Message{
{
Role: "user",
Content: []agentContext.ContentPart{
{
Type: agentContext.ContentFile,
File: &agentContext.FileAttachment{
URL: "__" + uploaderName + "://" + uploadedFile.ID,
Filename: "cache-test.txt",
},
},
{
Type: agentContext.ContentFile,
File: &agentContext.FileAttachment{
URL: "__" + uploaderName + "://" + uploadedFile.ID, // Same file
Filename: "cache-test.txt",
},
},
},
},
}
// 3. Call Vision
result, err := content.Vision(ctx, capabilities, messages, nil, false)
if err != nil {
t.Fatalf("Vision function failed: %v", err)
}
if len(result) != 1 {
t.Fatalf("Expected 1 message, got %d", len(result))
}
// 4. Verify both file references were processed
contentParts, ok := result[0].Content.([]agentContext.ContentPart)
if !ok {
t.Fatalf("Expected content to be []ContentPart")
}
if len(contentParts) != 2 {
t.Fatalf("Expected 2 content parts (both files), got %d", len(contentParts))
}
// Both should be text with same content
if contentParts[0].Type != agentContext.ContentText {
t.Errorf("First part: expected ContentText, got %s", contentParts[0].Type)
}
if contentParts[1].Type != agentContext.ContentText {
t.Errorf("Second part: expected ContentText, got %s", contentParts[1].Type)
}
if !strings.Contains(contentParts[0].Text, testContent) {
t.Errorf("First part text doesn't contain expected content")
}
if !strings.Contains(contentParts[1].Text, testContent) {
t.Errorf("Second part text doesn't contain expected content")
}
// Verify content was cached (check attachment manager)
cachedText, err := manager.GetText(context.Background(), uploadedFile.ID)
if err != nil {
t.Fatalf("Failed to get cached text: %v", err)
}
if cachedText == "" {
t.Error("Expected content to be cached in attachment manager")
}
t.Logf("✓ Content successfully cached and reused: %d characters", len(cachedText))
}
// TestVision_FileMetadataInMemory tests that file metadata is correctly passed to vision agent via ctx.Memory.Context
func TestVision_FileMetadataInMemory(t *testing.T) {
testutils.Prepare(t)
defer testutils.Clean(t)
// Setup test uploader
uploaderName := "test-vision-metadata"
manager := setupTestUploader(t, uploaderName)
defer cleanupTestUploader(uploaderName)
// 1. Generate and upload a test image
imageData := generateTestImage(t)
reader := strings.NewReader(string(imageData))
fileHeader := &attachment.FileHeader{
FileHeader: &multipart.FileHeader{
Filename: "test-metadata.png",
Size: int64(len(imageData)),
Header: make(map[string][]string),
},
}
fileHeader.Header.Set("Content-Type", "image/png")
uploadedFile, err := manager.Upload(context.Background(), fileHeader, reader, attachment.UploadOption{
Groups: []string{"vision", "test"},
})
if err != nil {
t.Fatalf("Failed to upload image: %v", err)
}
t.Logf("Uploaded file: %s (ID: %s)", uploadedFile.Filename, uploadedFile.ID)
// 2. Prepare Vision context - model doesn't support vision, use agent
ctx := agentContext.New(context.Background(), nil, "test")
// Capabilities without vision support
capabilities := &openai.Capabilities{
Vision: nil, // No vision support - force agent usage
}
// Uses configuration with vision-helper agent
uses := &agentContext.Uses{
Vision: "tests.vision-helper", // Use vision-helper agent that logs file metadata
}
messages := []agentContext.Message{
{
Role: "user",
Content: []agentContext.ContentPart{
{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: "__" + uploaderName + "://" + uploadedFile.ID,
},
},
},
},
}
// 3. Call Vision - should pass file metadata to agent via Space
result, err := content.Vision(ctx, capabilities, messages, uses, false)
if err != nil {
t.Fatalf("Vision function failed: %v", err)
}
if len(result) != 1 {
t.Fatalf("Expected 1 message, got %d", len(result))
}
// 4. Verify result contains file metadata validation info
contentParts, ok := result[0].Content.([]agentContext.ContentPart)
if !ok {
t.Fatalf("Expected content to be []ContentPart, got %T", result[0].Content)
}
if len(contentParts) != 1 {
t.Fatalf("Expected 1 content part, got %d", len(contentParts))
}
if contentParts[0].Type != agentContext.ContentText {
t.Errorf("Expected ContentText (from agent), got: %s", contentParts[0].Type)
}
// The vision-helper assistant should have logged file metadata from Space
// We can verify this through the response text (which should contain the description)
if contentParts[0].Text == "" {
t.Error("Expected non-empty text from vision agent")
}
t.Logf("✓ Vision agent processed image with file metadata")
t.Logf("Agent response: %s", contentParts[0].Text)
// Note: The actual validation of Space data happens in the vision-helper's Next hook
// which logs the file metadata. In a real test, we would need to capture those logs
// or have the agent return structured data that we can verify.
}
// TestVision_MultipleFilesMetadata tests file metadata handling with multiple attachments
func TestVision_MultipleFilesMetadata(t *testing.T) {
testutils.Prepare(t)
defer testutils.Clean(t)
// Setup test uploader
uploaderName := "test-vision-multi"
manager := setupTestUploader(t, uploaderName)
defer cleanupTestUploader(uploaderName)
// 1. Upload two test images
imageData1 := generateTestImage(t)
imageData2 := generateTestImage(t)
// Upload first image
reader1 := strings.NewReader(string(imageData1))
fileHeader1 := &attachment.FileHeader{
FileHeader: &multipart.FileHeader{
Filename: "test-image-1.png",
Size: int64(len(imageData1)),
Header: make(map[string][]string),
},
}
fileHeader1.Header.Set("Content-Type", "image/png")
uploadedFile1, err := manager.Upload(context.Background(), fileHeader1, reader1, attachment.UploadOption{
Groups: []string{"vision", "test"},
})
if err != nil {
t.Fatalf("Failed to upload first image: %v", err)
}
// Upload second image
reader2 := strings.NewReader(string(imageData2))
fileHeader2 := &attachment.FileHeader{
FileHeader: &multipart.FileHeader{
Filename: "test-image-2.png",
Size: int64(len(imageData2)),
Header: make(map[string][]string),
},
}
fileHeader2.Header.Set("Content-Type", "image/png")
uploadedFile2, err := manager.Upload(context.Background(), fileHeader2, reader2, attachment.UploadOption{
Groups: []string{"vision", "test"},
})
if err != nil {
t.Fatalf("Failed to upload second image: %v", err)
}
t.Logf("Uploaded files: %s (ID: %s), %s (ID: %s)",
uploadedFile1.Filename, uploadedFile1.ID,
uploadedFile2.Filename, uploadedFile2.ID)
// 2. Prepare Vision context with both images
ctx := agentContext.New(context.Background(), nil, "test")
capabilities := &openai.Capabilities{
Vision: nil, // No vision support
}
uses := &agentContext.Uses{
Vision: "tests.vision-helper",
}
messages := []agentContext.Message{
{
Role: "user",
Content: []agentContext.ContentPart{
{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: "__" + uploaderName + "://" + uploadedFile1.ID,
},
},
{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: "__" + uploaderName + "://" + uploadedFile2.ID,
},
},
},
},
}
// 3. Call Vision - should handle multiple files' metadata
result, err := content.Vision(ctx, capabilities, messages, uses, false)
if err != nil {
t.Fatalf("Vision function failed: %v", err)
}
if len(result) != 1 {
t.Fatalf("Expected 1 message, got %d", len(result))
}
// 4. Verify both images were processed
contentParts, ok := result[0].Content.([]agentContext.ContentPart)
if !ok {
t.Fatalf("Expected content to be []ContentPart, got %T", result[0].Content)
}
// Each image should be processed by the vision agent and return text
if len(contentParts) != 2 {
t.Fatalf("Expected 2 content parts (one per image), got %d", len(contentParts))
}
for i, part := range contentParts {
if part.Type != agentContext.ContentText {
t.Errorf("Part %d: expected ContentText, got: %s", i, part.Type)
}
if part.Text == "" {
t.Errorf("Part %d: expected non-empty text", i)
}
}
t.Logf("✓ Multiple files processed with metadata tracking")
t.Logf("File 1 response: %s", contentParts[0].Text)
t.Logf("File 2 response: %s", contentParts[1].Text)
}

View file

@ -1,48 +0,0 @@
package content
import (
"fmt"
"strings"
"github.com/yaoapp/gou/connector/openai"
agentContext "github.com/yaoapp/yao/agent/context"
)
// ExcelHandler handles Microsoft Excel spreadsheets
type ExcelHandler struct{}
// CanHandle checks if this handler can handle the content type
func (h *ExcelHandler) CanHandle(contentType string, fileType FileType) bool {
return fileType == FileTypeExcel ||
contentType == "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" ||
contentType == "application/vnd.ms-excel" ||
strings.Contains(contentType, "excel") ||
strings.Contains(contentType, "spreadsheet")
}
// Handle processes Excel spreadsheet content
func (h *ExcelHandler) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses, forceUses bool) (*Result, error) {
// TODO: Implement Excel handling
// 1. Extract data from .xlsx or .xls file
// 2. Convert to text format (e.g., CSV-like or structured text)
// 3. Handle multiple sheets
// 4. Return Result with formatted text
return nil, fmt.Errorf("not implemented")
}
// extractExcelText extracts text from Excel file
func extractExcelText(data []byte, contentType string) (string, error) {
// TODO: Implement Excel text extraction
// Handle both .xls (old format) and .xlsx (new format)
// Consider using libraries like:
// - github.com/360EntSecGroup-Skylar/excelize for .xlsx
// Format output as readable text or CSV
return "", fmt.Errorf("not implemented")
}
// formatExcelAsText formats Excel data as readable text
func formatExcelAsText(sheets map[string][][]string) string {
// TODO: Format multiple sheets into readable text
// Include sheet names, headers, and data
return ""
}

View file

@ -1,95 +0,0 @@
package content
import (
"fmt"
agentContext "github.com/yaoapp/yao/agent/context"
"github.com/yaoapp/yao/attachment"
)
// DefaultFetcher implements the Fetcher interface
type DefaultFetcher struct{}
// NewFetcher creates a new default fetcher
func NewFetcher() Fetcher {
return &DefaultFetcher{}
}
// Fetch retrieves content from HTTP URL or uploader wrapper
func (f *DefaultFetcher) Fetch(ctx *agentContext.Context, source Source, url string) (*Info, error) {
switch source {
case SourceHTTP:
return f.fetchHTTP(ctx, url)
case SourceUploader:
return f.fetchUploader(ctx, url)
default:
return nil, fmt.Errorf("unsupported source: %s", source)
}
}
// fetchHTTP fetches content from an HTTP(S) URL
func (f *DefaultFetcher) fetchHTTP(ctx *agentContext.Context, url string) (*Info, error) {
// TODO: Implement HTTP fetch logic
// 1. Download file from URL
// 2. Detect content type
// 3. Detect file type based on content type and extension
// 4. Return Info with data
return nil, fmt.Errorf("not implemented")
}
// fetchUploader fetches content from uploader wrapper (__uploader://fileid)
func (f *DefaultFetcher) fetchUploader(ctx *agentContext.Context, wrapper string) (*Info, error) {
// 1. Parse wrapper to get uploader name and file ID
uploaderName, fileID, ok := attachment.Parse(wrapper)
if !ok {
return nil, fmt.Errorf("invalid uploader wrapper format: %s", wrapper)
}
// 2. Get attachment manager
var manager attachment.FileManager
var exists bool
// Try to get manager by name
manager, exists = attachment.Managers[uploaderName]
if !exists {
return nil, fmt.Errorf("uploader '%s' not found", uploaderName)
}
// 3. Get file info
file, err := manager.Info(ctx.Context, fileID)
if err != nil {
return nil, fmt.Errorf("failed to get file info: %w", err)
}
// 4. Read file content
data, err := manager.Read(ctx.Context, fileID)
if err != nil {
return nil, fmt.Errorf("failed to read file: %w", err)
}
// 5. Return Info with data
return &Info{
Data: data,
ContentType: file.ContentType,
Filename: file.Filename,
FileType: DetectFileType(file.ContentType, file.Filename),
URL: wrapper,
Source: SourceUploader,
UploaderName: uploaderName,
FileID: fileID,
}, nil
}
// parseUploaderWrapper parses uploader wrapper format: __uploader://fileid
func parseUploaderWrapper(wrapper string) (uploaderName, fileID string, err error) {
// TODO: Implement wrapper parsing
// Format: __uploader://fileid
return "", "", fmt.Errorf("not implemented")
}
// detectFileType detects file type from content type and data
func detectFileType(contentType string, data []byte) FileType {
// TODO: Implement file type detection
// Based on content type and magic bytes
return FileTypeUnknown
}

View file

@ -1,188 +0,0 @@
package content
import (
"fmt"
"strings"
"github.com/yaoapp/gou/connector/openai"
agentContext "github.com/yaoapp/yao/agent/context"
)
// ImageHandler handles image content
type ImageHandler struct{}
// CanHandle checks if this handler can handle the content type
func (h *ImageHandler) CanHandle(contentType string, fileType FileType) bool {
return fileType == FileTypeImage || strings.HasPrefix(contentType, "image/")
}
// Handle processes image content
// Logic:
// 1. If forceUses is true and uses.Vision is specified -> use vision tool regardless of model capability
// 2. If model supports vision and forceUses is false -> convert to base64 or image_url format
// 3. If model doesn't support vision -> use agent/MCP specified in uses.Vision
func (h *ImageHandler) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses, forceUses bool) (*Result, error) {
if len(info.Data) == 0 {
return nil, fmt.Errorf("no image data to process")
}
if capabilities == nil {
return nil, fmt.Errorf("no capabilities provided")
}
// Check if model supports vision
supportsVision, visionFormat := agentContext.GetVisionSupport(capabilities)
// If forceUses is true and uses.Vision is specified, use vision tool regardless of model capability
if forceUses && uses != nil && uses.Vision != "" {
text, err := h.handleWithVisionAgent(ctx, info, uses.Vision)
if err != nil {
return nil, fmt.Errorf("failed to handle image with vision agent/MCP (forced): %w", err)
}
return &Result{
Text: text,
}, nil
}
if supportsVision {
// Model supports vision - return as image_url ContentPart
contentPart, err := h.handleWithVisionModel(ctx, info, visionFormat)
if err != nil {
return nil, fmt.Errorf("failed to handle image with vision model: %w", err)
}
return &Result{
ContentPart: contentPart,
}, nil
}
// Model doesn't support vision - use vision agent/MCP
visionTool := ""
if uses != nil && uses.Vision != "" {
visionTool = uses.Vision
}
if visionTool == "" {
return nil, fmt.Errorf("model doesn't support vision and no vision tool specified in uses.Vision")
}
// Call vision agent/MCP to extract text
text, err := h.handleWithVisionAgent(ctx, info, visionTool)
if err != nil {
return nil, fmt.Errorf("failed to handle image with vision agent/MCP: %w", err)
}
return &Result{
Text: text,
}, nil
}
// handleWithVisionModel processes image using model's vision capability
func (h *ImageHandler) handleWithVisionModel(ctx *agentContext.Context, info *Info, format agentContext.VisionFormat) (*agentContext.ContentPart, error) {
// Encode image to base64
base64Data := encodeImageBase64(info.Data, info.ContentType)
// Format according to model's vision format
switch format {
case agentContext.VisionFormatOpenAI:
// OpenAI format: image_url with data URI
return &agentContext.ContentPart{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: base64Data,
Detail: agentContext.DetailAuto,
},
}, nil
case agentContext.VisionFormatClaude:
// Claude format: also uses image_url but may have different handling
// For now, use the same format as OpenAI
return &agentContext.ContentPart{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: base64Data,
Detail: agentContext.DetailAuto,
},
}, nil
case agentContext.VisionFormatDefault, "":
// Default format (when Vision: true) - use OpenAI format
return &agentContext.ContentPart{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: base64Data,
Detail: agentContext.DetailAuto,
},
}, nil
default:
return nil, fmt.Errorf("unsupported vision format: %s", format)
}
}
// handleWithVisionAgent processes image using vision agent or MCP
func (h *ImageHandler) handleWithVisionAgent(ctx *agentContext.Context, info *Info, visionTool string) (string, error) {
// Parse vision tool format
// Format can be:
// - "agent_id" (call agent)
// - "mcp:server_id" (call MCP tool)
if strings.HasPrefix(visionTool, "mcp:") {
// MCP tool
serverID := strings.TrimPrefix(visionTool, "mcp:")
return h.callMCPVisionTool(ctx, serverID, info)
}
// Agent call
return h.callVisionAgent(ctx, visionTool, info)
}
// callVisionAgent calls a vision agent to describe the image
func (h *ImageHandler) callVisionAgent(ctx *agentContext.Context, agentID string, info *Info) (string, error) {
// Prepare message with image
base64Data := EncodeToBase64DataURI(info.Data, info.ContentType)
message := agentContext.Message{
Role: agentContext.RoleUser,
Content: []agentContext.ContentPart{
{
Type: agentContext.ContentText,
Text: "Please analyze this image.",
},
{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: base64Data,
Detail: agentContext.DetailAuto,
},
},
},
}
// Call agent with file metadata in context
// File info (filename, file_id, etc.) will be available in ctx.Metadata["file_info"]
// This allows hooks (especially Next hook) to access and format file information
return CallAgentWithFileInfo(ctx, agentID, message, info)
}
// callMCPVisionTool calls an MCP vision tool to describe the image
func (h *ImageHandler) callMCPVisionTool(ctx *agentContext.Context, serverID string, info *Info) (string, error) {
// Prepare base64 encoded image for MCP tool
base64Data := EncodeToBase64DataURI(info.Data, info.ContentType)
// Prepare arguments for MCP tool
arguments := map[string]interface{}{
"image": base64Data,
"content_type": info.ContentType,
}
// Call MCP tool (typically "describe_image" or similar)
return CallMCPTool(ctx, serverID, "describe_image", arguments)
}
// encodeImageBase64 encodes image data to base64 with data URI prefix
func encodeImageBase64(data []byte, contentType string) string {
// Use the common function
if contentType == "" {
contentType = "image/png" // default for images
}
return EncodeToBase64DataURI(data, contentType)
}

View file

@ -0,0 +1,406 @@
package image
import (
"encoding/base64"
"fmt"
"strings"
"github.com/yaoapp/yao/agent/content/tools"
"github.com/yaoapp/yao/agent/content/types"
agentContext "github.com/yaoapp/yao/agent/context"
"github.com/yaoapp/yao/agent/i18n"
"github.com/yaoapp/yao/agent/output/message"
searchTypes "github.com/yaoapp/yao/agent/search/types"
"github.com/yaoapp/yao/attachment"
)
// Image handles image content
type Image struct {
options *types.Options
}
// New creates a new image handler
func New(options *types.Options) *Image {
return &Image{options: options}
}
// Parse parses image content
// Logic:
// 1. Check model capabilities first
// 2. If forceUses is true and uses.Vision is specified -> use vision tool regardless of model capability
// 3. If model supports vision -> pass through or convert to base64 format
// 4. If model doesn't support vision -> use vision agent/MCP to extract text
func (h *Image) Parse(ctx *agentContext.Context, content agentContext.ContentPart) (agentContext.ContentPart, []*searchTypes.Reference, error) {
if content.ImageURL == nil || content.ImageURL.URL == "" {
return content, nil, fmt.Errorf("image_url content missing URL")
}
// Check model capabilities first
supportsVision, visionFormat := agentContext.GetVisionSupport(h.options.Capabilities)
// Check if we should force using Uses tools
forceUses := h.options.CompletionOptions != nil && h.options.CompletionOptions.ForceUses
// If forceUses is true and uses.Vision is specified, use vision tool regardless of model capability
if forceUses && h.options.CompletionOptions != nil && h.options.CompletionOptions.Uses != nil && h.options.CompletionOptions.Uses.Vision != "" {
// Check cache first before calling agent
cachedText, found, err := h.readFromCache(ctx, content.ImageURL.URL)
if err == nil && found {
return agentContext.ContentPart{
Type: agentContext.ContentText,
Text: cachedText,
}, nil, nil
}
return h.agent(ctx, content)
}
// If model supports vision
if supportsVision {
url := content.ImageURL.URL
// If it's already a data URI (base64), pass through directly
if strings.HasPrefix(url, "data:") {
return content, nil, nil
}
// Convert to base64 format
return h.base64(ctx, content, visionFormat)
}
// Model doesn't support vision - check cache first, then use vision agent/MCP
// Try to get cached text (from attachment's content_preview)
cachedText, found, err := h.readFromCache(ctx, content.ImageURL.URL)
if err == nil && found {
// Cache hit! Return as text content
return agentContext.ContentPart{
Type: agentContext.ContentText,
Text: cachedText,
}, nil, nil
}
// No cache, try to use vision agent/MCP
if h.options.CompletionOptions != nil && h.options.CompletionOptions.Uses != nil && h.options.CompletionOptions.Uses.Vision != "" {
return h.agent(ctx, content)
}
// No vision support and no vision tool specified, return error
return content, nil, fmt.Errorf("model doesn't support vision and no vision tool specified in uses.Vision")
}
// base64 encodes image content to base64 (for vision support)
func (h *Image) base64(ctx *agentContext.Context, content agentContext.ContentPart, format agentContext.VisionFormat) (agentContext.ContentPart, []*searchTypes.Reference, error) {
if content.ImageURL == nil || content.ImageURL.URL == "" {
return content, nil, fmt.Errorf("image_url content missing URL")
}
url := content.ImageURL.URL
// Read image data from source
data, contentType, err := h.read(ctx, url)
if err != nil {
return content, nil, fmt.Errorf("failed to read image: %w", err)
}
// Encode to base64 data URI
base64Data := EncodeToBase64DataURI(data, contentType)
// Return as image_url ContentPart
return agentContext.ContentPart{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: base64Data,
Detail: content.ImageURL.Detail,
},
}, nil, nil
}
// read reads image content from various sources
func (h *Image) read(ctx *agentContext.Context, url string) ([]byte, string, error) {
// Determine source type and read accordingly
if strings.HasPrefix(url, "data:") {
// Data URI format: data:image/png;base64,xxxxx
return h.readFromDataURI(url)
}
if strings.HasPrefix(url, "__") {
// Uploader wrapper format: __uploader://fileid
return h.readFromUploader(ctx, url)
}
if strings.HasPrefix(url, "http://") || strings.HasPrefix(url, "https://") {
// HTTP URL - for now return error, can be implemented later
return nil, "", fmt.Errorf("HTTP URL fetch not implemented yet: %s", url)
}
// Unknown source
return nil, "", fmt.Errorf("unsupported image source: %s", url)
}
// readFromDataURI reads image content from a data URI
func (h *Image) readFromDataURI(dataURI string) ([]byte, string, error) {
// Parse data URI: data:image/png;base64,xxxxx
if !strings.HasPrefix(dataURI, "data:") {
return nil, "", fmt.Errorf("invalid data URI format")
}
// Find the comma separator
commaIndex := strings.Index(dataURI, ",")
if commaIndex == -1 {
return nil, "", fmt.Errorf("invalid data URI: missing comma separator")
}
// Extract metadata part (e.g., "image/png;base64")
metadata := dataURI[5:commaIndex] // Skip "data:"
base64Data := dataURI[commaIndex+1:]
// Parse content type
contentType := "image/png" // default
if strings.Contains(metadata, ";") {
parts := strings.Split(metadata, ";")
if len(parts) > 0 && parts[0] != "" {
contentType = parts[0]
}
} else if metadata != "" && metadata != "base64" {
contentType = metadata
}
// Decode base64 data
data, err := base64.StdEncoding.DecodeString(base64Data)
if err != nil {
return nil, "", fmt.Errorf("failed to decode base64 data: %w", err)
}
return data, contentType, nil
}
// readFromUploader reads image content from file uploader __uploader://fileid
func (h *Image) readFromUploader(ctx *agentContext.Context, wrapper string) ([]byte, string, error) {
// Parse wrapper to get uploader name and file ID
uploaderName, fileID, ok := attachment.Parse(wrapper)
if !ok {
return nil, "", fmt.Errorf("invalid uploader wrapper format: %s", wrapper)
}
// Get attachment manager
manager, exists := attachment.Managers[uploaderName]
if !exists {
return nil, "", fmt.Errorf("uploader '%s' not found", uploaderName)
}
// Get file info
file, err := manager.Info(ctx.Context, fileID)
if err != nil {
return nil, "", fmt.Errorf("failed to get file info: %w", err)
}
// Read file content
data, err := manager.Read(ctx.Context, fileID)
if err != nil {
return nil, "", fmt.Errorf("failed to read file: %w", err)
}
return data, file.ContentType, nil
}
// readFromCache reads cached text content for an image
func (h *Image) readFromCache(ctx *agentContext.Context, url string) (string, bool, error) {
// Parse URL to check if it's an uploader wrapper
uploaderName, fileID, isWrapper := attachment.Parse(url)
if !isWrapper {
return "", false, nil // Not an uploader wrapper, no cache
}
// Try attachment manager's content_preview (cross-call cache)
manager, exists := attachment.Managers[uploaderName]
if !exists {
return "", false, nil
}
// GetText with fullContent=false to get preview (default)
text, err := manager.GetText(ctx.Context, fileID, false)
if err == nil && text != "" {
return text, true, nil
}
// No cache found
return "", false, nil
}
// saveToCache saves processed text to cache
func (h *Image) saveToCache(ctx *agentContext.Context, url string, text string) error {
// Parse URL to get uploader name and file ID
uploaderName, fileID, isWrapper := attachment.Parse(url)
if !isWrapper {
return nil // Not an uploader wrapper, nothing to cache
}
// Save to attachment manager for future calls
manager, exists := attachment.Managers[uploaderName]
if !exists {
return nil
}
return manager.SaveText(ctx.Context, fileID, text)
}
// agent calls image agent to parse image content
// Note: Cache check is done in Parse() before calling this method
func (h *Image) agent(ctx *agentContext.Context, content agentContext.ContentPart) (agentContext.ContentPart, []*searchTypes.Reference, error) {
if content.ImageURL == nil || content.ImageURL.URL == "" {
return content, nil, fmt.Errorf("image_url content missing URL")
}
url := content.ImageURL.URL
// Get vision tool from options
visionTool := ""
if h.options.CompletionOptions != nil && h.options.CompletionOptions.Uses != nil {
visionTool = h.options.CompletionOptions.Uses.Vision
}
if visionTool == "" {
return content, nil, fmt.Errorf("no vision tool specified in uses.Vision")
}
// Parse vision tool format
// Format can be:
// - "agent_id" (call agent)
// - "mcp:server_id" (call MCP tool)
var text string
var err error
if strings.HasPrefix(visionTool, "mcp:") {
// MCP tool
serverID := strings.TrimPrefix(visionTool, "mcp:")
text, err = h.callMCPVisionTool(ctx, serverID, content)
} else {
// Agent call
text, err = h.callVisionAgent(ctx, visionTool, content)
}
if err != nil {
return content, nil, fmt.Errorf("failed to process image with vision tool: %w", err)
}
// Cache the result
if cacheErr := h.saveToCache(ctx, url, text); cacheErr != nil {
// Log error but don't fail the request
fmt.Printf("Warning: failed to cache processed text: %v\n", cacheErr)
}
// Return as text content
return agentContext.ContentPart{
Type: agentContext.ContentText,
Text: text,
}, nil, nil
}
// callVisionAgent calls a vision agent to describe the image
func (h *Image) callVisionAgent(ctx *agentContext.Context, agentID string, content agentContext.ContentPart) (string, error) {
// Read image data and convert to base64
data, contentType, err := h.read(ctx, content.ImageURL.URL)
if err != nil {
return "", fmt.Errorf("failed to read image: %w", err)
}
base64Data := EncodeToBase64DataURI(data, contentType)
// Prepare message with image
message := agentContext.Message{
Role: agentContext.RoleUser,
Content: []agentContext.ContentPart{
{
Type: agentContext.ContentText,
Text: "Please analyze this image.",
},
{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: base64Data,
Detail: agentContext.DetailAuto,
},
},
},
}
// Send loading message
loadingID := h.sendLoading(ctx, i18n.T(ctx.Locale, "content.image.analyzing"))
// Call agent using the tools package
result, err := tools.CallAgent(ctx, agentID, message)
// Send done message
h.sendLoadingDone(ctx, loadingID)
return result, err
}
// callMCPVisionTool calls an MCP vision tool to describe the image
func (h *Image) callMCPVisionTool(ctx *agentContext.Context, serverID string, content agentContext.ContentPart) (string, error) {
// Read image data and convert to base64
data, contentType, err := h.read(ctx, content.ImageURL.URL)
if err != nil {
return "", fmt.Errorf("failed to read image: %w", err)
}
base64Data := EncodeToBase64DataURI(data, contentType)
// Prepare arguments for MCP tool
arguments := map[string]interface{}{
"image": base64Data,
"content_type": contentType,
}
// Send loading message
loadingID := h.sendLoading(ctx, i18n.T(ctx.Locale, "content.image.analyzing"))
// Call MCP tool (typically "describe_image" or similar)
result, err := tools.CallMCPTool(ctx, serverID, "describe_image", arguments)
// Send done message
h.sendLoadingDone(ctx, loadingID)
return result, err
}
// sendLoading sends a loading message and returns the message ID
func (h *Image) sendLoading(ctx *agentContext.Context, msg string) string {
loadingMsg := &message.Message{
Type: message.TypeLoading,
Props: map[string]interface{}{
"message": msg,
},
}
msgID, err := ctx.SendStream(loadingMsg)
if err != nil {
return ""
}
return msgID
}
// sendLoadingDone marks the loading message as done
func (h *Image) sendLoadingDone(ctx *agentContext.Context, loadingID string) {
if loadingID == "" {
return
}
doneMsg := &message.Message{
MessageID: loadingID,
Delta: true,
DeltaAction: message.DeltaReplace,
Type: message.TypeLoading,
Props: map[string]interface{}{
"done": true,
},
}
ctx.Send(doneMsg)
}
// EncodeToBase64DataURI encodes data to base64 with data URI prefix
func EncodeToBase64DataURI(data []byte, contentType string) string {
if contentType == "" {
contentType = "image/png" // default for images
}
encoded := base64.StdEncoding.EncodeToString(data)
return fmt.Sprintf("data:%s;base64,%s", contentType, encoded)
}

View file

@ -0,0 +1,374 @@
package image_test
import (
stdContext "context"
"encoding/base64"
"strings"
"testing"
"github.com/stretchr/testify/assert"
"github.com/yaoapp/gou/connector/openai"
"github.com/yaoapp/yao/agent/content/image"
contentTypes "github.com/yaoapp/yao/agent/content/types"
agentContext "github.com/yaoapp/yao/agent/context"
"github.com/yaoapp/yao/agent/output/message"
"github.com/yaoapp/yao/agent/testutils"
oauthTypes "github.com/yaoapp/yao/openapi/oauth/types"
)
// newTestContext creates a Context for testing with commonly used fields pre-populated
func newTestContext(capabilities *openai.Capabilities) *agentContext.Context {
authorized := &oauthTypes.AuthorizedInfo{
Subject: "test-user",
ClientID: "test-client-id",
UserID: "test-user-123",
TeamID: "test-team-456",
TenantID: "test-tenant-789",
}
ctx := agentContext.New(stdContext.Background(), authorized, "test-chat")
ctx.AssistantID = "test-assistant"
ctx.Locale = "en-us"
ctx.Theme = "light"
ctx.Client = agentContext.Client{
Type: "web",
UserAgent: "TestAgent/1.0",
IP: "127.0.0.1",
}
ctx.Referer = agentContext.RefererAPI
ctx.Accept = agentContext.AcceptWebCUI
ctx.Route = ""
ctx.Metadata = make(map[string]interface{})
ctx.Capabilities = capabilities
ctx.IDGenerator = message.NewIDGenerator()
return ctx
}
// newTestOptions creates test options with the given capabilities
func newTestOptions(capabilities *openai.Capabilities, completionOptions *agentContext.CompletionOptions) *contentTypes.Options {
return &contentTypes.Options{
Capabilities: capabilities,
CompletionOptions: completionOptions,
}
}
// TestParseWithVisionSupport tests parsing image when model supports vision
func TestParseWithVisionSupport(t *testing.T) {
if testing.Short() {
t.Skip("Skipping integration test")
}
testutils.Prepare(t)
defer testutils.Clean(t)
// Create capabilities with vision support
capabilities := &openai.Capabilities{
Vision: "openai",
}
options := newTestOptions(capabilities, nil)
ctx := newTestContext(capabilities)
// Create test image content with data URI
base64Data := "data:image/png;base64," + base64.StdEncoding.EncodeToString(createTestPNG())
content := agentContext.ContentPart{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: base64Data,
Detail: agentContext.DetailAuto,
},
}
handler := image.New(options)
result, refs, err := handler.Parse(ctx, content)
assert.NoError(t, err)
assert.Nil(t, refs)
assert.Equal(t, agentContext.ContentImageURL, result.Type)
assert.NotNil(t, result.ImageURL)
assert.Equal(t, base64Data, result.ImageURL.URL) // Should pass through unchanged
}
// TestParseWithoutVisionSupport tests parsing image when model doesn't support vision
func TestParseWithoutVisionSupport(t *testing.T) {
if testing.Short() {
t.Skip("Skipping integration test")
}
testutils.Prepare(t)
defer testutils.Clean(t)
// Create capabilities WITHOUT vision support
capabilities := &openai.Capabilities{
Vision: nil,
}
options := newTestOptions(capabilities, nil)
ctx := newTestContext(capabilities)
// Create test image content
base64Data := "data:image/png;base64," + base64.StdEncoding.EncodeToString(createTestPNG())
content := agentContext.ContentPart{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: base64Data,
Detail: agentContext.DetailAuto,
},
}
handler := image.New(options)
_, _, err := handler.Parse(ctx, content)
// Should return error because no vision support and no vision tool specified
assert.Error(t, err)
assert.Contains(t, err.Error(), "no vision tool specified")
}
// TestParseWithEmptyURL tests parsing image with empty URL
func TestParseWithEmptyURL(t *testing.T) {
if testing.Short() {
t.Skip("Skipping integration test")
}
testutils.Prepare(t)
defer testutils.Clean(t)
capabilities := &openai.Capabilities{
Vision: "openai",
}
options := newTestOptions(capabilities, nil)
ctx := newTestContext(capabilities)
// Create content with empty URL
content := agentContext.ContentPart{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: "",
},
}
handler := image.New(options)
_, _, err := handler.Parse(ctx, content)
assert.Error(t, err)
assert.Contains(t, err.Error(), "missing URL")
}
// TestParseWithNilImageURL tests parsing image with nil ImageURL
func TestParseWithNilImageURL(t *testing.T) {
if testing.Short() {
t.Skip("Skipping integration test")
}
testutils.Prepare(t)
defer testutils.Clean(t)
capabilities := &openai.Capabilities{
Vision: "openai",
}
options := newTestOptions(capabilities, nil)
ctx := newTestContext(capabilities)
// Create content with nil ImageURL
content := agentContext.ContentPart{
Type: agentContext.ContentImageURL,
ImageURL: nil,
}
handler := image.New(options)
_, _, err := handler.Parse(ctx, content)
assert.Error(t, err)
assert.Contains(t, err.Error(), "missing URL")
}
// TestEncodeToBase64DataURI tests base64 encoding
func TestEncodeToBase64DataURI(t *testing.T) {
tests := []struct {
name string
data []byte
contentType string
wantPrefix string
}{
{
name: "PNG image",
data: []byte{0x89, 0x50, 0x4E, 0x47},
contentType: "image/png",
wantPrefix: "data:image/png;base64,",
},
{
name: "JPEG image",
data: []byte{0xFF, 0xD8, 0xFF},
contentType: "image/jpeg",
wantPrefix: "data:image/jpeg;base64,",
},
{
name: "Empty content type defaults to PNG",
data: []byte{0x01, 0x02, 0x03},
contentType: "",
wantPrefix: "data:image/png;base64,",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
result := image.EncodeToBase64DataURI(tt.data, tt.contentType)
// Check prefix
assert.True(t, strings.HasPrefix(result, tt.wantPrefix))
// Verify base64 encoding by decoding
base64Part := result[len(tt.wantPrefix):]
decoded, err := base64.StdEncoding.DecodeString(base64Part)
assert.NoError(t, err)
// Verify decoded data matches original
assert.Equal(t, tt.data, decoded)
})
}
}
// TestParseDataURIPassthrough tests that data URI images pass through unchanged when vision is supported
func TestParseDataURIPassthrough(t *testing.T) {
if testing.Short() {
t.Skip("Skipping integration test")
}
testutils.Prepare(t)
defer testutils.Clean(t)
capabilities := &openai.Capabilities{
Vision: "openai",
}
options := newTestOptions(capabilities, nil)
ctx := newTestContext(capabilities)
// Create test image content with data URI
originalURL := "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mNk+M9QDwADhgGAWjR9awAAAABJRU5ErkJggg=="
content := agentContext.ContentPart{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: originalURL,
Detail: agentContext.DetailHigh,
},
}
handler := image.New(options)
result, refs, err := handler.Parse(ctx, content)
assert.NoError(t, err)
assert.Nil(t, refs)
assert.Equal(t, agentContext.ContentImageURL, result.Type)
assert.NotNil(t, result.ImageURL)
assert.Equal(t, originalURL, result.ImageURL.URL)
assert.Equal(t, agentContext.DetailHigh, result.ImageURL.Detail)
}
// TestParseWithVisionAgent tests parsing image using a vision agent when model doesn't support vision
func TestParseWithVisionAgent(t *testing.T) {
if testing.Short() {
t.Skip("Skipping integration test")
}
testutils.Prepare(t)
defer testutils.Clean(t)
// Create capabilities WITHOUT vision support
capabilities := &openai.Capabilities{
Vision: nil, // Model doesn't support vision
}
// Configure to use vision agent
completionOptions := &agentContext.CompletionOptions{
Uses: &agentContext.Uses{
Vision: "tests.vision-test", // Use our test vision agent
},
}
options := newTestOptions(capabilities, completionOptions)
ctx := newTestContext(capabilities)
// Create test image content with data URI (1x1 red PNG)
base64Data := "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8DwHwAFBQIAX8jx0gAAAABJRU5ErkJggg=="
content := agentContext.ContentPart{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: base64Data,
Detail: agentContext.DetailAuto,
},
}
handler := image.New(options)
result, refs, err := handler.Parse(ctx, content)
// Should succeed and return text content (image description from agent)
assert.NoError(t, err)
assert.Nil(t, refs)
assert.Equal(t, agentContext.ContentText, result.Type)
assert.NotEmpty(t, result.Text) // Agent should return some description
t.Logf("Vision agent response: %s", result.Text)
}
// TestParseWithForceUsesVisionAgent tests forceUses flag with vision agent
func TestParseWithForceUsesVisionAgent(t *testing.T) {
if testing.Short() {
t.Skip("Skipping integration test")
}
testutils.Prepare(t)
defer testutils.Clean(t)
// Create capabilities WITH vision support
capabilities := &openai.Capabilities{
Vision: "openai", // Model supports vision
}
// Configure to FORCE use vision agent (even though model supports vision)
completionOptions := &agentContext.CompletionOptions{
ForceUses: true, // Force using the vision tool
Uses: &agentContext.Uses{
Vision: "tests.vision-test", // Use our test vision agent
},
}
options := newTestOptions(capabilities, completionOptions)
ctx := newTestContext(capabilities)
// Create test image content with data URI
base64Data := "data:image/png;base64,iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8DwHwAFBQIAX8jx0gAAAABJRU5ErkJggg=="
content := agentContext.ContentPart{
Type: agentContext.ContentImageURL,
ImageURL: &agentContext.ImageURL{
URL: base64Data,
Detail: agentContext.DetailAuto,
},
}
handler := image.New(options)
result, refs, err := handler.Parse(ctx, content)
// Should succeed and return text content (forced to use agent even though model supports vision)
assert.NoError(t, err)
assert.Nil(t, refs)
assert.Equal(t, agentContext.ContentText, result.Type)
assert.NotEmpty(t, result.Text) // Agent should return some description
t.Logf("Vision agent (forced) response: %s", result.Text)
}
// createTestPNG creates a minimal valid PNG image (1x1 red pixel)
func createTestPNG() []byte {
return []byte{
0x89, 0x50, 0x4E, 0x47, 0x0D, 0x0A, 0x1A, 0x0A, // PNG signature
0x00, 0x00, 0x00, 0x0D, 0x49, 0x48, 0x44, 0x52, // IHDR chunk
0x00, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01, // 1x1 dimensions
0x08, 0x02, 0x00, 0x00, 0x00, 0x90, 0x77, 0x53,
0xDE, 0x00, 0x00, 0x00, 0x0C, 0x49, 0x44, 0x41, // IDAT chunk
0x54, 0x08, 0xD7, 0x63, 0xF8, 0xCF, 0xC0, 0x00,
0x00, 0x03, 0x01, 0x01, 0x00, 0x18, 0xDD, 0x8D,
0xB4, 0x00, 0x00, 0x00, 0x00, 0x49, 0x45, 0x4E, // IEND chunk
0x44, 0xAE, 0x42, 0x60, 0x82,
}
}

View file

@ -1,261 +0,0 @@
package content
import (
stdContext "context"
"encoding/base64"
"os"
"strings"
"testing"
"github.com/yaoapp/gou/connector/openai"
agentContext "github.com/yaoapp/yao/agent/context"
"github.com/yaoapp/yao/config"
"github.com/yaoapp/yao/openapi/oauth/types"
"github.com/yaoapp/yao/test"
)
func TestMain(m *testing.M) {
// Setup test environment
test.Prepare(nil, config.Conf)
defer test.Clean()
// Run tests
code := m.Run()
os.Exit(code)
}
// newTestContext creates a Context for testing with commonly used fields pre-populated
func newTestContext(capabilities *openai.Capabilities) *agentContext.Context {
authorized := &types.AuthorizedInfo{
Subject: "test-user",
ClientID: "test-client-id",
UserID: "test-user-123",
TeamID: "test-team-456",
TenantID: "test-tenant-789",
}
ctx := agentContext.New(stdContext.Background(), authorized, "test-chat")
ctx.AssistantID = "test-assistant"
ctx.Locale = "en-us"
ctx.Theme = "light"
ctx.Client = agentContext.Client{
Type: "web",
UserAgent: "TestAgent/1.0",
IP: "127.0.0.1",
}
ctx.Referer = agentContext.RefererAPI
ctx.Accept = agentContext.AcceptWebCUI
ctx.Route = ""
ctx.Metadata = make(map[string]interface{})
ctx.Capabilities = capabilities
return ctx
}
func TestImageHandler_CanHandle(t *testing.T) {
handler := &ImageHandler{}
tests := []struct {
name string
contentType string
fileType FileType
want bool
}{
{"PNG image", "image/png", FileTypeImage, true},
{"JPEG image", "image/jpeg", FileTypeImage, true},
{"GIF image", "image/gif", FileTypeImage, true},
{"WebP image", "image/webp", FileTypeImage, true},
{"Text (should not handle)", "text/plain", FileTypeText, false},
{"PDF (should not handle)", "application/pdf", FileTypePDF, false},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
got := handler.CanHandle(tt.contentType, tt.fileType)
if got != tt.want {
t.Errorf("CanHandle(%q, %q) = %v, want %v", tt.contentType, tt.fileType, got, tt.want)
}
})
}
}
func TestImageHandler_Handle_WithVisionSupport(t *testing.T) {
test.Prepare(t, config.Conf)
defer test.Clean()
handler := &ImageHandler{}
// Create a simple test image (1x1 red PNG)
pngData := createTestPNG()
// Create capabilities with vision support
capabilities := &openai.Capabilities{
Vision: "openai", // Vision is enabled with OpenAI format
}
// Create test context
ctx := newTestContext(capabilities)
info := &Info{
FileType: FileTypeImage,
ContentType: "image/png",
Data: pngData,
}
result, err := handler.Handle(ctx, info, capabilities, nil, false)
if err != nil {
t.Fatalf("Handle() error = %v", err)
}
if result == nil {
t.Fatal("Expected non-nil result")
}
if result.ContentPart == nil {
t.Fatal("Expected ContentPart for vision-supported model")
}
if result.ContentPart.Type != agentContext.ContentImageURL {
t.Errorf("Expected ContentPart type = %v, got %v", agentContext.ContentImageURL, result.ContentPart.Type)
}
if result.ContentPart.ImageURL == nil {
t.Fatal("Expected ImageURL to be set")
}
// Verify base64 encoding
if result.ContentPart.ImageURL.URL == "" {
t.Error("Expected non-empty URL")
}
// Should be data URI format
if len(result.ContentPart.ImageURL.URL) < 20 {
t.Error("Expected data URI to be longer")
}
}
func TestImageHandler_Handle_WithoutVisionSupport(t *testing.T) {
test.Prepare(t, config.Conf)
defer test.Clean()
handler := &ImageHandler{}
// Create a simple test image
pngData := createTestPNG()
// Create capabilities WITHOUT vision support
capabilities := &openai.Capabilities{
Vision: nil, // No vision support
}
// Create test context
ctx := newTestContext(capabilities)
info := &Info{
FileType: FileTypeImage,
ContentType: "image/png",
Data: pngData,
}
// Should return error because no vision support and no tool
_, err := handler.Handle(ctx, info, capabilities, nil, false)
if err == nil {
t.Error("Expected error when no vision support and no tool specified")
}
}
func TestImageHandler_Handle_EmptyData(t *testing.T) {
test.Prepare(t, config.Conf)
defer test.Clean()
handler := &ImageHandler{}
capabilities := &openai.Capabilities{
Vision: "openai",
}
// Create test context
ctx := newTestContext(capabilities)
info := &Info{
FileType: FileTypeImage,
ContentType: "image/png",
Data: []byte{}, // Empty data
}
_, err := handler.Handle(ctx, info, capabilities, nil, false)
if err == nil {
t.Error("Expected error for empty image data")
}
}
func TestEncodeImageBase64(t *testing.T) {
tests := []struct {
name string
data []byte
contentType string
wantPrefix string
}{
{
name: "PNG image",
data: []byte{0x89, 0x50, 0x4E, 0x47}, // PNG magic number
contentType: "image/png",
wantPrefix: "data:image/png;base64,",
},
{
name: "JPEG image",
data: []byte{0xFF, 0xD8, 0xFF}, // JPEG magic number
contentType: "image/jpeg",
wantPrefix: "data:image/jpeg;base64,",
},
{
name: "Empty content type defaults to PNG",
data: []byte{0x01, 0x02, 0x03},
contentType: "",
wantPrefix: "data:image/png;base64,",
},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
result := encodeImageBase64(tt.data, tt.contentType)
// Check prefix
if !strings.HasPrefix(result, tt.wantPrefix) {
t.Errorf("Expected prefix %q, got %q", tt.wantPrefix, result[:len(tt.wantPrefix)])
}
// Verify base64 encoding by decoding
base64Part := result[len(tt.wantPrefix):]
decoded, err := base64.StdEncoding.DecodeString(base64Part)
if err != nil {
t.Errorf("Failed to decode base64: %v", err)
}
// Verify decoded data matches original
if len(decoded) != len(tt.data) {
t.Errorf("Decoded length = %d, want %d", len(decoded), len(tt.data))
}
for i := range decoded {
if decoded[i] != tt.data[i] {
t.Errorf("Decoded byte[%d] = %x, want %x", i, decoded[i], tt.data[i])
}
}
})
}
}
// createTestPNG creates a minimal valid PNG image (1x1 red pixel)
func createTestPNG() []byte {
// This is a minimal valid 1x1 red PNG image
return []byte{
0x89, 0x50, 0x4E, 0x47, 0x0D, 0x0A, 0x1A, 0x0A, // PNG signature
0x00, 0x00, 0x00, 0x0D, 0x49, 0x48, 0x44, 0x52, // IHDR chunk
0x00, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01, // 1x1 dimensions
0x08, 0x02, 0x00, 0x00, 0x00, 0x90, 0x77, 0x53,
0xDE, 0x00, 0x00, 0x00, 0x0C, 0x49, 0x44, 0x41, // IDAT chunk
0x54, 0x08, 0xD7, 0x63, 0xF8, 0xCF, 0xC0, 0x00,
0x00, 0x03, 0x01, 0x01, 0x00, 0x18, 0xDD, 0x8D,
0xB4, 0x00, 0x00, 0x00, 0x00, 0x49, 0x45, 0x4E, // IEND chunk
0x44, 0xAE, 0x42, 0x60, 0x82,
}
}

View file

@ -1,26 +0,0 @@
package content
import (
"github.com/yaoapp/gou/connector/openai"
agentContext "github.com/yaoapp/yao/agent/context"
)
// Handler defines the interface for handling different content types
// Converts content (images, documents, etc.) to text or standard formats
type Handler interface {
// CanHandle checks if this handler can handle the given content type
CanHandle(contentType string, fileType FileType) bool
// Handle converts the content and returns processed result
// ctx: agent context (passed from Vision function)
// capabilities: model capabilities (for vision/audio support detection)
// uses: configuration for external tools (agents/MCP servers)
// forceUses: if true, force using Uses tools even when model has native capabilities
Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses, forceUses bool) (*Result, error)
}
// Fetcher defines the interface for fetching content from different sources
type Fetcher interface {
// Fetch retrieves content from a URL or file ID
Fetch(ctx *agentContext.Context, source Source, url string) (*Info, error)
}

View file

@ -0,0 +1 @@
package link

View file

@ -1,50 +0,0 @@
package content
import (
"fmt"
"strings"
"github.com/yaoapp/gou/connector/openai"
agentContext "github.com/yaoapp/yao/agent/context"
)
// PDFHandler handles PDF documents
type PDFHandler struct{}
// CanHandle checks if this handler can handle the content type
func (h *PDFHandler) CanHandle(contentType string, fileType FileType) bool {
return fileType == FileTypePDF ||
contentType == "application/pdf" ||
strings.Contains(contentType, "pdf")
}
// Handle processes PDF content
// Logic:
// 1. Check if uses.Vision is specified and supports PDF
// 2. If yes, use vision tool to handle PDF (images + text)
// 3. If no, extract text directly from PDF
func (h *PDFHandler) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses, forceUses bool) (*Result, error) {
// TODO: Implement PDF handling
// 1. Check if vision tool supports PDF
// 2. If yes:
// - Call vision tool to handle PDF (handles both text and images)
// 3. If no:
// - Extract text from PDF using default library
// 4. Return Result with extracted text
return nil, fmt.Errorf("not implemented")
}
// extractPDFText extracts text content from PDF
func extractPDFText(data []byte) (string, error) {
// TODO: Implement PDF text extraction
// Use a PDF library to extract text
// Consider preserving layout/structure
return "", fmt.Errorf("not implemented")
}
// handleWithVisionTool processes PDF using vision tool (for PDFs with images)
func handleWithVisionTool(ctx *agentContext.Context, data []byte, visionTool string) (string, error) {
// TODO: Implement vision tool PDF processing
// Some vision tools can handle PDF directly and extract both text and images
return "", fmt.Errorf("not implemented")
}

49
agent/content/pdf/pdf.go Normal file
View file

@ -0,0 +1,49 @@
package pdf
import (
"fmt"
"github.com/yaoapp/yao/agent/content/types"
agentContext "github.com/yaoapp/yao/agent/context"
searchTypes "github.com/yaoapp/yao/agent/search/types"
)
// PDF handlePDF content
type PDF struct {
options *types.Options
}
// New creates a new image handler
func New(options *types.Options) *PDF {
return &PDF{options: options}
}
// Parse parses pdf content
func (h *PDF) Parse(ctx *agentContext.Context, content agentContext.ContentPart) (agentContext.ContentPart, []*searchTypes.Reference, error) {
return content, nil, nil
}
// asImages converts pdf content to images then parse as image content
func (h *PDF) asImages(ctx *agentContext.Context, content agentContext.ContentPart) (agentContext.ContentPart, []*searchTypes.Reference, error) {
return agentContext.ContentPart{}, nil, nil
}
// base64 encodes image content to base64 ( for PDF support )
func (h *PDF) base64(ctx *agentContext.Context, content agentContext.ContentPart) (agentContext.ContentPart, []*searchTypes.Reference, error) {
return agentContext.ContentPart{}, nil, nil
}
// read image content from file uploader __uploader://fileid return agentContext.ContentPart
func (h *PDF) read(ctx *agentContext.Context, content agentContext.ContentPart) (agentContext.ContentPart, []*searchTypes.Reference, error) {
return agentContext.ContentPart{}, nil, fmt.Errorf("not implemented")
}
// read image content from file uploader __uploader://fileid return agentContext.ContentPart
func (h *PDF) readFromCache(ctx *agentContext.Context, content agentContext.ContentPart) (agentContext.ContentPart, []*searchTypes.Reference, error) {
return agentContext.ContentPart{}, nil, fmt.Errorf("not implemented")
}
// read image content from file uploader __uploader://fileid return base64 encoded string
func (h *PDF) readFromUploader(ctx *agentContext.Context, content agentContext.ContentPart) (string, error) {
return "", fmt.Errorf("not implemented")
}

View file

@ -1,47 +0,0 @@
package content
import (
"fmt"
"github.com/yaoapp/gou/connector/openai"
agentContext "github.com/yaoapp/yao/agent/context"
)
// Registry holds all registered content handlers
type Registry struct {
handlers []Handler
}
// NewRegistry creates a new handler registry with default handlers
func NewRegistry() *Registry {
return &Registry{
handlers: []Handler{
&ImageHandler{},
&AudioHandler{},
&PDFHandler{},
&WordHandler{},
&ExcelHandler{},
&TextHandler{},
},
}
}
// GetHandler finds the appropriate handler for the given content
func (r *Registry) GetHandler(contentType string, fileType FileType) Handler {
for _, handler := range r.handlers {
if handler.CanHandle(contentType, fileType) {
return handler
}
}
return nil
}
// Handle processes content using the appropriate handler
func (r *Registry) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses, forceUses bool) (*Result, error) {
handler := r.GetHandler(info.ContentType, info.FileType)
if handler == nil {
return nil, fmt.Errorf("no handler found for content type: %s, file type: %s", info.ContentType, info.FileType)
}
return handler.Handle(ctx, info, capabilities, uses, forceUses)
}

View file

@ -1,123 +0,0 @@
package content
import (
"encoding/json"
"fmt"
"strings"
"github.com/yaoapp/gou/connector/openai"
agentContext "github.com/yaoapp/yao/agent/context"
)
// TextHandler handles plain text, code files, CSV, JSON, XML, Markdown, etc.
type TextHandler struct{}
// CanHandle checks if this handler can handle the content type
func (h *TextHandler) CanHandle(contentType string, fileType FileType) bool {
// Handle explicit text file types
if fileType == FileTypeText || fileType == FileTypeCSV || fileType == FileTypeJSON {
return true
}
// Handle text-based MIME types
if strings.HasPrefix(contentType, "text/") {
return true
}
// Handle common text-based content types
textContentTypes := []string{
"application/json",
"application/xml",
"application/javascript",
"application/typescript",
"application/x-yaml",
"application/yaml",
"application/toml",
"application/x-sh",
"application/x-python",
"application/x-ruby",
"application/x-perl",
}
for _, ct := range textContentTypes {
if contentType == ct || strings.Contains(contentType, ct) {
return true
}
}
return false
}
// Handle processes text content
func (h *TextHandler) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses, forceUses bool) (*Result, error) {
if len(info.Data) == 0 {
return nil, fmt.Errorf("no data to process")
}
var text string
var err error
// Handle different text formats
switch {
case info.FileType == FileTypeCSV || strings.Contains(info.ContentType, "csv"):
// Format CSV as readable text (for now, just return as-is, can enhance later)
text = string(info.Data)
case info.FileType == FileTypeJSON ||
info.ContentType == "application/json" ||
strings.Contains(info.ContentType, "json"):
// Pretty print JSON
text, err = formatJSONAsText(info.Data)
if err != nil {
// If JSON parsing fails, return raw text
text = string(info.Data)
}
case info.ContentType == "application/xml" ||
strings.Contains(info.ContentType, "xml"):
// For now, return XML as-is (can enhance formatting later)
text = string(info.Data)
default:
// Plain text, code files, markdown, etc.
text, err = readTextContent(info.Data, info.ContentType)
if err != nil {
return nil, fmt.Errorf("failed to read text content: %w", err)
}
}
return &Result{
Text: text,
}, nil
}
// readTextContent reads text content from data
func readTextContent(data []byte, contentType string) (string, error) {
// For now, assume UTF-8 encoding
// TODO: Add encoding detection if needed (e.g., using golang.org/x/text/encoding)
return string(data), nil
}
// formatCSVAsText formats CSV data as readable text
func formatCSVAsText(data []byte) (string, error) {
// TODO: Parse CSV and format as readable table
// Consider using encoding/csv package
// For now, just return as-is
return string(data), nil
}
// formatJSONAsText formats JSON data as readable text
func formatJSONAsText(data []byte) (string, error) {
// Pretty print JSON with indentation
var obj interface{}
if err := json.Unmarshal(data, &obj); err != nil {
return "", err
}
pretty, err := json.MarshalIndent(obj, "", " ")
if err != nil {
return "", err
}
return string(pretty), nil
}

View file

@ -1,152 +0,0 @@
package content
import (
"testing"
)
func TestTextHandler_CanHandle(t *testing.T) {
handler := &TextHandler{}
tests := []struct {
name string
contentType string
fileType FileType
want bool
}{
{"Plain text", "text/plain", FileTypeText, true},
{"Markdown", "text/markdown", FileTypeText, true},
{"HTML", "text/html", FileTypeText, true},
{"JSON", "application/json", FileTypeJSON, true},
{"JavaScript", "application/javascript", FileTypeText, true},
{"TypeScript", "application/typescript", FileTypeText, true},
{"YAML", "application/yaml", FileTypeText, true},
{"CSV", "text/csv", FileTypeCSV, true},
{"XML", "application/xml", FileTypeText, true},
{"PDF (should not handle)", "application/pdf", FileTypePDF, false},
{"Image (should not handle)", "image/png", FileTypeImage, false},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
got := handler.CanHandle(tt.contentType, tt.fileType)
if got != tt.want {
t.Errorf("CanHandle(%q, %q) = %v, want %v", tt.contentType, tt.fileType, got, tt.want)
}
})
}
}
func TestTextHandler_Handle(t *testing.T) {
handler := &TextHandler{}
tests := []struct {
name string
info *Info
wantErr bool
checkResult func(*testing.T, *Result)
}{
{
name: "Plain text",
info: &Info{
FileType: FileTypeText,
ContentType: "text/plain",
Data: []byte("Hello, World!"),
},
wantErr: false,
checkResult: func(t *testing.T, r *Result) {
if r.Text != "Hello, World!" {
t.Errorf("Expected 'Hello, World!', got %q", r.Text)
}
},
},
{
name: "JSON with pretty print",
info: &Info{
FileType: FileTypeJSON,
ContentType: "application/json",
Data: []byte(`{"name":"test","value":123}`),
},
wantErr: false,
checkResult: func(t *testing.T, r *Result) {
// Should be pretty printed
if len(r.Text) <= len(`{"name":"test","value":123}`) {
t.Errorf("JSON should be pretty printed, got: %q", r.Text)
}
},
},
{
name: "Code file (Go)",
info: &Info{
FileType: FileTypeText,
ContentType: "text/plain",
Data: []byte("package main\n\nfunc main() {\n\tprintln(\"Hello\")\n}"),
},
wantErr: false,
checkResult: func(t *testing.T, r *Result) {
if r.Text == "" {
t.Error("Expected non-empty text for Go code")
}
},
},
{
name: "Empty data",
info: &Info{
FileType: FileTypeText,
ContentType: "text/plain",
Data: []byte{},
},
wantErr: true,
},
}
// Create test context
testCtx := newTestContext(nil)
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
result, err := handler.Handle(testCtx, tt.info, nil, nil, false)
if (err != nil) != tt.wantErr {
t.Errorf("Handle() error = %v, wantErr %v", err, tt.wantErr)
return
}
if !tt.wantErr && tt.checkResult != nil {
tt.checkResult(t, result)
}
})
}
}
func TestDetectFileType(t *testing.T) {
tests := []struct {
name string
contentType string
filename string
want FileType
}{
{"Go file", "text/plain", "main.go", FileTypeText},
{"Python file", "text/plain", "script.py", FileTypeText},
{"JavaScript file", "application/javascript", "app.js", FileTypeText},
{"TypeScript file", "text/plain", "index.ts", FileTypeText},
{"Markdown file", "text/markdown", "README.md", FileTypeText},
{"JSON file", "application/json", "config.json", FileTypeJSON},
{"YAML file", "text/plain", "config.yml", FileTypeText},
{"PDF file", "application/pdf", "document.pdf", FileTypePDF},
{"Image file", "image/png", "photo.png", FileTypeImage},
{"CSV file", "text/csv", "data.csv", FileTypeCSV},
{"XML file", "application/xml", "config.xml", FileTypeXML},
{"Shell script", "text/plain", "script.sh", FileTypeText},
{"Dockerfile", "text/plain", "Dockerfile", FileTypeText},
{"gitignore", "text/plain", ".gitignore", FileTypeText},
{"HTML", "text/html", "index.html", FileTypeText},
{"CSS", "text/css", "styles.css", FileTypeText},
}
for _, tt := range tests {
t.Run(tt.name, func(t *testing.T) {
got := DetectFileType(tt.contentType, tt.filename)
if got != tt.want {
t.Errorf("DetectFileType(%q, %q) = %v, want %v", tt.contentType, tt.filename, got, tt.want)
}
})
}
}

View file

@ -1,240 +0,0 @@
package content
import (
"context"
"encoding/base64"
"fmt"
"sync"
jsoniter "github.com/json-iterator/go"
"github.com/yaoapp/gou/mcp"
"github.com/yaoapp/kun/log"
"github.com/yaoapp/yao/agent/caller"
agentContext "github.com/yaoapp/yao/agent/context"
)
// fileInfoMutex protects concurrent access to files_info list in Space
var fileInfoMutex sync.Mutex
// CallAgent calls an agent to process content (vision, audio, etc.)
// This is a generic function that can be used by any handler
func CallAgent(ctx *agentContext.Context, agentID string, message agentContext.Message) (string, error) {
if caller.AgentGetterFunc == nil {
return "", fmt.Errorf("AgentGetterFunc not initialized")
}
// Load the agent by ID using the injected function
agent, err := caller.AgentGetterFunc(agentID)
if err != nil {
return "", fmt.Errorf("failed to load agent %s: %w", agentID, err)
}
// Call the agent with the message
messages := []agentContext.Message{message}
// Note: Connector is now in Options (call-level parameter), not Context
// For A2A calls, skip history and output (we only need the response data)
opts := &agentContext.Options{Skip: &agentContext.Skip{History: true, Output: true}} // Skip history and output
response, err := agent.Stream(ctx, messages, opts)
if err != nil {
return "", fmt.Errorf("failed to call agent %s: %w", agentID, err)
}
// Extract text from agent response
// Two formats are supported:
// 1. Custom Hook response (from Next hook) - response.Next
// 2. Standard Agent Stream response (LLM completion) - response.Completion
return extractTextFromResponse(response)
}
// CallAgentWithFileInfo calls an agent to process content with file metadata
// The file metadata is passed via ctx.Memory.Context for access by hooks (especially Next hook)
// Uses Memory.Context (request-scoped) to avoid creating context copies and ensure proper cleanup
//
// Memory Keys (with agent ID as namespace prefix to avoid conflicts between different agents):
// - {agentID}:files_info - List of all files being processed by this agent (array)
// - {agentID}:current_file - Currently processing file (single object)
func CallAgentWithFileInfo(ctx *agentContext.Context, agentID string, message agentContext.Message, info *Info) (string, error) {
// Store file information in Memory.Context if available
if info != nil && ctx.Memory != nil && ctx.Memory.Context != nil {
fileInfo := map[string]interface{}{
"url": info.URL,
"filename": info.Filename,
"content_type": info.ContentType,
"file_type": string(info.FileType),
"source": string(info.Source),
}
// Add uploader-specific information if available
if info.UploaderName != "" {
fileInfo["uploader_name"] = info.UploaderName
}
if info.FileID != "" {
fileInfo["file_id"] = info.FileID
}
// Use agent ID as namespace prefix for Memory keys
filesListKey := agentID + ":files_info"
currentFileKey := agentID + ":current_file"
// Thread-safe: append current file to files list
fileInfoMutex.Lock()
var filesList []map[string]interface{}
if existing, ok := ctx.Memory.Context.Get(filesListKey); ok {
// Convert existing data to []map[string]interface{}
if existingList, ok := existing.([]interface{}); ok {
for _, item := range existingList {
if itemMap, ok := item.(map[string]interface{}); ok {
filesList = append(filesList, itemMap)
}
}
} else if existingList, ok := existing.([]map[string]interface{}); ok {
filesList = existingList
}
}
// Append current file to list
filesList = append(filesList, fileInfo)
ctx.Memory.Context.Set(filesListKey, filesList, 0)
fileInfoMutex.Unlock()
// Store current file in Memory.Context
if err := ctx.Memory.Context.Set(currentFileKey, fileInfo, 0); err != nil {
log.Trace("[Content] Failed to set current file info in Memory.Context: %v", err)
}
// Ensure cleanup after agent call completes
defer func() {
// Clean up current file
if err := ctx.Memory.Context.Del(currentFileKey); err != nil {
log.Trace("[Content] Failed to delete current file info from Memory.Context: %v", err)
}
// Clean up files list (reset for next call)
if err := ctx.Memory.Context.Del(filesListKey); err != nil {
log.Trace("[Content] Failed to delete files list from Memory.Context: %v", err)
}
}()
}
// Call the agent with the original context
return CallAgent(ctx, agentID, message)
}
// extractTextFromResponse extracts text from agent response
// Now that agent.Stream() returns *agentContext.Response directly,
// we can access fields without type assertions or JSON conversion.
//
// Priority:
// 1. Check response.Next (custom hook data) → return complete data
// 2. Check response.Completion (standard LLM response) → extract text only
func extractTextFromResponse(response *agentContext.Response) (string, error) {
if response == nil {
return "", fmt.Errorf("agent returned nil response")
}
// Priority 1: Check Next field (custom hook data)
// If Next hook returns custom data, return the complete structure
if response.Next != nil {
// If next is a string, return directly
if nextStr, ok := response.Next.(string); ok {
return nextStr, nil
}
// Otherwise, JSON stringify to preserve complete structure
jsonBytes, err := jsoniter.Marshal(response.Next)
if err != nil {
return "", fmt.Errorf("failed to serialize next hook data: %w", err)
}
return string(jsonBytes), nil
}
// Priority 2: Check Completion field (standard LLM response)
// Extract text content from the LLM completion
if response.Completion != nil {
// Content can be string or []ContentPart (multimodal)
switch v := response.Completion.Content.(type) {
case string:
// Simple text content
return v, nil
case []interface{}:
// Multimodal content array - extract all text parts
var text string
for _, part := range v {
if partMap, ok := part.(map[string]interface{}); ok {
if partType, _ := partMap["type"].(string); partType == "text" {
if textContent, ok := partMap["text"].(string); ok {
text += textContent
}
}
}
}
if text != "" {
return text, nil
}
// No text found in content parts
return "", fmt.Errorf("no text content found in completion content parts")
}
}
// No content found
return "", fmt.Errorf("no content found in agent response")
}
// CallMCPTool calls an MCP tool to process content
// This is a generic function that can be used by any handler
func CallMCPTool(ctx *agentContext.Context, serverID string, toolName string, arguments map[string]interface{}) (string, error) {
// Get MCP context for cancellation/timeout control
mcpCtx := ctx.Context
if mcpCtx == nil {
mcpCtx = context.Background()
}
// Get MCP client
client, err := mcp.Select(serverID)
if err != nil {
return "", fmt.Errorf("failed to select MCP client '%s': %w", serverID, err)
}
// Call the tool
log.Trace("[Content] Calling MCP tool: %s (server: %s)", toolName, serverID)
callResult, err := client.CallTool(mcpCtx, toolName, arguments)
if err != nil {
return "", fmt.Errorf("MCP tool call failed: %w", err)
}
// Check if result is an error
if callResult.IsError {
return "", fmt.Errorf("MCP tool returned error: %v", callResult.Content)
}
// Extract text content from result
// callResult.Content is []ToolContent
var text string
for _, content := range callResult.Content {
if content.Type == "text" {
text += content.Text
}
// Can also handle other types like image, resource if needed
}
if text == "" {
// If no text content found, return error
return "", fmt.Errorf("MCP tool returned no text content")
}
return text, nil
}
// EncodeToBase64DataURI encodes data to base64 with data URI prefix
// This is useful for encoding images, audio, or other binary data
func EncodeToBase64DataURI(data []byte, contentType string) string {
// Ensure we have a valid content type
if contentType == "" {
contentType = "application/octet-stream" // default
}
// Encode to base64
encoded := base64.StdEncoding.EncodeToString(data)
// Return data URI format
return fmt.Sprintf("data:%s;base64,%s", contentType, encoded)
}

View file

@ -0,0 +1,125 @@
package tools
import (
"context"
"fmt"
jsoniter "github.com/json-iterator/go"
"github.com/yaoapp/gou/mcp"
"github.com/yaoapp/kun/log"
"github.com/yaoapp/yao/agent/caller"
agentContext "github.com/yaoapp/yao/agent/context"
)
// CallAgent calls an agent to process content (vision, audio, etc.)
func CallAgent(ctx *agentContext.Context, agentID string, message agentContext.Message) (string, error) {
if caller.AgentGetterFunc == nil {
return "", fmt.Errorf("AgentGetterFunc not initialized")
}
// Load the agent by ID using the injected function
agent, err := caller.AgentGetterFunc(agentID)
if err != nil {
return "", fmt.Errorf("failed to load agent %s: %w", agentID, err)
}
// Call the agent with the message
messages := []agentContext.Message{message}
// For A2A calls, skip history and output (we only need the response data)
opts := &agentContext.Options{Skip: &agentContext.Skip{History: true, Output: true}}
response, err := agent.Stream(ctx, messages, opts)
if err != nil {
return "", fmt.Errorf("failed to call agent %s: %w", agentID, err)
}
// Extract text from agent response
return ExtractTextFromResponse(response)
}
// ExtractTextFromResponse extracts text from agent response
func ExtractTextFromResponse(response *agentContext.Response) (string, error) {
if response == nil {
return "", fmt.Errorf("agent returned nil response")
}
// Priority 1: Check Next field (custom hook data)
if response.Next != nil {
if nextStr, ok := response.Next.(string); ok {
return nextStr, nil
}
// Otherwise, JSON stringify to preserve complete structure
jsonBytes, err := jsoniter.Marshal(response.Next)
if err != nil {
return "", fmt.Errorf("failed to serialize next hook data: %w", err)
}
return string(jsonBytes), nil
}
// Priority 2: Check Completion field (standard LLM response)
if response.Completion != nil {
switch v := response.Completion.Content.(type) {
case string:
return v, nil
case []interface{}:
// Multimodal content array - extract all text parts
var text string
for _, part := range v {
if partMap, ok := part.(map[string]interface{}); ok {
if partType, _ := partMap["type"].(string); partType == "text" {
if textContent, ok := partMap["text"].(string); ok {
text += textContent
}
}
}
}
if text != "" {
return text, nil
}
return "", fmt.Errorf("no text content found in completion content parts")
}
}
return "", fmt.Errorf("no content found in agent response")
}
// CallMCPTool calls an MCP tool to process content
func CallMCPTool(ctx *agentContext.Context, serverID string, toolName string, arguments map[string]interface{}) (string, error) {
// Get MCP context for cancellation/timeout control
mcpCtx := ctx.Context
if mcpCtx == nil {
mcpCtx = context.Background()
}
// Get MCP client
client, err := mcp.Select(serverID)
if err != nil {
return "", fmt.Errorf("failed to select MCP client '%s': %w", serverID, err)
}
// Call the tool
log.Trace("[Content] Calling MCP tool: %s (server: %s)", toolName, serverID)
callResult, err := client.CallTool(mcpCtx, toolName, arguments)
if err != nil {
return "", fmt.Errorf("MCP tool call failed: %w", err)
}
// Check if result is an error
if callResult.IsError {
return "", fmt.Errorf("MCP tool returned error: %v", callResult.Content)
}
// Extract text content from result
var text string
for _, content := range callResult.Content {
if content.Type == "text" {
text += content.Text
}
}
if text == "" {
return "", fmt.Errorf("MCP tool returned no text content")
}
return text, nil
}

View file

@ -1,262 +0,0 @@
package content
import "github.com/yaoapp/yao/agent/context"
// FileType represents the type of file content
type FileType string
const (
// Image types
FileTypeImage FileType = "image"
// Audio types
FileTypeAudio FileType = "audio"
// Document types
FileTypeText FileType = "text"
FileTypePDF FileType = "pdf"
FileTypeWord FileType = "word"
FileTypeExcel FileType = "excel"
FileTypePPT FileType = "ppt"
FileTypeCSV FileType = "csv"
// Data types
FileTypeJSON FileType = "json"
FileTypeXML FileType = "xml"
// Binary
FileTypeBinary FileType = "binary"
// Other
FileTypeUnknown FileType = "unknown"
)
// Source represents where the content comes from
type Source string
const (
SourceHTTP Source = "http" // HTTP(S) URL
SourceUploader Source = "uploader" // Uploader wrapper: __uploader://fileid
SourceBase64 Source = "base64" // Base64 encoded data
SourceLocal Source = "local" // Local file path
)
// Result represents the result of content handling
type Result struct {
Text string // Extracted text content
ContentPart *context.ContentPart // Processed ContentPart (for model input)
Metadata map[string]interface{} // Additional metadata
Error error // Error if handling failed
}
// Info holds information about a content part to be handled
type Info struct {
Source Source // Where the content comes from
FileType FileType // Type of the file
ContentType string // MIME content type
URL string // Original URL or file ID
Filename string // Original filename (if available)
Data []byte // File data (if already fetched)
// For uploader wrapper
UploaderName string // Uploader name from wrapper
FileID string // File ID from wrapper
}
// DetectFileType detects file type from content type, filename, and file extension
func DetectFileType(contentType, filename string) FileType {
// Check by content type first
switch {
case contentType == "application/pdf":
return FileTypePDF
case contentType == "application/json":
return FileTypeJSON
case contentType == "application/xml" || contentType == "text/xml":
return FileTypeXML
case contentType == "text/csv":
return FileTypeCSV
case contentType == "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
contentType == "application/msword":
return FileTypeWord
case contentType == "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
contentType == "application/vnd.ms-excel":
return FileTypeExcel
case contentType == "application/vnd.openxmlformats-officedocument.presentationml.presentation",
contentType == "application/vnd.ms-powerpoint":
return FileTypePPT
}
// Check image types
if isImageContentType(contentType) {
return FileTypeImage
}
// Check audio types
if isAudioContentType(contentType) {
return FileTypeAudio
}
// Check text types
if isTextContentType(contentType) {
return FileTypeText
}
// If content type doesn't help, check file extension
if filename != "" {
if ext := getFileExtension(filename); ext != "" {
return detectTypeByExtension(ext)
}
}
return FileTypeUnknown
}
// isImageContentType checks if content type is an image
func isImageContentType(contentType string) bool {
return contentType != "" &&
(contentType == "image/png" ||
contentType == "image/jpeg" ||
contentType == "image/jpg" ||
contentType == "image/gif" ||
contentType == "image/webp" ||
contentType == "image/svg+xml" ||
contentType == "image/bmp")
}
// isAudioContentType checks if content type is audio
func isAudioContentType(contentType string) bool {
return contentType != "" &&
(contentType == "audio/mpeg" ||
contentType == "audio/mp3" ||
contentType == "audio/wav" ||
contentType == "audio/ogg" ||
contentType == "audio/flac" ||
contentType == "audio/aac")
}
// isTextContentType checks if content type is text-based
func isTextContentType(contentType string) bool {
if contentType == "" {
return false
}
// Common text MIME types
textTypes := []string{
"text/plain",
"text/html",
"text/css",
"text/javascript",
"text/markdown",
"text/x-markdown",
"application/javascript",
"application/typescript",
"application/x-yaml",
"application/yaml",
"application/toml",
"application/x-sh",
"application/x-python",
"application/x-ruby",
"application/x-perl",
"application/x-php",
"application/x-go",
}
for _, t := range textTypes {
if contentType == t {
return true
}
}
return false
}
// getFileExtension extracts file extension from filename (without dot)
func getFileExtension(filename string) string {
for i := len(filename) - 1; i >= 0; i-- {
if filename[i] == '.' {
return filename[i+1:]
}
if filename[i] == '/' || filename[i] == '\\' {
break
}
}
return ""
}
// detectTypeByExtension detects file type by file extension
func detectTypeByExtension(ext string) FileType {
// Normalize to lowercase
ext = toLower(ext)
// Image extensions
imageExts := []string{"png", "jpg", "jpeg", "gif", "webp", "svg", "bmp", "ico"}
for _, e := range imageExts {
if ext == e {
return FileTypeImage
}
}
// Audio extensions
audioExts := []string{"mp3", "wav", "ogg", "flac", "aac", "m4a"}
for _, e := range audioExts {
if ext == e {
return FileTypeAudio
}
}
// Document extensions
switch ext {
case "pdf":
return FileTypePDF
case "doc", "docx":
return FileTypeWord
case "xls", "xlsx":
return FileTypeExcel
case "ppt", "pptx":
return FileTypePPT
case "csv":
return FileTypeCSV
case "json":
return FileTypeJSON
case "xml":
return FileTypeXML
}
// Code and text file extensions (very comprehensive list)
textExts := []string{
"txt", "text", "md", "markdown", "rst",
// Programming languages
"go", "py", "js", "ts", "jsx", "tsx", "java", "c", "cpp", "h", "hpp",
"cs", "rb", "php", "pl", "swift", "kt", "rs", "scala", "clj",
// Web
"html", "htm", "css", "scss", "sass", "less",
// Config
"yaml", "yml", "toml", "ini", "conf", "config",
// Shell
"sh", "bash", "zsh", "fish",
// Data
"sql", "graphql", "proto",
// Others
"log", "gitignore", "env", "dockerfile",
}
for _, e := range textExts {
if ext == e {
return FileTypeText
}
}
return FileTypeUnknown
}
// toLower converts ASCII string to lowercase (simple version)
func toLower(s string) string {
result := make([]byte, len(s))
for i := 0; i < len(s); i++ {
c := s[i]
if c >= 'A' && c <= 'Z' {
c += 'a' - 'A'
}
result[i] = c
}
return string(result)
}

View file

@ -0,0 +1,23 @@
package types
import (
"github.com/yaoapp/gou/connector"
"github.com/yaoapp/gou/connector/openai"
agentContext "github.com/yaoapp/yao/agent/context"
)
// Options represents the options for the content
type Options struct {
// Connector, Current connector instance
Connector connector.Connector
// Capabilities, Current capabilities instance
Capabilities *openai.Capabilities
// CompletionOptions, Current completion options instance
CompletionOptions *agentContext.CompletionOptions
// StreamOptions, Current stream options instance
StreamOptions *agentContext.StreamOptions
}

View file

@ -1,39 +0,0 @@
package content
import (
"fmt"
"strings"
"github.com/yaoapp/gou/connector/openai"
agentContext "github.com/yaoapp/yao/agent/context"
)
// WordHandler handles Microsoft Word documents
type WordHandler struct{}
// CanHandle checks if this handler can handle the content type
func (h *WordHandler) CanHandle(contentType string, fileType FileType) bool {
return fileType == FileTypeWord ||
contentType == "application/vnd.openxmlformats-officedocument.wordprocessingml.document" ||
contentType == "application/msword" ||
strings.Contains(contentType, "word")
}
// Handle processes Word document content
func (h *WordHandler) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses, forceUses bool) (*Result, error) {
// TODO: Implement Word document handling
// 1. Extract text from .docx or .doc file
// 2. Preserve formatting information if needed
// 3. Return Result with extracted text
return nil, fmt.Errorf("not implemented")
}
// extractWordText extracts text from Word document
func extractWordText(data []byte, contentType string) (string, error) {
// TODO: Implement Word text extraction
// Handle both .doc (old format) and .docx (new format)
// Consider using libraries like:
// - github.com/unidoc/unioffice for .docx
// - Other libraries for .doc
return "", fmt.Errorf("not implemented")
}

View file

@ -302,6 +302,9 @@ type Options struct {
// Agent mode, use to select the mode of the request, default is "chat"
Mode string `json:"mode,omitempty"` // Agent mode, use to select the mode of the request, default is "chat"
// Uses configuration, allow hook to override wrapper configurations for vision, audio, search, and fetch
Uses *Uses `json:"uses,omitempty"` // Uses configuration, allow hook to override wrapper configurations for vision, audio, search, and fetch
// Metadata for passing custom data to hooks (e.g., scenario selection)
Metadata map[string]any `json:"metadata,omitempty"` // Custom metadata passed to Create/Next hooks
}

View file

@ -99,6 +99,9 @@ func init() {
"kb.chat.name": "Chat Knowledge Base",
"kb.chat.description": "Auto-created knowledge base collection for chat sessions",
// Content: content/image/image.go - Image processing messages
"content.image.analyzing": "Analyzing image...",
// Search: assistant/search.go - Output messages
"search.loading": "Searching...",
"search.success": "Found %d references",
@ -192,6 +195,9 @@ func init() {
"kb.chat.name": "聊天知识库",
"kb.chat.description": "自动为聊天会话创建的知识库集合",
// Content: content/image/image.go - Image processing messages
"content.image.analyzing": "正在分析图片...",
// Search: assistant/search.go - Output messages
"search.loading": "正在搜索...",
"search.success": "找到 %d 条参考资料",
@ -313,6 +319,9 @@ func init() {
"kb.chat.name": "聊天知识库",
"kb.chat.description": "自动为聊天会话创建的知识库集合",
// Content: content/image/image.go - Image processing messages
"content.image.analyzing": "正在分析图片...",
// Search: assistant/search.go - Output messages
"search.loading": "正在搜索...",
"search.success": "找到 %d 条参考资料",