Enhance Assistant functionality with improved content handling and debugging
- Added support for converting extended types (file, data) to standard LLM types (text, image_url, input_audio) in the Stream method. - Introduced a new agentCallerWrapper to facilitate agent calls from the content package. - Implemented ThreadID management for nested agent calls to improve concurrent stream identification. - Enhanced message handling to include metadata for message_start and message_end events, allowing for better tracking of message states. - Removed deprecated vision capability checks from the Assistant initialization process, streamlining the codebase. - Added new content types (file, data) to the context types for improved message content handling.
This commit is contained in:
parent
1060c0f0e0
commit
beb63b77de
25 changed files with 2922 additions and 34 deletions
|
|
@ -8,6 +8,7 @@ import (
|
|||
"github.com/yaoapp/gou/connector"
|
||||
"github.com/yaoapp/gou/connector/openai"
|
||||
"github.com/yaoapp/kun/log"
|
||||
"github.com/yaoapp/kun/utils"
|
||||
"github.com/yaoapp/yao/agent/assistant/handlers"
|
||||
"github.com/yaoapp/yao/agent/context"
|
||||
"github.com/yaoapp/yao/agent/i18n"
|
||||
|
|
@ -42,6 +43,13 @@ func (ast *Assistant) Stream(ctx *context.Context, inputMessages []context.Messa
|
|||
_, _, done := context.EnterStack(ctx, ast.ID, ctx.Referer)
|
||||
defer done()
|
||||
|
||||
fmt.Println("--- Stack debug ---")
|
||||
if ctx.Stack != nil {
|
||||
fmt.Println(ctx.Stack.IsRoot())
|
||||
utils.Dump(ctx.Stack)
|
||||
}
|
||||
fmt.Println("------ end stack debug ------")
|
||||
|
||||
// Determine stream handler
|
||||
streamHandler := ast.getStreamHandler(ctx, handler...)
|
||||
|
||||
|
|
@ -106,6 +114,14 @@ func (ast *Assistant) Stream(ctx *context.Context, inputMessages []context.Messa
|
|||
return nil, err
|
||||
}
|
||||
|
||||
// Build content - convert extended types (file, data) to standard LLM types (text, image_url, input_audio)
|
||||
completionMessages, err = ast.BuildContent(ctx, completionMessages, completionOptions)
|
||||
if err != nil {
|
||||
ast.traceAgentFail(agentNode, err)
|
||||
ast.sendStreamEndOnError(ctx, streamHandler, streamStartTime, err)
|
||||
return nil, err
|
||||
}
|
||||
|
||||
// Execute the LLM streaming call
|
||||
completionResponse, err = ast.executeLLMStream(ctx, completionMessages, completionOptions, agentNode, streamHandler)
|
||||
if err != nil {
|
||||
|
|
|
|||
|
|
@ -5,11 +5,34 @@ import (
|
|||
"path"
|
||||
|
||||
"github.com/yaoapp/gou/fs"
|
||||
"github.com/yaoapp/yao/agent/content"
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
"github.com/yaoapp/yao/agent/i18n"
|
||||
store "github.com/yaoapp/yao/agent/store/types"
|
||||
sui "github.com/yaoapp/yao/sui/core"
|
||||
)
|
||||
|
||||
func init() {
|
||||
// Initialize AgentGetterFunc to allow content package to call agents
|
||||
content.AgentGetterFunc = func(agentID string) (content.AgentCaller, error) {
|
||||
ast, err := Get(agentID)
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
// Return a wrapper that implements AgentCaller interface
|
||||
return &agentCallerWrapper{ast: ast}, nil
|
||||
}
|
||||
}
|
||||
|
||||
// agentCallerWrapper wraps Assistant to implement AgentCaller interface
|
||||
type agentCallerWrapper struct {
|
||||
ast *Assistant
|
||||
}
|
||||
|
||||
func (w *agentCallerWrapper) Stream(ctx *agentContext.Context, messages []agentContext.Message) (interface{}, error) {
|
||||
return w.ast.Stream(ctx, messages)
|
||||
}
|
||||
|
||||
// Get get the assistant by id
|
||||
func Get(id string) (*Assistant, error) {
|
||||
return LoadStore(id)
|
||||
|
|
|
|||
31
agent/assistant/build_content.go
Normal file
31
agent/assistant/build_content.go
Normal file
|
|
@ -0,0 +1,31 @@
|
|||
package assistant
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"github.com/yaoapp/yao/agent/content"
|
||||
"github.com/yaoapp/yao/agent/context"
|
||||
)
|
||||
|
||||
// BuildContent processes messages through Vision function to convert extended content types
|
||||
// (file, data) to standard LLM-compatible types (text, image_url, input_audio)
|
||||
//
|
||||
// This should be called after BuildRequest and before executing LLM call
|
||||
func (ast *Assistant) BuildContent(ctx *context.Context, messages []context.Message, options *context.CompletionOptions) ([]context.Message, error) {
|
||||
// Get connector and capabilities
|
||||
_, capabilities, err := ast.GetConnector(ctx)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to get connector: %w", err)
|
||||
}
|
||||
|
||||
// Get Uses configuration from options (already merged in BuildRequest)
|
||||
uses := options.Uses
|
||||
|
||||
// Process content through Vision function
|
||||
processedMessages, err := content.Vision(ctx, capabilities, messages, uses)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to process content: %w", err)
|
||||
}
|
||||
|
||||
return processedMessages, nil
|
||||
}
|
||||
|
|
@ -107,6 +107,11 @@ func (s *streamState) handleMessageStart(data []byte) int {
|
|||
startData.MessageID = messageID
|
||||
}
|
||||
|
||||
// Auto-set ThreadID from Stack for nested agent calls
|
||||
if startData.ThreadID == "" && s.ctx.Stack != nil && !s.ctx.Stack.IsRoot() {
|
||||
startData.ThreadID = s.ctx.Stack.ID
|
||||
}
|
||||
|
||||
// Initialize message state with the correct message ID
|
||||
s.inGroup = true
|
||||
s.currentGroupID = messageID
|
||||
|
|
@ -312,11 +317,18 @@ func (s *streamState) handleMessageEnd(data []byte) int {
|
|||
msgType = message.TypeText // Fallback to text if type not set
|
||||
}
|
||||
|
||||
// Get ThreadID from Stack for nested agent calls
|
||||
var threadID string
|
||||
if s.ctx.Stack != nil && !s.ctx.Stack.IsRoot() {
|
||||
threadID = s.ctx.Stack.ID
|
||||
}
|
||||
|
||||
// Build EventMessageEndData with complete content
|
||||
endData := message.EventMessageEndData{
|
||||
MessageID: s.currentGroupID, // Use the message ID
|
||||
Type: msgType,
|
||||
Timestamp: time.Now().UnixMilli(),
|
||||
ThreadID: threadID, // Include ThreadID for concurrent stream identification
|
||||
DurationMs: durationMs,
|
||||
ChunkCount: s.chunkCount,
|
||||
Status: "completed",
|
||||
|
|
|
|||
|
|
@ -751,15 +751,6 @@ func (ast *Assistant) initialize() error {
|
|||
}
|
||||
ast.openai = api
|
||||
|
||||
// Check if the assistant supports vision
|
||||
model := api.Model()
|
||||
if v, ok := ast.Options["model"].(string); ok {
|
||||
model = strings.TrimLeft(v, "moapi:")
|
||||
}
|
||||
if _, ok := VisionCapableModels[model]; ok {
|
||||
ast.vision = true
|
||||
}
|
||||
|
||||
// Check if the assistant has an init hook
|
||||
if ast.Script != nil {
|
||||
scriptCtx, err := ast.Script.NewContext("", nil)
|
||||
|
|
|
|||
|
|
@ -40,31 +40,6 @@ type Assistant struct {
|
|||
// toolCalls bool // Whether this assistant supports tool_calls
|
||||
}
|
||||
|
||||
// VisionCapableModels list of LLM models that support vision capabilities
|
||||
var VisionCapableModels = map[string]bool{
|
||||
// OpenAI Models
|
||||
"gpt-4-vision-preview": true,
|
||||
"gpt-4v": true, // Alias for gpt-4-vision-preview
|
||||
|
||||
// Anthropic Models
|
||||
"claude-3-opus": true, // Most capable Claude model
|
||||
"claude-3-sonnet": true, // Balanced Claude model
|
||||
"claude-3-haiku": true, // Fast and efficient Claude model
|
||||
|
||||
// Google Models
|
||||
"gemini-pro-vision": true,
|
||||
|
||||
// Open Source Models
|
||||
"llava-13b": true,
|
||||
"cogvlm": true,
|
||||
"qwen-vl": true,
|
||||
"yi-vl": true,
|
||||
|
||||
// Custom Models
|
||||
"gpt-4o": true, // Custom OpenAI compatible model
|
||||
"gpt-4o-mini": true, // Custom OpenAI compatible model - mini version
|
||||
}
|
||||
|
||||
// MCPTool represents a simplified MCP tool for building LLM requests
|
||||
// This is an internal representation used when collecting tools from MCP servers
|
||||
// and preparing them for the LLM's tool calling interface
|
||||
|
|
|
|||
326
agent/content/README.md
Normal file
326
agent/content/README.md
Normal file
|
|
@ -0,0 +1,326 @@
|
|||
# Content Processing Package
|
||||
|
||||
This package handles content transformation for multimodal messages in agent conversations. It is called **BEFORE** sending messages to the LLM and converts extended content types into standard LLM-compatible formats.
|
||||
|
||||
## ⚠️ Critical Design Principle
|
||||
|
||||
**Input**: Messages with extended content types (`file`, `data`, etc.)
|
||||
**Output**: Messages with ONLY standard LLM-compatible types (`text`, `image_url`, `input_audio`)
|
||||
|
||||
The LLM should NEVER receive `type="file"` or `type="data"` content parts. These MUST be converted to `text` (or `image_url` for images if model supports vision).
|
||||
|
||||
## Architecture
|
||||
|
||||
```
|
||||
Vision (main entry)
|
||||
↓
|
||||
Initialize processedFiles cache (map[fileID]text)
|
||||
↓
|
||||
processMessage (for each message)
|
||||
↓
|
||||
processContentPart (for each content part)
|
||||
↓
|
||||
Is uploader wrapper?
|
||||
├── Yes → Check cache
|
||||
│ ├── In cache? → Use cached text ✅
|
||||
│ └── Not in cache → Try GetText(fileID) preview
|
||||
│ ├── Has preview? → Use preview + cache ✅
|
||||
│ └── No preview → Proceed to full processing ↓
|
||||
└── No (HTTP/other) → Proceed to full processing ↓
|
||||
↓
|
||||
├── Fetch content (if needed)
|
||||
│ ├── HTTP URL
|
||||
│ └── Uploader Wrapper (__uploader://fileid)
|
||||
↓
|
||||
├── Determine Processing Strategy
|
||||
│ ├── Model supports? → Format for model
|
||||
│ └── Model doesn't support? → Use agent/MCP
|
||||
↓
|
||||
ProcessorRegistry
|
||||
↓
|
||||
├── ImageProcessor
|
||||
├── AudioProcessor
|
||||
├── PDFProcessor
|
||||
├── WordProcessor
|
||||
├── ExcelProcessor
|
||||
└── TextProcessor
|
||||
↓
|
||||
Cache result (if uploader wrapper)
|
||||
```
|
||||
|
||||
## Content Type Transformation
|
||||
|
||||
### Input → Output Mapping
|
||||
|
||||
| Input Type | Model Supports? | Output Type | Processing |
|
||||
|------------|-----------------|-------------|------------|
|
||||
| `text` | - | `text` | Pass through |
|
||||
| `image_url` | ✅ Yes | `image_url` | Convert format if needed (base64/URL) |
|
||||
| `image_url` | ❌ No | `text` | Use vision agent/MCP to describe |
|
||||
| `input_audio` | ✅ Yes | `input_audio` | Keep as audio |
|
||||
| `input_audio` | ❌ No | `text` | Transcribe using audio agent/MCP |
|
||||
| `file` (image) | ✅ Yes | `image_url` | Same as image_url processing |
|
||||
| `file` (image) | ❌ No | `text` | Use vision tool to describe |
|
||||
| `file` (document) | - | `text` | Extract text from PDF/Word/Excel/etc |
|
||||
| `data` | - | `text` | Fetch and format data sources |
|
||||
|
||||
### 1. Images and Audio
|
||||
|
||||
**If model supports (vision/audio capability):**
|
||||
|
||||
- Keep as multimodal content:
|
||||
- `image_url`: Convert to appropriate format (OpenAI URL vs Claude base64)
|
||||
- `input_audio`: Convert to base64 format
|
||||
|
||||
**If model doesn't support:**
|
||||
|
||||
- Convert to text:
|
||||
- Use agent/MCP specified in `uses.Vision` or `uses.Audio`
|
||||
- Extract text description or transcription
|
||||
- Return as `type="text"` content
|
||||
|
||||
**HTTP URLs:**
|
||||
|
||||
- Fetch content first
|
||||
- Then process the same way as above
|
||||
|
||||
### 2. Files (type="file")
|
||||
|
||||
**Critical**: All `type="file"` content MUST be converted to `text` or `image_url` (if image and model supports).
|
||||
|
||||
**Processing Steps:**
|
||||
|
||||
1. **Fetch file content**:
|
||||
- Uploader wrapper: `__uploader://fileid` → Parse and fetch from attachment manager
|
||||
- HTTP URL: Download from URL
|
||||
|
||||
2. **Detect file type** from content-type and magic bytes
|
||||
|
||||
3. **Process based on file type**:
|
||||
|
||||
| File Type | Output Type | Processing Method |
|
||||
| ------------ | ----------- | -------------------------------------------------------------------------------------------------------- |
|
||||
| **Image** | `image_url` or `text` | If model supports vision → `image_url`<br>If not → use vision tool → `text` |
|
||||
| **PDF** | `text` | If `uses.Vision` supports PDF → use vision tool<br>Otherwise → extract text directly |
|
||||
| **Word** | `text` | Extract text using Word document parser |
|
||||
| **Excel** | `text` | Extract and format as readable table/CSV |
|
||||
| **PPT** | `text` | Extract text and slide content |
|
||||
| **CSV** | `text` | Format as readable table |
|
||||
| **Text** | `text` | Read directly (with encoding detection) |
|
||||
| **JSON/XML** | `text` | Pretty print for readability |
|
||||
|
||||
### 3. Data Sources (type="data")
|
||||
|
||||
**Critical**: All `type="data"` content MUST be converted to `text`.
|
||||
|
||||
**Processing Steps:**
|
||||
|
||||
1. **Parse DataContent.Sources** array
|
||||
2. **Fetch data** from each source:
|
||||
- `model`: Query data model
|
||||
- `kb_collection`: Search knowledge base collection
|
||||
- `kb_document`: Get document content
|
||||
- `table`: Query database table
|
||||
- `api`: Call API endpoint
|
||||
- `mcp_resource`: Fetch MCP resource
|
||||
3. **Format as readable text**:
|
||||
- Tables: Format as markdown tables or CSV
|
||||
- Documents: Include title and content
|
||||
- JSON: Pretty print
|
||||
4. **Return as** `type="text"` content
|
||||
|
||||
## Components
|
||||
|
||||
### Core Files
|
||||
|
||||
- **content.go** - Main entry point (`Vision` function)
|
||||
- **types.go** - Type definitions and constants
|
||||
- **interfaces.go** - Interface definitions
|
||||
|
||||
### Fetching
|
||||
|
||||
- **fetch.go** - Fetch content from HTTP or uploader
|
||||
|
||||
### Processors
|
||||
|
||||
- **processor.go** - Processor registry and routing
|
||||
- **image.go** - Image processing
|
||||
- **audio.go** - Audio processing
|
||||
- **pdf.go** - PDF document processing
|
||||
- **word.go** - Word document processing
|
||||
- **excel.go** - Excel spreadsheet processing
|
||||
- **text.go** - Plain text and CSV processing
|
||||
|
||||
## Frontend Message Format
|
||||
|
||||
The frontend (InputArea) sends messages in the following format:
|
||||
|
||||
### Image Attachments
|
||||
```json
|
||||
{
|
||||
"type": "image_url",
|
||||
"image_url": {
|
||||
"url": "__yao.attachment://file_id",
|
||||
"detail": "auto"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### File Attachments
|
||||
```json
|
||||
{
|
||||
"type": "file",
|
||||
"file": {
|
||||
"url": "__yao.attachment://file_id",
|
||||
"filename": "document.pdf"
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
The `url` field contains an uploader wrapper in the format `__uploader://fileid`.
|
||||
|
||||
## Data Structures
|
||||
|
||||
### ContentInfo
|
||||
|
||||
Holds information about content to be processed:
|
||||
|
||||
```go
|
||||
type ContentInfo struct {
|
||||
Source ContentSource // http, uploader, base64, local
|
||||
FileType FileType // image, audio, pdf, word, excel, etc.
|
||||
ContentType string // MIME type
|
||||
URL string // Original URL or file ID
|
||||
Data []byte // File data
|
||||
|
||||
// For uploader wrapper
|
||||
UploaderName string
|
||||
FileID string
|
||||
}
|
||||
```
|
||||
|
||||
### ProcessedContent
|
||||
|
||||
Result of content processing:
|
||||
|
||||
```go
|
||||
type ProcessedContent struct {
|
||||
Text string // Extracted text
|
||||
ContentPart *context.ContentPart // For model input
|
||||
Metadata map[string]interface{}
|
||||
Error error
|
||||
}
|
||||
```
|
||||
|
||||
## Usage Example
|
||||
|
||||
```go
|
||||
import (
|
||||
"github.com/yaoapp/yao/agent/content"
|
||||
"github.com/yaoapp/yao/agent/context"
|
||||
)
|
||||
|
||||
// Process messages before sending to LLM
|
||||
processedMessages, err := content.Vision(
|
||||
ctx,
|
||||
capabilities, // Model capabilities
|
||||
messages, // Original messages
|
||||
uses, // Tool specifications (vision, audio, etc.)
|
||||
)
|
||||
```
|
||||
|
||||
## Performance Optimization
|
||||
|
||||
### File Processing Cache
|
||||
|
||||
**Problem**: Same file (uploader wrapper) might appear in multiple messages or be referenced multiple times.
|
||||
|
||||
**Solution**: Three-level caching strategy:
|
||||
|
||||
1. **In-memory cache** (`processedFiles` map):
|
||||
- Caches processed text for the duration of the Vision() call
|
||||
- Key: file ID from uploader wrapper
|
||||
- Value: extracted text content
|
||||
|
||||
2. **Attachment preview** (attachment.GetText with preview):
|
||||
- Tries to get preview (first 2000 chars) from attachment manager
|
||||
- If file was previously processed and saved, preview is available immediately
|
||||
- Much faster than full file processing
|
||||
|
||||
3. **Full processing** (only if needed):
|
||||
- Falls back to complete file processing if no cache/preview available
|
||||
- Result is cached in memory and optionally saved to attachment manager
|
||||
|
||||
### Cache Flow
|
||||
|
||||
```go
|
||||
// For uploader://file_id
|
||||
1. Check processedFiles[file_id]
|
||||
└── Found? → Return cached text ⚡ (fastest)
|
||||
|
||||
2. Not in cache → Call attachment.GetText(file_id, false) // preview only
|
||||
└── Has preview? → Cache and return ⚡ (fast)
|
||||
|
||||
3. No preview → Process file fully 🔄 (slower)
|
||||
└── Cache result in processedFiles
|
||||
└── Optional: Save to attachment using SaveText for future use
|
||||
```
|
||||
|
||||
### Benefits
|
||||
|
||||
- **Avoid duplicate processing**: Same file processed only once per Vision() call
|
||||
- **Fast preview access**: Leverage pre-processed content from attachment manager
|
||||
- **Reduced latency**: Especially important for large documents (PDFs, Word, Excel)
|
||||
- **Resource efficient**: Less CPU/memory usage for repeated file references
|
||||
|
||||
## Implementation Status
|
||||
|
||||
### ✅ Completed
|
||||
|
||||
- [x] Package structure
|
||||
- [x] Type definitions
|
||||
- [x] Interface definitions
|
||||
- [x] Skeleton functions with TODO comments
|
||||
- [x] File processing cache infrastructure
|
||||
- [x] Cache helper functions (tryGetCachedText, cacheProcessedText)
|
||||
|
||||
### 🚧 To Implement
|
||||
|
||||
- [ ] tryGetCachedText implementation (attachment.GetText integration)
|
||||
- [ ] cacheProcessedText implementation (attachment.SaveText integration)
|
||||
- [ ] HTTP fetching logic
|
||||
- [ ] Uploader wrapper parsing and fetching
|
||||
- [ ] Image processing (base64, vision API)
|
||||
- [ ] Audio processing (transcription)
|
||||
- [ ] PDF text extraction
|
||||
- [ ] Word document parsing
|
||||
- [ ] Excel spreadsheet parsing
|
||||
- [ ] Text/CSV formatting
|
||||
- [ ] Content part processing logic
|
||||
- [ ] Model capability detection
|
||||
- [ ] Agent/MCP tool invocation
|
||||
|
||||
## Configuration
|
||||
|
||||
Content processing behavior is controlled by:
|
||||
|
||||
1. **Model Capabilities** (`openai.Capabilities`)
|
||||
|
||||
- Determines if model can handle images/audio directly
|
||||
- Specifies vision format (OpenAI vs Claude)
|
||||
|
||||
2. **Uses** (`context.Uses`)
|
||||
```go
|
||||
type Uses struct {
|
||||
Vision string // "agent" or "mcp:server_id"
|
||||
Audio string // "agent" or "mcp:server_id"
|
||||
Search string
|
||||
Fetch string
|
||||
}
|
||||
```
|
||||
|
||||
## Error Handling
|
||||
|
||||
- Errors during processing are logged but don't stop the entire pipeline
|
||||
- Original content is kept if processing fails
|
||||
- Graceful degradation: if advanced processing fails, fall back to simpler methods
|
||||
61
agent/content/audio.go
Normal file
61
agent/content/audio.go
Normal file
|
|
@ -0,0 +1,61 @@
|
|||
package content
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"github.com/yaoapp/gou/connector/openai"
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
)
|
||||
|
||||
// AudioHandler handles audio content
|
||||
type AudioHandler struct{}
|
||||
|
||||
// CanHandle checks if this handler can handle the content type
|
||||
func (h *AudioHandler) CanHandle(contentType string, fileType FileType) bool {
|
||||
return fileType == FileTypeAudio || strings.HasPrefix(contentType, "audio/")
|
||||
}
|
||||
|
||||
// Handle processes audio content
|
||||
// Logic similar to image:
|
||||
// 1. If model supports audio input -> convert to base64 format
|
||||
// 2. If model doesn't support audio -> use agent/MCP specified in uses.Audio
|
||||
func (h *AudioHandler) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses) (*Result, error) {
|
||||
// TODO: Implement audio handling
|
||||
// 1. Check model audio capabilities
|
||||
// 2. If supported:
|
||||
// - Encode audio as base64 with proper format
|
||||
// 3. If not supported:
|
||||
// - Call audio agent/MCP to transcribe audio to text
|
||||
// 4. Return Result with text or ContentPart
|
||||
return nil, fmt.Errorf("not implemented")
|
||||
}
|
||||
|
||||
// handleWithAudioModel processes audio using model's audio capability
|
||||
func (h *AudioHandler) handleWithAudioModel(ctx *agentContext.Context, info *Info) (*Result, error) {
|
||||
// TODO: Implement audio model processing
|
||||
// Format audio according to model's audio input format
|
||||
return nil, fmt.Errorf("not implemented")
|
||||
}
|
||||
|
||||
// handleWithAudioAgent processes audio using audio agent or MCP
|
||||
func (h *AudioHandler) handleWithAudioAgent(ctx *agentContext.Context, info *Info, audioTool string) (string, error) {
|
||||
// TODO: Implement audio agent/MCP processing
|
||||
// 1. Parse audioTool (format: "agent" or "mcp:server_id")
|
||||
// 2. Call appropriate tool to transcribe audio
|
||||
// 3. Return transcribed text
|
||||
return "", fmt.Errorf("not implemented")
|
||||
}
|
||||
|
||||
// encodeAudioBase64 encodes audio data to base64 with proper format
|
||||
func encodeAudioBase64(data []byte, contentType string) string {
|
||||
// TODO: Implement audio base64 encoding
|
||||
return ""
|
||||
}
|
||||
|
||||
// detectAudioFormat detects audio format from content type or data
|
||||
func detectAudioFormat(contentType string, data []byte) string {
|
||||
// TODO: Implement audio format detection
|
||||
// Return format like "wav", "mp3", "flac", etc.
|
||||
return ""
|
||||
}
|
||||
470
agent/content/content.go
Normal file
470
agent/content/content.go
Normal file
|
|
@ -0,0 +1,470 @@
|
|||
package content
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"github.com/yaoapp/gou/connector/openai"
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
"github.com/yaoapp/yao/attachment"
|
||||
)
|
||||
|
||||
// Vision transforms extended content types to LLM-compatible formats
|
||||
// This is the main entry point for content preprocessing before sending to LLM
|
||||
//
|
||||
// IMPORTANT: This function is called BEFORE sending messages to LLM (in agent.executeLLMStream)
|
||||
// It must convert all extended content types to standard LLM-compatible types.
|
||||
//
|
||||
// Input Content Types (Extended):
|
||||
// - type="text" -> Pass through (already standard)
|
||||
// - type="image_url" -> Process based on model capability (may need base64 conversion or vision tool)
|
||||
// - type="input_audio" -> Process based on model capability (may need transcription)
|
||||
// - type="file" -> Convert to text or image_url (MUST be converted)
|
||||
// - type="data" -> Convert to text (MUST be converted)
|
||||
//
|
||||
// Output Content Types (LLM-compatible only):
|
||||
// - type="text" -> Text content
|
||||
// - type="image_url" -> Image (only if model supports vision)
|
||||
// - type="input_audio" -> Audio (only if model supports audio)
|
||||
//
|
||||
// Processing Logic:
|
||||
// 1. For images (image_url):
|
||||
// - If model supports vision -> keep as image_url (may convert URL to base64)
|
||||
// - If model doesn't support -> use vision agent/MCP to extract text -> convert to type="text"
|
||||
//
|
||||
// 2. For audio (input_audio):
|
||||
// - If model supports audio -> keep as input_audio
|
||||
// - If model doesn't support -> use audio agent/MCP to transcribe -> convert to type="text"
|
||||
//
|
||||
// 3. For files (type="file"):
|
||||
// - Parse uploader wrapper (__uploader://fileid) or fetch HTTP URL
|
||||
// - Detect file type (PDF, Word, Excel, Image, etc.)
|
||||
// - Process based on file type:
|
||||
// - Images: same as image processing above
|
||||
// - PDF: use vision tool if available, otherwise extract text -> type="text"
|
||||
// - Word/Excel/PPT/CSV: extract text -> type="text"
|
||||
// - MUST convert to type="text" or type="image_url" (if image and model supports)
|
||||
//
|
||||
// 4. For data (type="data"):
|
||||
// - Fetch data from sources (models, KB, MCP resources, etc.)
|
||||
// - Format as readable text
|
||||
// - MUST convert to type="text"
|
||||
//
|
||||
// Return: Messages with only standard LLM-compatible content types (text, image_url, input_audio)
|
||||
func Vision(ctx *agentContext.Context, capabilities *openai.Capabilities, messages []agentContext.Message, uses *agentContext.Uses) ([]agentContext.Message, error) {
|
||||
// Initialize handlers and fetcher
|
||||
registry := NewRegistry()
|
||||
fetcher := NewFetcher()
|
||||
|
||||
// Cache for processed files (uploader wrapper -> extracted text)
|
||||
// Ensures each file is only processed once
|
||||
processedFiles := make(map[string]string)
|
||||
|
||||
// Process each message
|
||||
processedMessages := make([]agentContext.Message, 0, len(messages))
|
||||
|
||||
for _, msg := range messages {
|
||||
processedMsg, err := processMessage(ctx, &msg, capabilities, uses, registry, fetcher, processedFiles)
|
||||
if err != nil {
|
||||
// Log error but continue processing other messages
|
||||
// TODO: Add proper logging
|
||||
fmt.Printf("Warning: failed to process message: %v\n", err)
|
||||
processedMessages = append(processedMessages, msg) // Keep original on error
|
||||
continue
|
||||
}
|
||||
processedMessages = append(processedMessages, processedMsg)
|
||||
}
|
||||
|
||||
return processedMessages, nil
|
||||
}
|
||||
|
||||
// processMessage processes a single message and its content parts
|
||||
func processMessage(
|
||||
ctx *agentContext.Context,
|
||||
msg *agentContext.Message,
|
||||
capabilities *openai.Capabilities,
|
||||
uses *agentContext.Uses,
|
||||
registry *Registry,
|
||||
fetcher Fetcher,
|
||||
processedFiles map[string]string,
|
||||
) (agentContext.Message, error) {
|
||||
// If content is simple string, no processing needed
|
||||
if _, ok := msg.GetContentAsString(); ok {
|
||||
return *msg, nil
|
||||
}
|
||||
|
||||
// Get content parts
|
||||
parts, ok := msg.GetContentAsParts()
|
||||
if !ok {
|
||||
return *msg, nil
|
||||
}
|
||||
|
||||
// Process each content part
|
||||
processedParts := make([]agentContext.ContentPart, 0, len(parts))
|
||||
for _, part := range parts {
|
||||
processedPart, err := processContentPart(ctx, &part, capabilities, uses, registry, fetcher, processedFiles)
|
||||
if err != nil {
|
||||
// Log error and handle gracefully
|
||||
fmt.Printf("Warning: failed to process content part: %v\n", err)
|
||||
|
||||
// For image_url that failed to process, convert to text description
|
||||
// This prevents sending unsupported multimodal content to non-vision models
|
||||
if part.Type == agentContext.ContentImageURL {
|
||||
processedParts = append(processedParts, agentContext.ContentPart{
|
||||
Type: agentContext.ContentText,
|
||||
Text: fmt.Sprintf("[Image processing failed: %s]", part.ImageURL.URL),
|
||||
})
|
||||
} else {
|
||||
// For other types, keep original
|
||||
processedParts = append(processedParts, part)
|
||||
}
|
||||
continue
|
||||
}
|
||||
|
||||
// If handling returned text, convert to text part
|
||||
if processedPart.Text != "" {
|
||||
processedParts = append(processedParts, agentContext.ContentPart{
|
||||
Type: agentContext.ContentText,
|
||||
Text: processedPart.Text,
|
||||
})
|
||||
} else if processedPart.ContentPart != nil {
|
||||
// Use the processed content part (e.g., base64 image)
|
||||
processedParts = append(processedParts, *processedPart.ContentPart)
|
||||
} else {
|
||||
// Keep original if no handling result
|
||||
processedParts = append(processedParts, part)
|
||||
}
|
||||
}
|
||||
|
||||
// Return new message with processed content
|
||||
return agentContext.Message{
|
||||
Role: msg.Role,
|
||||
Content: processedParts,
|
||||
Name: msg.Name,
|
||||
ToolCallID: msg.ToolCallID,
|
||||
ToolCalls: msg.ToolCalls,
|
||||
Refusal: msg.Refusal,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// processContentPart processes a single content part
|
||||
// IMPORTANT: Must convert extended types (file, data) to standard types (text, image_url, input_audio)
|
||||
func processContentPart(
|
||||
ctx *agentContext.Context,
|
||||
part *agentContext.ContentPart,
|
||||
capabilities *openai.Capabilities,
|
||||
uses *agentContext.Uses,
|
||||
registry *Registry,
|
||||
fetcher Fetcher,
|
||||
processedFiles map[string]string,
|
||||
) (*Result, error) {
|
||||
// 1. Handle standard types - pass through
|
||||
switch part.Type {
|
||||
case agentContext.ContentText:
|
||||
// Text is already standard, pass through
|
||||
return &Result{
|
||||
ContentPart: part,
|
||||
}, nil
|
||||
|
||||
case agentContext.ContentImageURL:
|
||||
// Image URL - check if it needs processing
|
||||
return processImageURLContent(ctx, part, capabilities, uses, registry, fetcher, processedFiles)
|
||||
|
||||
case agentContext.ContentInputAudio:
|
||||
// Audio - check if it needs processing
|
||||
return processAudioContent(ctx, part, capabilities, uses, registry, fetcher, processedFiles)
|
||||
}
|
||||
|
||||
// 2. Handle extended types - MUST convert to standard types
|
||||
switch part.Type {
|
||||
case agentContext.ContentFile:
|
||||
return processFileContent(ctx, part, capabilities, uses, registry, fetcher, processedFiles)
|
||||
|
||||
case agentContext.ContentData:
|
||||
return processDataContent(ctx, part)
|
||||
|
||||
default:
|
||||
// Unknown type, return error
|
||||
return nil, fmt.Errorf("unsupported content type: %s", part.Type)
|
||||
}
|
||||
}
|
||||
|
||||
// processFileContent processes file content with caching
|
||||
func processFileContent(
|
||||
ctx *agentContext.Context,
|
||||
part *agentContext.ContentPart,
|
||||
capabilities *openai.Capabilities,
|
||||
uses *agentContext.Uses,
|
||||
registry *Registry,
|
||||
fetcher Fetcher,
|
||||
processedFiles map[string]string,
|
||||
) (*Result, error) {
|
||||
if part.File == nil || part.File.URL == "" {
|
||||
return nil, fmt.Errorf("file content part missing URL")
|
||||
}
|
||||
|
||||
url := part.File.URL
|
||||
|
||||
// Step 1: Try to get cached text (three-tier cache)
|
||||
cachedText, found, err := tryGetCachedText(ctx, url, processedFiles)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to check cache: %w", err)
|
||||
}
|
||||
if found {
|
||||
// Cache hit! Return as text
|
||||
return &Result{
|
||||
Text: cachedText,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// Step 2: No cache, need to process the file
|
||||
// Determine content source
|
||||
source, sourceURL, err := determineContentSource(part)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to determine content source: %w", err)
|
||||
}
|
||||
|
||||
// Fetch content
|
||||
info, err := fetcher.Fetch(ctx, source, sourceURL)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to fetch content: %w", err)
|
||||
}
|
||||
|
||||
// Detect file type if not already set
|
||||
if info.FileType == FileTypeUnknown {
|
||||
info.FileType = DetectFileType(info.ContentType, part.File.Filename)
|
||||
}
|
||||
|
||||
// Process with appropriate handler
|
||||
result, err := registry.Handle(ctx, info, capabilities, uses)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to handle content: %w", err)
|
||||
}
|
||||
|
||||
// Step 3: Cache the result if it's text
|
||||
if result.Text != "" {
|
||||
if cacheErr := cacheProcessedText(ctx, url, result.Text, processedFiles); cacheErr != nil {
|
||||
// Log error but don't fail the request
|
||||
fmt.Printf("Warning: failed to cache processed text: %v\n", cacheErr)
|
||||
}
|
||||
}
|
||||
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// processImageURLContent processes image_url content
|
||||
// If URL is uploader wrapper or HTTP, fetch and process it
|
||||
func processImageURLContent(
|
||||
ctx *agentContext.Context,
|
||||
part *agentContext.ContentPart,
|
||||
capabilities *openai.Capabilities,
|
||||
uses *agentContext.Uses,
|
||||
registry *Registry,
|
||||
fetcher Fetcher,
|
||||
processedFiles map[string]string,
|
||||
) (*Result, error) {
|
||||
if part.ImageURL == nil || part.ImageURL.URL == "" {
|
||||
return nil, fmt.Errorf("image_url content missing URL")
|
||||
}
|
||||
|
||||
url := part.ImageURL.URL
|
||||
|
||||
// If it's a data URI (base64), pass through
|
||||
if strings.HasPrefix(url, "data:") {
|
||||
return &Result{
|
||||
ContentPart: part,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// If it's uploader wrapper or HTTP URL, need to process
|
||||
// Check cache first
|
||||
cachedText, found, err := tryGetCachedText(ctx, url, processedFiles)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to check cache: %w", err)
|
||||
}
|
||||
if found {
|
||||
// Cache hit! Return as text
|
||||
return &Result{
|
||||
Text: cachedText,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// Determine source
|
||||
source, sourceURL, err := determineContentSource(part)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to determine content source: %w", err)
|
||||
}
|
||||
|
||||
// Fetch content
|
||||
info, err := fetcher.Fetch(ctx, source, sourceURL)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to fetch image: %w", err)
|
||||
}
|
||||
|
||||
// Set file type as image
|
||||
info.FileType = FileTypeImage
|
||||
|
||||
// Process with image handler
|
||||
result, err := registry.Handle(ctx, info, capabilities, uses)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to handle image: %w", err)
|
||||
}
|
||||
|
||||
// Cache if result is text
|
||||
if result.Text != "" {
|
||||
if cacheErr := cacheProcessedText(ctx, url, result.Text, processedFiles); cacheErr != nil {
|
||||
fmt.Printf("Warning: failed to cache processed text: %v\n", cacheErr)
|
||||
}
|
||||
}
|
||||
|
||||
return result, nil
|
||||
}
|
||||
|
||||
// processAudioContent processes input_audio content
|
||||
func processAudioContent(
|
||||
ctx *agentContext.Context,
|
||||
part *agentContext.ContentPart,
|
||||
capabilities *openai.Capabilities,
|
||||
uses *agentContext.Uses,
|
||||
registry *Registry,
|
||||
fetcher Fetcher,
|
||||
processedFiles map[string]string,
|
||||
) (*Result, error) {
|
||||
if part.InputAudio == nil || part.InputAudio.Data == "" {
|
||||
return nil, fmt.Errorf("input_audio content missing data")
|
||||
}
|
||||
|
||||
// For now, pass through audio as-is
|
||||
// TODO: Implement audio processing (transcription, etc.)
|
||||
return &Result{
|
||||
ContentPart: part,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// processDataContent processes data content (converts to text)
|
||||
func processDataContent(ctx *agentContext.Context, part *agentContext.ContentPart) (*Result, error) {
|
||||
if part.Data == nil {
|
||||
return nil, fmt.Errorf("data content part missing data")
|
||||
}
|
||||
|
||||
// TODO: Implement data processing
|
||||
// For now, just return error
|
||||
return nil, fmt.Errorf("data content processing not implemented yet")
|
||||
}
|
||||
|
||||
// determineContentSource determines where the content comes from
|
||||
func determineContentSource(part *agentContext.ContentPart) (Source, string, error) {
|
||||
var url string
|
||||
|
||||
// Extract URL based on content type
|
||||
switch part.Type {
|
||||
case agentContext.ContentFile:
|
||||
if part.File == nil || part.File.URL == "" {
|
||||
return "", "", fmt.Errorf("file content missing URL")
|
||||
}
|
||||
url = part.File.URL
|
||||
|
||||
case agentContext.ContentImageURL:
|
||||
if part.ImageURL == nil || part.ImageURL.URL == "" {
|
||||
return "", "", fmt.Errorf("image_url content missing URL")
|
||||
}
|
||||
url = part.ImageURL.URL
|
||||
|
||||
case agentContext.ContentInputAudio:
|
||||
if part.InputAudio == nil || part.InputAudio.Data == "" {
|
||||
return "", "", fmt.Errorf("input_audio content missing data")
|
||||
}
|
||||
// Audio data is base64, treat as base64 source
|
||||
return SourceBase64, part.InputAudio.Data, nil
|
||||
|
||||
default:
|
||||
return "", "", fmt.Errorf("unsupported content type for source detection: %s", part.Type)
|
||||
}
|
||||
|
||||
// Determine source type based on URL format
|
||||
if strings.HasPrefix(url, "http://") || strings.HasPrefix(url, "https://") {
|
||||
return SourceHTTP, url, nil
|
||||
}
|
||||
|
||||
if strings.HasPrefix(url, "__") {
|
||||
// Uploader wrapper format: __uploader://fileid
|
||||
return SourceUploader, url, nil
|
||||
}
|
||||
|
||||
if strings.HasPrefix(url, "data:") {
|
||||
// Data URI (base64)
|
||||
return SourceBase64, url, nil
|
||||
}
|
||||
|
||||
// Default to treating as uploader if no prefix matches
|
||||
return SourceUploader, url, nil
|
||||
}
|
||||
|
||||
// shouldProcessWithModel checks if content should be processed by the model directly
|
||||
func shouldProcessWithModel(capabilities *openai.Capabilities, fileType FileType) (bool, agentContext.VisionFormat) {
|
||||
// TODO: Implement model capability check
|
||||
// For images: check if model supports vision
|
||||
// For audio: check if model supports audio input
|
||||
// Return whether to use model and the format to use
|
||||
return false, agentContext.VisionFormatNone
|
||||
}
|
||||
|
||||
// getToolForProcessing gets the agent/MCP tool to use for processing
|
||||
func getToolForProcessing(uses *agentContext.Uses, fileType FileType) string {
|
||||
// TODO: Implement tool selection
|
||||
// Based on file type, return the appropriate tool from uses
|
||||
// - Images -> uses.Vision
|
||||
// - Audio -> uses.Audio
|
||||
// - PDF (if vision available) -> uses.Vision
|
||||
return ""
|
||||
}
|
||||
|
||||
// tryGetCachedText checks if the URL is an uploader wrapper and tries to get cached text
|
||||
// Returns (text, found, error)
|
||||
func tryGetCachedText(ctx *agentContext.Context, url string, processedFiles map[string]string) (string, bool, error) {
|
||||
// Parse URL to check if it's an uploader wrapper
|
||||
uploaderName, fileID, isWrapper := attachment.Parse(url)
|
||||
if !isWrapper {
|
||||
return "", false, nil // Not an uploader wrapper, no cache
|
||||
}
|
||||
|
||||
// 1. Check in-memory cache for this Vision call
|
||||
if text, ok := processedFiles[fileID]; ok {
|
||||
return text, true, nil
|
||||
}
|
||||
|
||||
// 2. Try attachment manager's content_preview (cross-call cache)
|
||||
manager, exists := attachment.Managers[uploaderName]
|
||||
if exists {
|
||||
// GetText with fullContent=false to get preview (default)
|
||||
text, err := manager.GetText(ctx.Context, fileID, false)
|
||||
if err == nil && text != "" {
|
||||
// Cache in-memory for this Vision call
|
||||
processedFiles[fileID] = text
|
||||
return text, true, nil
|
||||
}
|
||||
}
|
||||
|
||||
// No cache found
|
||||
return "", false, nil
|
||||
}
|
||||
|
||||
// cacheProcessedText caches the processed text for an uploader wrapper
|
||||
func cacheProcessedText(ctx *agentContext.Context, url string, text string, processedFiles map[string]string) error {
|
||||
// Parse URL to get uploader name and file ID
|
||||
uploaderName, fileID, isWrapper := attachment.Parse(url)
|
||||
if !isWrapper {
|
||||
return nil // Not an uploader wrapper, nothing to cache
|
||||
}
|
||||
|
||||
// 1. Cache in-memory for this Vision call
|
||||
processedFiles[fileID] = text
|
||||
|
||||
// 2. Save to attachment manager for future Vision calls
|
||||
manager, exists := attachment.Managers[uploaderName]
|
||||
if exists {
|
||||
return manager.SaveText(ctx.Context, fileID, text)
|
||||
}
|
||||
|
||||
return nil
|
||||
}
|
||||
458
agent/content/content_vision_test.go
Normal file
458
agent/content/content_vision_test.go
Normal file
|
|
@ -0,0 +1,458 @@
|
|||
package content_test
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"context"
|
||||
"image"
|
||||
"image/color"
|
||||
"image/png"
|
||||
"mime/multipart"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/yaoapp/gou/connector/openai"
|
||||
"github.com/yaoapp/yao/agent/content"
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
"github.com/yaoapp/yao/agent/testutils"
|
||||
"github.com/yaoapp/yao/attachment"
|
||||
)
|
||||
|
||||
// setupTestUploader creates and registers a test uploader manager
|
||||
// The manager will be registered with "__" prefix as required by attachment.Parse
|
||||
func setupTestUploader(t *testing.T, name string) attachment.FileManager {
|
||||
// Register with __ prefix to match Parse behavior
|
||||
managerName := "__" + name
|
||||
manager, err := attachment.Register(managerName, "local", attachment.ManagerOption{
|
||||
Driver: "local",
|
||||
MaxSize: "10M",
|
||||
AllowedTypes: []string{"text/*", "image/*", "application/*"},
|
||||
Options: map[string]interface{}{
|
||||
"path": "/tmp/test_vision_attachments_" + name,
|
||||
},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to register attachment manager '%s': %v", managerName, err)
|
||||
}
|
||||
return manager
|
||||
}
|
||||
|
||||
// cleanupTestUploader removes the test uploader from registry
|
||||
func cleanupTestUploader(name string) {
|
||||
delete(attachment.Managers, "__"+name)
|
||||
}
|
||||
|
||||
// generateTestImage creates a valid PNG image (100x100 red square)
|
||||
func generateTestImage(t *testing.T) []byte {
|
||||
img := image.NewRGBA(image.Rect(0, 0, 100, 100))
|
||||
red := color.RGBA{255, 0, 0, 255}
|
||||
for y := 0; y < 100; y++ {
|
||||
for x := 0; x < 100; x++ {
|
||||
img.Set(x, y, red)
|
||||
}
|
||||
}
|
||||
|
||||
var buf bytes.Buffer
|
||||
if err := png.Encode(&buf, img); err != nil {
|
||||
t.Fatalf("Failed to encode test image: %v", err)
|
||||
}
|
||||
return buf.Bytes()
|
||||
}
|
||||
|
||||
// TestVision_TextFile tests Vision function with text/code file parsing
|
||||
func TestVision_TextFile(t *testing.T) {
|
||||
testutils.Prepare(t)
|
||||
defer testutils.Clean(t)
|
||||
|
||||
// Setup test uploader
|
||||
uploaderName := "test-vision-text"
|
||||
manager := setupTestUploader(t, uploaderName)
|
||||
defer cleanupTestUploader(uploaderName)
|
||||
|
||||
// 1. Create and upload a Go source file
|
||||
testContent := `package main
|
||||
|
||||
import "fmt"
|
||||
|
||||
func main() {
|
||||
fmt.Println("Hello, Vision Test!")
|
||||
}
|
||||
`
|
||||
|
||||
// Upload file
|
||||
reader := strings.NewReader(testContent)
|
||||
fileHeader := &attachment.FileHeader{
|
||||
FileHeader: &multipart.FileHeader{
|
||||
Filename: "main.go",
|
||||
Size: int64(len(testContent)),
|
||||
Header: make(map[string][]string),
|
||||
},
|
||||
}
|
||||
fileHeader.Header.Set("Content-Type", "text/x-go")
|
||||
|
||||
uploadedFile, err := manager.Upload(context.Background(), fileHeader, reader, attachment.UploadOption{
|
||||
Groups: []string{"vision", "test"},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to upload file: %v", err)
|
||||
}
|
||||
|
||||
t.Logf("Uploaded file ID: %s", uploadedFile.ID)
|
||||
|
||||
// 2. Prepare Vision context (text files don't need special capabilities)
|
||||
ctx := agentContext.New(context.Background(), nil, "test")
|
||||
|
||||
capabilities := &openai.Capabilities{}
|
||||
|
||||
messages := []agentContext.Message{
|
||||
{
|
||||
Role: "user",
|
||||
Content: []agentContext.ContentPart{
|
||||
{
|
||||
Type: agentContext.ContentFile,
|
||||
File: &agentContext.FileAttachment{
|
||||
URL: "__" + uploaderName + "://" + uploadedFile.ID,
|
||||
Filename: "main.go",
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
// 3. Call Vision function
|
||||
result, err := content.Vision(ctx, capabilities, messages, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("Vision function failed: %v", err)
|
||||
}
|
||||
|
||||
if len(result) != 1 {
|
||||
t.Fatalf("Expected 1 message, got %d", len(result))
|
||||
}
|
||||
|
||||
// 4. Verify result
|
||||
contentParts, ok := result[0].Content.([]agentContext.ContentPart)
|
||||
if !ok {
|
||||
t.Fatalf("Expected content to be []ContentPart, got %T", result[0].Content)
|
||||
}
|
||||
|
||||
if len(contentParts) != 1 {
|
||||
t.Fatalf("Expected 1 content part, got %d", len(contentParts))
|
||||
}
|
||||
|
||||
// Should be converted to text
|
||||
if contentParts[0].Type != agentContext.ContentText {
|
||||
t.Errorf("Expected ContentText type, got %s", contentParts[0].Type)
|
||||
}
|
||||
|
||||
if !strings.Contains(contentParts[0].Text, "package main") {
|
||||
t.Errorf("Expected text to contain 'package main', got: %s", contentParts[0].Text)
|
||||
}
|
||||
|
||||
if !strings.Contains(contentParts[0].Text, "Hello, Vision Test!") {
|
||||
t.Errorf("Expected text to contain 'Hello, Vision Test!', got: %s", contentParts[0].Text)
|
||||
}
|
||||
|
||||
t.Logf("✓ Text file successfully parsed: %d characters", len(contentParts[0].Text))
|
||||
}
|
||||
|
||||
// TestVision_ImageWithVisionSupport tests image processing with vision-capable model
|
||||
func TestVision_ImageWithVisionSupport(t *testing.T) {
|
||||
testutils.Prepare(t)
|
||||
defer testutils.Clean(t)
|
||||
|
||||
// Setup test uploader
|
||||
uploaderName := "test-vision-image"
|
||||
manager := setupTestUploader(t, uploaderName)
|
||||
defer cleanupTestUploader(uploaderName)
|
||||
|
||||
// 1. Create and upload a test image (1x1 red PNG)
|
||||
imageData := []byte{
|
||||
0x89, 0x50, 0x4E, 0x47, 0x0D, 0x0A, 0x1A, 0x0A,
|
||||
0x00, 0x00, 0x00, 0x0D, 0x49, 0x48, 0x44, 0x52,
|
||||
0x00, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01,
|
||||
0x08, 0x02, 0x00, 0x00, 0x00, 0x90, 0x77, 0x53,
|
||||
0xDE, 0x00, 0x00, 0x00, 0x0C, 0x49, 0x44, 0x41,
|
||||
0x54, 0x08, 0xD7, 0x63, 0xF8, 0xCF, 0xC0, 0x00,
|
||||
0x00, 0x03, 0x01, 0x01, 0x00, 0x18, 0xDD, 0x8D,
|
||||
0xB4, 0x00, 0x00, 0x00, 0x00, 0x49, 0x45, 0x4E,
|
||||
0x44, 0xAE, 0x42, 0x60, 0x82,
|
||||
}
|
||||
|
||||
// Upload image
|
||||
reader := strings.NewReader(string(imageData))
|
||||
fileHeader := &attachment.FileHeader{
|
||||
FileHeader: &multipart.FileHeader{
|
||||
Filename: "test.png",
|
||||
Size: int64(len(imageData)),
|
||||
Header: make(map[string][]string),
|
||||
},
|
||||
}
|
||||
fileHeader.Header.Set("Content-Type", "image/png")
|
||||
|
||||
uploadedFile, err := manager.Upload(context.Background(), fileHeader, reader, attachment.UploadOption{
|
||||
Groups: []string{"vision", "test"},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to upload image: %v", err)
|
||||
}
|
||||
|
||||
// 2. Prepare Vision context with vision-capable model
|
||||
ctx := agentContext.New(context.Background(), nil, "test")
|
||||
|
||||
// Construct capabilities with vision support (OpenAI format)
|
||||
capabilities := &openai.Capabilities{
|
||||
Vision: agentContext.VisionFormatOpenAI, // OpenAI vision format
|
||||
}
|
||||
|
||||
messages := []agentContext.Message{
|
||||
{
|
||||
Role: "user",
|
||||
Content: []agentContext.ContentPart{
|
||||
{
|
||||
Type: agentContext.ContentImageURL,
|
||||
ImageURL: &agentContext.ImageURL{
|
||||
URL: "__" + uploaderName + "://" + uploadedFile.ID,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
// 3. Call Vision function (no uses needed for direct vision support)
|
||||
result, err := content.Vision(ctx, capabilities, messages, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("Vision function failed: %v", err)
|
||||
}
|
||||
|
||||
if len(result) != 1 {
|
||||
t.Fatalf("Expected 1 message, got %d", len(result))
|
||||
}
|
||||
|
||||
// 4. Verify result
|
||||
contentParts, ok := result[0].Content.([]agentContext.ContentPart)
|
||||
if !ok {
|
||||
t.Fatalf("Expected content to be []ContentPart, got %T", result[0].Content)
|
||||
}
|
||||
|
||||
if len(contentParts) != 1 {
|
||||
t.Fatalf("Expected 1 content part, got %d", len(contentParts))
|
||||
}
|
||||
|
||||
// If model supports vision, should be image_url with base64
|
||||
if capabilities.Vision != nil {
|
||||
if contentParts[0].Type != agentContext.ContentImageURL {
|
||||
t.Errorf("Expected ContentImageURL type, got %s", contentParts[0].Type)
|
||||
}
|
||||
|
||||
if contentParts[0].ImageURL == nil {
|
||||
t.Fatal("Expected ImageURL to be set")
|
||||
}
|
||||
|
||||
if !strings.Contains(contentParts[0].ImageURL.URL, "data:image/png;base64,") {
|
||||
t.Errorf("Expected base64 data URI, got: %s", contentParts[0].ImageURL.URL)
|
||||
}
|
||||
|
||||
t.Logf("✓ Image processed with vision support: %d bytes (base64)", len(contentParts[0].ImageURL.URL))
|
||||
} else {
|
||||
// If no vision support, should fall back to text (via agent/MCP)
|
||||
t.Logf("ℹ Model doesn't support vision, result type: %s", contentParts[0].Type)
|
||||
}
|
||||
}
|
||||
|
||||
// TestVision_ImageWithAgent tests image processing with vision agent when model doesn't support vision
|
||||
// Note: This test demonstrates the agent fallback mechanism when the model doesn't support vision
|
||||
func TestVision_ImageWithAgent(t *testing.T) {
|
||||
testutils.Prepare(t)
|
||||
defer testutils.Clean(t)
|
||||
|
||||
// Setup test uploader
|
||||
uploaderName := "test-vision-agent"
|
||||
manager := setupTestUploader(t, uploaderName)
|
||||
defer cleanupTestUploader(uploaderName)
|
||||
|
||||
// 1. Generate and upload a valid test image (100x100 red PNG)
|
||||
imageData := generateTestImage(t)
|
||||
|
||||
reader := strings.NewReader(string(imageData))
|
||||
fileHeader := &attachment.FileHeader{
|
||||
FileHeader: &multipart.FileHeader{
|
||||
Filename: "test.png",
|
||||
Size: int64(len(imageData)),
|
||||
Header: make(map[string][]string),
|
||||
},
|
||||
}
|
||||
fileHeader.Header.Set("Content-Type", "image/png")
|
||||
|
||||
uploadedFile, err := manager.Upload(context.Background(), fileHeader, reader, attachment.UploadOption{
|
||||
Groups: []string{"vision", "test"},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to upload image: %v", err)
|
||||
}
|
||||
|
||||
// 2. Prepare Vision context with proper setup
|
||||
// Model does NOT support vision, but uses.Vision specifies a vision agent
|
||||
ctx := agentContext.New(context.Background(), nil, "test")
|
||||
|
||||
// Capabilities without vision support
|
||||
capabilities := &openai.Capabilities{
|
||||
Vision: nil, // No vision support
|
||||
}
|
||||
|
||||
// Uses configuration with vision agent
|
||||
uses := &agentContext.Uses{
|
||||
Vision: "tests.vision-helper", // Use vision-helper agent
|
||||
}
|
||||
|
||||
messages := []agentContext.Message{
|
||||
{
|
||||
Role: "user",
|
||||
Content: []agentContext.ContentPart{
|
||||
{
|
||||
Type: agentContext.ContentImageURL,
|
||||
ImageURL: &agentContext.ImageURL{
|
||||
URL: "__" + uploaderName + "://" + uploadedFile.ID,
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
// 3. Call Vision - should use agent since model doesn't support vision
|
||||
result, err := content.Vision(ctx, capabilities, messages, uses)
|
||||
if err != nil {
|
||||
t.Fatalf("Vision function failed: %v", err)
|
||||
}
|
||||
|
||||
if len(result) != 1 {
|
||||
t.Fatalf("Expected 1 message, got %d", len(result))
|
||||
}
|
||||
|
||||
// 4. Verify result is text (processed by vision agent)
|
||||
contentParts, ok := result[0].Content.([]agentContext.ContentPart)
|
||||
if !ok {
|
||||
t.Fatalf("Expected content to be []ContentPart, got %T", result[0].Content)
|
||||
}
|
||||
|
||||
if len(contentParts) != 1 {
|
||||
t.Fatalf("Expected 1 content part, got %d", len(contentParts))
|
||||
}
|
||||
|
||||
if contentParts[0].Type != agentContext.ContentText {
|
||||
t.Errorf("Expected ContentText (from agent), got: %s", contentParts[0].Type)
|
||||
}
|
||||
|
||||
if contentParts[0].Text == "" {
|
||||
t.Error("Expected non-empty text from vision agent processing")
|
||||
}
|
||||
|
||||
t.Logf("✓ Vision agent processed image to text: %d characters", len(contentParts[0].Text))
|
||||
t.Logf("Agent response text:\n%s", contentParts[0].Text)
|
||||
}
|
||||
|
||||
// TestVision_CachedContent tests that file content is cached and reused
|
||||
func TestVision_CachedContent(t *testing.T) {
|
||||
testutils.Prepare(t)
|
||||
defer testutils.Clean(t)
|
||||
|
||||
// Setup test uploader
|
||||
uploaderName := "test-vision-cache"
|
||||
manager := setupTestUploader(t, uploaderName)
|
||||
defer cleanupTestUploader(uploaderName)
|
||||
|
||||
// 1. Upload a text file
|
||||
testContent := "Test content for caching verification"
|
||||
|
||||
reader := strings.NewReader(testContent)
|
||||
fileHeader := &attachment.FileHeader{
|
||||
FileHeader: &multipart.FileHeader{
|
||||
Filename: "cache-test.txt",
|
||||
Size: int64(len(testContent)),
|
||||
Header: make(map[string][]string),
|
||||
},
|
||||
}
|
||||
fileHeader.Header.Set("Content-Type", "text/plain")
|
||||
|
||||
uploadedFile, err := manager.Upload(context.Background(), fileHeader, reader, attachment.UploadOption{
|
||||
Groups: []string{"vision", "test"},
|
||||
})
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to upload file: %v", err)
|
||||
}
|
||||
|
||||
// 2. Prepare Vision context with same file referenced twice
|
||||
ctx := agentContext.New(context.Background(), nil, "test")
|
||||
|
||||
// Construct simple capabilities (text files don't need vision)
|
||||
capabilities := &openai.Capabilities{}
|
||||
|
||||
messages := []agentContext.Message{
|
||||
{
|
||||
Role: "user",
|
||||
Content: []agentContext.ContentPart{
|
||||
{
|
||||
Type: agentContext.ContentFile,
|
||||
File: &agentContext.FileAttachment{
|
||||
URL: "__" + uploaderName + "://" + uploadedFile.ID,
|
||||
Filename: "cache-test.txt",
|
||||
},
|
||||
},
|
||||
{
|
||||
Type: agentContext.ContentFile,
|
||||
File: &agentContext.FileAttachment{
|
||||
URL: "__" + uploaderName + "://" + uploadedFile.ID, // Same file
|
||||
Filename: "cache-test.txt",
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
// 3. Call Vision
|
||||
result, err := content.Vision(ctx, capabilities, messages, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("Vision function failed: %v", err)
|
||||
}
|
||||
|
||||
if len(result) != 1 {
|
||||
t.Fatalf("Expected 1 message, got %d", len(result))
|
||||
}
|
||||
|
||||
// 4. Verify both file references were processed
|
||||
contentParts, ok := result[0].Content.([]agentContext.ContentPart)
|
||||
if !ok {
|
||||
t.Fatalf("Expected content to be []ContentPart")
|
||||
}
|
||||
|
||||
if len(contentParts) != 2 {
|
||||
t.Fatalf("Expected 2 content parts (both files), got %d", len(contentParts))
|
||||
}
|
||||
|
||||
// Both should be text with same content
|
||||
if contentParts[0].Type != agentContext.ContentText {
|
||||
t.Errorf("First part: expected ContentText, got %s", contentParts[0].Type)
|
||||
}
|
||||
|
||||
if contentParts[1].Type != agentContext.ContentText {
|
||||
t.Errorf("Second part: expected ContentText, got %s", contentParts[1].Type)
|
||||
}
|
||||
|
||||
if !strings.Contains(contentParts[0].Text, testContent) {
|
||||
t.Errorf("First part text doesn't contain expected content")
|
||||
}
|
||||
|
||||
if !strings.Contains(contentParts[1].Text, testContent) {
|
||||
t.Errorf("Second part text doesn't contain expected content")
|
||||
}
|
||||
|
||||
// Verify content was cached (check attachment manager)
|
||||
cachedText, err := manager.GetText(context.Background(), uploadedFile.ID)
|
||||
if err != nil {
|
||||
t.Fatalf("Failed to get cached text: %v", err)
|
||||
}
|
||||
|
||||
if cachedText == "" {
|
||||
t.Error("Expected content to be cached in attachment manager")
|
||||
}
|
||||
|
||||
t.Logf("✓ Content successfully cached and reused: %d characters", len(cachedText))
|
||||
}
|
||||
48
agent/content/excel.go
Normal file
48
agent/content/excel.go
Normal file
|
|
@ -0,0 +1,48 @@
|
|||
package content
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"github.com/yaoapp/gou/connector/openai"
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
)
|
||||
|
||||
// ExcelHandler handles Microsoft Excel spreadsheets
|
||||
type ExcelHandler struct{}
|
||||
|
||||
// CanHandle checks if this handler can handle the content type
|
||||
func (h *ExcelHandler) CanHandle(contentType string, fileType FileType) bool {
|
||||
return fileType == FileTypeExcel ||
|
||||
contentType == "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" ||
|
||||
contentType == "application/vnd.ms-excel" ||
|
||||
strings.Contains(contentType, "excel") ||
|
||||
strings.Contains(contentType, "spreadsheet")
|
||||
}
|
||||
|
||||
// Handle processes Excel spreadsheet content
|
||||
func (h *ExcelHandler) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses) (*Result, error) {
|
||||
// TODO: Implement Excel handling
|
||||
// 1. Extract data from .xlsx or .xls file
|
||||
// 2. Convert to text format (e.g., CSV-like or structured text)
|
||||
// 3. Handle multiple sheets
|
||||
// 4. Return Result with formatted text
|
||||
return nil, fmt.Errorf("not implemented")
|
||||
}
|
||||
|
||||
// extractExcelText extracts text from Excel file
|
||||
func extractExcelText(data []byte, contentType string) (string, error) {
|
||||
// TODO: Implement Excel text extraction
|
||||
// Handle both .xls (old format) and .xlsx (new format)
|
||||
// Consider using libraries like:
|
||||
// - github.com/360EntSecGroup-Skylar/excelize for .xlsx
|
||||
// Format output as readable text or CSV
|
||||
return "", fmt.Errorf("not implemented")
|
||||
}
|
||||
|
||||
// formatExcelAsText formats Excel data as readable text
|
||||
func formatExcelAsText(sheets map[string][][]string) string {
|
||||
// TODO: Format multiple sheets into readable text
|
||||
// Include sheet names, headers, and data
|
||||
return ""
|
||||
}
|
||||
90
agent/content/fetch.go
Normal file
90
agent/content/fetch.go
Normal file
|
|
@ -0,0 +1,90 @@
|
|||
package content
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
"github.com/yaoapp/yao/attachment"
|
||||
)
|
||||
|
||||
// DefaultFetcher implements the Fetcher interface
|
||||
type DefaultFetcher struct{}
|
||||
|
||||
// NewFetcher creates a new default fetcher
|
||||
func NewFetcher() Fetcher {
|
||||
return &DefaultFetcher{}
|
||||
}
|
||||
|
||||
// Fetch retrieves content from HTTP URL or uploader wrapper
|
||||
func (f *DefaultFetcher) Fetch(ctx *agentContext.Context, source Source, url string) (*Info, error) {
|
||||
switch source {
|
||||
case SourceHTTP:
|
||||
return f.fetchHTTP(ctx, url)
|
||||
case SourceUploader:
|
||||
return f.fetchUploader(ctx, url)
|
||||
default:
|
||||
return nil, fmt.Errorf("unsupported source: %s", source)
|
||||
}
|
||||
}
|
||||
|
||||
// fetchHTTP fetches content from an HTTP(S) URL
|
||||
func (f *DefaultFetcher) fetchHTTP(ctx *agentContext.Context, url string) (*Info, error) {
|
||||
// TODO: Implement HTTP fetch logic
|
||||
// 1. Download file from URL
|
||||
// 2. Detect content type
|
||||
// 3. Detect file type based on content type and extension
|
||||
// 4. Return Info with data
|
||||
return nil, fmt.Errorf("not implemented")
|
||||
}
|
||||
|
||||
// fetchUploader fetches content from uploader wrapper (__uploader://fileid)
|
||||
func (f *DefaultFetcher) fetchUploader(ctx *agentContext.Context, wrapper string) (*Info, error) {
|
||||
// 1. Parse wrapper to get uploader name and file ID
|
||||
uploaderName, fileID, ok := attachment.Parse(wrapper)
|
||||
if !ok {
|
||||
return nil, fmt.Errorf("invalid uploader wrapper format: %s", wrapper)
|
||||
}
|
||||
|
||||
// 2. Get attachment manager
|
||||
var manager attachment.FileManager
|
||||
var exists bool
|
||||
|
||||
// Try to get manager by name
|
||||
manager, exists = attachment.Managers[uploaderName]
|
||||
if !exists {
|
||||
return nil, fmt.Errorf("uploader '%s' not found", uploaderName)
|
||||
}
|
||||
|
||||
// 3. Get file info
|
||||
file, err := manager.Info(ctx.Context, fileID)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to get file info: %w", err)
|
||||
}
|
||||
|
||||
// 4. Read file content
|
||||
data, err := manager.Read(ctx.Context, fileID)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to read file: %w", err)
|
||||
}
|
||||
|
||||
// 5. Return Info with data
|
||||
return &Info{
|
||||
Data: data,
|
||||
ContentType: file.ContentType,
|
||||
FileType: DetectFileType(file.ContentType, file.Filename),
|
||||
}, nil
|
||||
}
|
||||
|
||||
// parseUploaderWrapper parses uploader wrapper format: __uploader://fileid
|
||||
func parseUploaderWrapper(wrapper string) (uploaderName, fileID string, err error) {
|
||||
// TODO: Implement wrapper parsing
|
||||
// Format: __uploader://fileid
|
||||
return "", "", fmt.Errorf("not implemented")
|
||||
}
|
||||
|
||||
// detectFileType detects file type from content type and data
|
||||
func detectFileType(contentType string, data []byte) FileType {
|
||||
// TODO: Implement file type detection
|
||||
// Based on content type and magic bytes
|
||||
return FileTypeUnknown
|
||||
}
|
||||
173
agent/content/image.go
Normal file
173
agent/content/image.go
Normal file
|
|
@ -0,0 +1,173 @@
|
|||
package content
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"github.com/yaoapp/gou/connector/openai"
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
)
|
||||
|
||||
// ImageHandler handles image content
|
||||
type ImageHandler struct{}
|
||||
|
||||
// CanHandle checks if this handler can handle the content type
|
||||
func (h *ImageHandler) CanHandle(contentType string, fileType FileType) bool {
|
||||
return fileType == FileTypeImage || strings.HasPrefix(contentType, "image/")
|
||||
}
|
||||
|
||||
// Handle processes image content
|
||||
// Logic:
|
||||
// 1. If model supports vision -> convert to base64 or image_url format
|
||||
// 2. If model doesn't support vision -> use agent/MCP specified in uses.Vision
|
||||
func (h *ImageHandler) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses) (*Result, error) {
|
||||
if len(info.Data) == 0 {
|
||||
return nil, fmt.Errorf("no image data to process")
|
||||
}
|
||||
|
||||
if capabilities == nil {
|
||||
return nil, fmt.Errorf("no capabilities provided")
|
||||
}
|
||||
|
||||
// Check if model supports vision
|
||||
supportsVision, visionFormat := agentContext.GetVisionSupport(capabilities)
|
||||
|
||||
if supportsVision {
|
||||
// Model supports vision - return as image_url ContentPart
|
||||
contentPart, err := h.handleWithVisionModel(ctx, info, visionFormat)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to handle image with vision model: %w", err)
|
||||
}
|
||||
return &Result{
|
||||
ContentPart: contentPart,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// Model doesn't support vision - use vision agent/MCP
|
||||
visionTool := ""
|
||||
if uses != nil && uses.Vision != "" {
|
||||
visionTool = uses.Vision
|
||||
}
|
||||
|
||||
if visionTool == "" {
|
||||
return nil, fmt.Errorf("model doesn't support vision and no vision tool specified in uses.Vision")
|
||||
}
|
||||
|
||||
// Call vision agent/MCP to extract text
|
||||
text, err := h.handleWithVisionAgent(ctx, info, visionTool)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to handle image with vision agent/MCP: %w", err)
|
||||
}
|
||||
|
||||
return &Result{
|
||||
Text: text,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// handleWithVisionModel processes image using model's vision capability
|
||||
func (h *ImageHandler) handleWithVisionModel(ctx *agentContext.Context, info *Info, format agentContext.VisionFormat) (*agentContext.ContentPart, error) {
|
||||
// Encode image to base64
|
||||
base64Data := encodeImageBase64(info.Data, info.ContentType)
|
||||
|
||||
// Format according to model's vision format
|
||||
switch format {
|
||||
case agentContext.VisionFormatOpenAI:
|
||||
// OpenAI format: image_url with data URI
|
||||
return &agentContext.ContentPart{
|
||||
Type: agentContext.ContentImageURL,
|
||||
ImageURL: &agentContext.ImageURL{
|
||||
URL: base64Data,
|
||||
Detail: agentContext.DetailAuto,
|
||||
},
|
||||
}, nil
|
||||
|
||||
case agentContext.VisionFormatClaude:
|
||||
// Claude format: also uses image_url but may have different handling
|
||||
// For now, use the same format as OpenAI
|
||||
return &agentContext.ContentPart{
|
||||
Type: agentContext.ContentImageURL,
|
||||
ImageURL: &agentContext.ImageURL{
|
||||
URL: base64Data,
|
||||
Detail: agentContext.DetailAuto,
|
||||
},
|
||||
}, nil
|
||||
|
||||
case agentContext.VisionFormatDefault, "":
|
||||
// Default format (when Vision: true) - use OpenAI format
|
||||
return &agentContext.ContentPart{
|
||||
Type: agentContext.ContentImageURL,
|
||||
ImageURL: &agentContext.ImageURL{
|
||||
URL: base64Data,
|
||||
Detail: agentContext.DetailAuto,
|
||||
},
|
||||
}, nil
|
||||
|
||||
default:
|
||||
return nil, fmt.Errorf("unsupported vision format: %s", format)
|
||||
}
|
||||
}
|
||||
|
||||
// handleWithVisionAgent processes image using vision agent or MCP
|
||||
func (h *ImageHandler) handleWithVisionAgent(ctx *agentContext.Context, info *Info, visionTool string) (string, error) {
|
||||
// Parse vision tool format
|
||||
// Format can be:
|
||||
// - "agent_id" (call agent)
|
||||
// - "mcp:server_id" (call MCP tool)
|
||||
if strings.HasPrefix(visionTool, "mcp:") {
|
||||
// MCP tool
|
||||
serverID := strings.TrimPrefix(visionTool, "mcp:")
|
||||
return h.callMCPVisionTool(ctx, serverID, info)
|
||||
}
|
||||
|
||||
// Agent call
|
||||
return h.callVisionAgent(ctx, visionTool, info)
|
||||
}
|
||||
|
||||
// callVisionAgent calls a vision agent to describe the image
|
||||
func (h *ImageHandler) callVisionAgent(ctx *agentContext.Context, agentID string, info *Info) (string, error) {
|
||||
// Prepare message with image
|
||||
base64Data := EncodeToBase64DataURI(info.Data, info.ContentType)
|
||||
|
||||
message := agentContext.Message{
|
||||
Role: agentContext.RoleUser,
|
||||
Content: []agentContext.ContentPart{
|
||||
{
|
||||
Type: agentContext.ContentText,
|
||||
Text: "Please describe this image in detail.",
|
||||
},
|
||||
{
|
||||
Type: agentContext.ContentImageURL,
|
||||
ImageURL: &agentContext.ImageURL{
|
||||
URL: base64Data,
|
||||
Detail: agentContext.DetailAuto,
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
|
||||
return CallAgent(ctx, agentID, message)
|
||||
}
|
||||
|
||||
// callMCPVisionTool calls an MCP vision tool to describe the image
|
||||
func (h *ImageHandler) callMCPVisionTool(ctx *agentContext.Context, serverID string, info *Info) (string, error) {
|
||||
// Prepare base64 encoded image for MCP tool
|
||||
base64Data := EncodeToBase64DataURI(info.Data, info.ContentType)
|
||||
|
||||
// Prepare arguments for MCP tool
|
||||
arguments := map[string]interface{}{
|
||||
"image": base64Data,
|
||||
"content_type": info.ContentType,
|
||||
}
|
||||
|
||||
// Call MCP tool (typically "describe_image" or similar)
|
||||
return CallMCPTool(ctx, serverID, "describe_image", arguments)
|
||||
}
|
||||
|
||||
// encodeImageBase64 encodes image data to base64 with data URI prefix
|
||||
func encodeImageBase64(data []byte, contentType string) string {
|
||||
// Use the common function
|
||||
if contentType == "" {
|
||||
contentType = "image/png" // default for images
|
||||
}
|
||||
return EncodeToBase64DataURI(data, contentType)
|
||||
}
|
||||
265
agent/content/image_test.go
Normal file
265
agent/content/image_test.go
Normal file
|
|
@ -0,0 +1,265 @@
|
|||
package content
|
||||
|
||||
import (
|
||||
stdContext "context"
|
||||
"encoding/base64"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/yaoapp/gou/connector/openai"
|
||||
"github.com/yaoapp/gou/plan"
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
"github.com/yaoapp/yao/config"
|
||||
"github.com/yaoapp/yao/openapi/oauth/types"
|
||||
"github.com/yaoapp/yao/test"
|
||||
)
|
||||
|
||||
func TestMain(m *testing.M) {
|
||||
// Setup test environment
|
||||
test.Prepare(nil, config.Conf)
|
||||
defer test.Clean()
|
||||
|
||||
// Run tests
|
||||
code := m.Run()
|
||||
os.Exit(code)
|
||||
}
|
||||
|
||||
// newTestContext creates a Context for testing with commonly used fields pre-populated
|
||||
func newTestContext(capabilities *openai.Capabilities) *agentContext.Context {
|
||||
return &agentContext.Context{
|
||||
Context: stdContext.Background(),
|
||||
Space: plan.NewMemorySharedSpace(),
|
||||
ChatID: "test-chat",
|
||||
AssistantID: "test-assistant",
|
||||
Connector: "openai",
|
||||
Locale: "en-us",
|
||||
Theme: "light",
|
||||
Client: agentContext.Client{
|
||||
Type: "web",
|
||||
UserAgent: "TestAgent/1.0",
|
||||
IP: "127.0.0.1",
|
||||
},
|
||||
Referer: agentContext.RefererAPI,
|
||||
Accept: agentContext.AcceptWebCUI,
|
||||
Route: "",
|
||||
Metadata: make(map[string]interface{}),
|
||||
Capabilities: capabilities,
|
||||
Authorized: &types.AuthorizedInfo{
|
||||
Subject: "test-user",
|
||||
ClientID: "test-client-id",
|
||||
UserID: "test-user-123",
|
||||
TeamID: "test-team-456",
|
||||
TenantID: "test-tenant-789",
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func TestImageHandler_CanHandle(t *testing.T) {
|
||||
handler := &ImageHandler{}
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
contentType string
|
||||
fileType FileType
|
||||
want bool
|
||||
}{
|
||||
{"PNG image", "image/png", FileTypeImage, true},
|
||||
{"JPEG image", "image/jpeg", FileTypeImage, true},
|
||||
{"GIF image", "image/gif", FileTypeImage, true},
|
||||
{"WebP image", "image/webp", FileTypeImage, true},
|
||||
{"Text (should not handle)", "text/plain", FileTypeText, false},
|
||||
{"PDF (should not handle)", "application/pdf", FileTypePDF, false},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := handler.CanHandle(tt.contentType, tt.fileType)
|
||||
if got != tt.want {
|
||||
t.Errorf("CanHandle(%q, %q) = %v, want %v", tt.contentType, tt.fileType, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestImageHandler_Handle_WithVisionSupport(t *testing.T) {
|
||||
test.Prepare(t, config.Conf)
|
||||
defer test.Clean()
|
||||
|
||||
handler := &ImageHandler{}
|
||||
|
||||
// Create a simple test image (1x1 red PNG)
|
||||
pngData := createTestPNG()
|
||||
|
||||
// Create capabilities with vision support
|
||||
capabilities := &openai.Capabilities{
|
||||
Vision: "openai", // Vision is enabled with OpenAI format
|
||||
}
|
||||
|
||||
// Create test context
|
||||
ctx := newTestContext(capabilities)
|
||||
|
||||
info := &Info{
|
||||
FileType: FileTypeImage,
|
||||
ContentType: "image/png",
|
||||
Data: pngData,
|
||||
}
|
||||
|
||||
result, err := handler.Handle(ctx, info, capabilities, nil)
|
||||
if err != nil {
|
||||
t.Fatalf("Handle() error = %v", err)
|
||||
}
|
||||
|
||||
if result == nil {
|
||||
t.Fatal("Expected non-nil result")
|
||||
}
|
||||
|
||||
if result.ContentPart == nil {
|
||||
t.Fatal("Expected ContentPart for vision-supported model")
|
||||
}
|
||||
|
||||
if result.ContentPart.Type != agentContext.ContentImageURL {
|
||||
t.Errorf("Expected ContentPart type = %v, got %v", agentContext.ContentImageURL, result.ContentPart.Type)
|
||||
}
|
||||
|
||||
if result.ContentPart.ImageURL == nil {
|
||||
t.Fatal("Expected ImageURL to be set")
|
||||
}
|
||||
|
||||
// Verify base64 encoding
|
||||
if result.ContentPart.ImageURL.URL == "" {
|
||||
t.Error("Expected non-empty URL")
|
||||
}
|
||||
|
||||
// Should be data URI format
|
||||
if len(result.ContentPart.ImageURL.URL) < 20 {
|
||||
t.Error("Expected data URI to be longer")
|
||||
}
|
||||
}
|
||||
|
||||
func TestImageHandler_Handle_WithoutVisionSupport(t *testing.T) {
|
||||
test.Prepare(t, config.Conf)
|
||||
defer test.Clean()
|
||||
|
||||
handler := &ImageHandler{}
|
||||
|
||||
// Create a simple test image
|
||||
pngData := createTestPNG()
|
||||
|
||||
// Create capabilities WITHOUT vision support
|
||||
capabilities := &openai.Capabilities{
|
||||
Vision: nil, // No vision support
|
||||
}
|
||||
|
||||
// Create test context
|
||||
ctx := newTestContext(capabilities)
|
||||
|
||||
info := &Info{
|
||||
FileType: FileTypeImage,
|
||||
ContentType: "image/png",
|
||||
Data: pngData,
|
||||
}
|
||||
|
||||
// Should return error because no vision support and no tool
|
||||
_, err := handler.Handle(ctx, info, capabilities, nil)
|
||||
if err == nil {
|
||||
t.Error("Expected error when no vision support and no tool specified")
|
||||
}
|
||||
}
|
||||
|
||||
func TestImageHandler_Handle_EmptyData(t *testing.T) {
|
||||
test.Prepare(t, config.Conf)
|
||||
defer test.Clean()
|
||||
|
||||
handler := &ImageHandler{}
|
||||
|
||||
capabilities := &openai.Capabilities{
|
||||
Vision: "openai",
|
||||
}
|
||||
|
||||
// Create test context
|
||||
ctx := newTestContext(capabilities)
|
||||
|
||||
info := &Info{
|
||||
FileType: FileTypeImage,
|
||||
ContentType: "image/png",
|
||||
Data: []byte{}, // Empty data
|
||||
}
|
||||
|
||||
_, err := handler.Handle(ctx, info, capabilities, nil)
|
||||
if err == nil {
|
||||
t.Error("Expected error for empty image data")
|
||||
}
|
||||
}
|
||||
|
||||
func TestEncodeImageBase64(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
data []byte
|
||||
contentType string
|
||||
wantPrefix string
|
||||
}{
|
||||
{
|
||||
name: "PNG image",
|
||||
data: []byte{0x89, 0x50, 0x4E, 0x47}, // PNG magic number
|
||||
contentType: "image/png",
|
||||
wantPrefix: "data:image/png;base64,",
|
||||
},
|
||||
{
|
||||
name: "JPEG image",
|
||||
data: []byte{0xFF, 0xD8, 0xFF}, // JPEG magic number
|
||||
contentType: "image/jpeg",
|
||||
wantPrefix: "data:image/jpeg;base64,",
|
||||
},
|
||||
{
|
||||
name: "Empty content type defaults to PNG",
|
||||
data: []byte{0x01, 0x02, 0x03},
|
||||
contentType: "",
|
||||
wantPrefix: "data:image/png;base64,",
|
||||
},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result := encodeImageBase64(tt.data, tt.contentType)
|
||||
|
||||
// Check prefix
|
||||
if !strings.HasPrefix(result, tt.wantPrefix) {
|
||||
t.Errorf("Expected prefix %q, got %q", tt.wantPrefix, result[:len(tt.wantPrefix)])
|
||||
}
|
||||
|
||||
// Verify base64 encoding by decoding
|
||||
base64Part := result[len(tt.wantPrefix):]
|
||||
decoded, err := base64.StdEncoding.DecodeString(base64Part)
|
||||
if err != nil {
|
||||
t.Errorf("Failed to decode base64: %v", err)
|
||||
}
|
||||
|
||||
// Verify decoded data matches original
|
||||
if len(decoded) != len(tt.data) {
|
||||
t.Errorf("Decoded length = %d, want %d", len(decoded), len(tt.data))
|
||||
}
|
||||
for i := range decoded {
|
||||
if decoded[i] != tt.data[i] {
|
||||
t.Errorf("Decoded byte[%d] = %x, want %x", i, decoded[i], tt.data[i])
|
||||
}
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// createTestPNG creates a minimal valid PNG image (1x1 red pixel)
|
||||
func createTestPNG() []byte {
|
||||
// This is a minimal valid 1x1 red PNG image
|
||||
return []byte{
|
||||
0x89, 0x50, 0x4E, 0x47, 0x0D, 0x0A, 0x1A, 0x0A, // PNG signature
|
||||
0x00, 0x00, 0x00, 0x0D, 0x49, 0x48, 0x44, 0x52, // IHDR chunk
|
||||
0x00, 0x00, 0x00, 0x01, 0x00, 0x00, 0x00, 0x01, // 1x1 dimensions
|
||||
0x08, 0x02, 0x00, 0x00, 0x00, 0x90, 0x77, 0x53,
|
||||
0xDE, 0x00, 0x00, 0x00, 0x0C, 0x49, 0x44, 0x41, // IDAT chunk
|
||||
0x54, 0x08, 0xD7, 0x63, 0xF8, 0xCF, 0xC0, 0x00,
|
||||
0x00, 0x03, 0x01, 0x01, 0x00, 0x18, 0xDD, 0x8D,
|
||||
0xB4, 0x00, 0x00, 0x00, 0x00, 0x49, 0x45, 0x4E, // IEND chunk
|
||||
0x44, 0xAE, 0x42, 0x60, 0x82,
|
||||
}
|
||||
}
|
||||
25
agent/content/interfaces.go
Normal file
25
agent/content/interfaces.go
Normal file
|
|
@ -0,0 +1,25 @@
|
|||
package content
|
||||
|
||||
import (
|
||||
"github.com/yaoapp/gou/connector/openai"
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
)
|
||||
|
||||
// Handler defines the interface for handling different content types
|
||||
// Converts content (images, documents, etc.) to text or standard formats
|
||||
type Handler interface {
|
||||
// CanHandle checks if this handler can handle the given content type
|
||||
CanHandle(contentType string, fileType FileType) bool
|
||||
|
||||
// Handle converts the content and returns processed result
|
||||
// ctx: agent context (passed from Vision function)
|
||||
// capabilities: model capabilities (for vision/audio support detection)
|
||||
// uses: configuration for external tools (agents/MCP servers)
|
||||
Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses) (*Result, error)
|
||||
}
|
||||
|
||||
// Fetcher defines the interface for fetching content from different sources
|
||||
type Fetcher interface {
|
||||
// Fetch retrieves content from a URL or file ID
|
||||
Fetch(ctx *agentContext.Context, source Source, url string) (*Info, error)
|
||||
}
|
||||
50
agent/content/pdf.go
Normal file
50
agent/content/pdf.go
Normal file
|
|
@ -0,0 +1,50 @@
|
|||
package content
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"github.com/yaoapp/gou/connector/openai"
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
)
|
||||
|
||||
// PDFHandler handles PDF documents
|
||||
type PDFHandler struct{}
|
||||
|
||||
// CanHandle checks if this handler can handle the content type
|
||||
func (h *PDFHandler) CanHandle(contentType string, fileType FileType) bool {
|
||||
return fileType == FileTypePDF ||
|
||||
contentType == "application/pdf" ||
|
||||
strings.Contains(contentType, "pdf")
|
||||
}
|
||||
|
||||
// Handle processes PDF content
|
||||
// Logic:
|
||||
// 1. Check if uses.Vision is specified and supports PDF
|
||||
// 2. If yes, use vision tool to handle PDF (images + text)
|
||||
// 3. If no, extract text directly from PDF
|
||||
func (h *PDFHandler) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses) (*Result, error) {
|
||||
// TODO: Implement PDF handling
|
||||
// 1. Check if vision tool supports PDF
|
||||
// 2. If yes:
|
||||
// - Call vision tool to handle PDF (handles both text and images)
|
||||
// 3. If no:
|
||||
// - Extract text from PDF using default library
|
||||
// 4. Return Result with extracted text
|
||||
return nil, fmt.Errorf("not implemented")
|
||||
}
|
||||
|
||||
// extractPDFText extracts text content from PDF
|
||||
func extractPDFText(data []byte) (string, error) {
|
||||
// TODO: Implement PDF text extraction
|
||||
// Use a PDF library to extract text
|
||||
// Consider preserving layout/structure
|
||||
return "", fmt.Errorf("not implemented")
|
||||
}
|
||||
|
||||
// handleWithVisionTool processes PDF using vision tool (for PDFs with images)
|
||||
func handleWithVisionTool(ctx *agentContext.Context, data []byte, visionTool string) (string, error) {
|
||||
// TODO: Implement vision tool PDF processing
|
||||
// Some vision tools can handle PDF directly and extract both text and images
|
||||
return "", fmt.Errorf("not implemented")
|
||||
}
|
||||
47
agent/content/registry.go
Normal file
47
agent/content/registry.go
Normal file
|
|
@ -0,0 +1,47 @@
|
|||
package content
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
"github.com/yaoapp/gou/connector/openai"
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
)
|
||||
|
||||
// Registry holds all registered content handlers
|
||||
type Registry struct {
|
||||
handlers []Handler
|
||||
}
|
||||
|
||||
// NewRegistry creates a new handler registry with default handlers
|
||||
func NewRegistry() *Registry {
|
||||
return &Registry{
|
||||
handlers: []Handler{
|
||||
&ImageHandler{},
|
||||
&AudioHandler{},
|
||||
&PDFHandler{},
|
||||
&WordHandler{},
|
||||
&ExcelHandler{},
|
||||
&TextHandler{},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// GetHandler finds the appropriate handler for the given content
|
||||
func (r *Registry) GetHandler(contentType string, fileType FileType) Handler {
|
||||
for _, handler := range r.handlers {
|
||||
if handler.CanHandle(contentType, fileType) {
|
||||
return handler
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// Handle processes content using the appropriate handler
|
||||
func (r *Registry) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses) (*Result, error) {
|
||||
handler := r.GetHandler(info.ContentType, info.FileType)
|
||||
if handler == nil {
|
||||
return nil, fmt.Errorf("no handler found for content type: %s, file type: %s", info.ContentType, info.FileType)
|
||||
}
|
||||
|
||||
return handler.Handle(ctx, info, capabilities, uses)
|
||||
}
|
||||
123
agent/content/text.go
Normal file
123
agent/content/text.go
Normal file
|
|
@ -0,0 +1,123 @@
|
|||
package content
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"github.com/yaoapp/gou/connector/openai"
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
)
|
||||
|
||||
// TextHandler handles plain text, code files, CSV, JSON, XML, Markdown, etc.
|
||||
type TextHandler struct{}
|
||||
|
||||
// CanHandle checks if this handler can handle the content type
|
||||
func (h *TextHandler) CanHandle(contentType string, fileType FileType) bool {
|
||||
// Handle explicit text file types
|
||||
if fileType == FileTypeText || fileType == FileTypeCSV || fileType == FileTypeJSON {
|
||||
return true
|
||||
}
|
||||
|
||||
// Handle text-based MIME types
|
||||
if strings.HasPrefix(contentType, "text/") {
|
||||
return true
|
||||
}
|
||||
|
||||
// Handle common text-based content types
|
||||
textContentTypes := []string{
|
||||
"application/json",
|
||||
"application/xml",
|
||||
"application/javascript",
|
||||
"application/typescript",
|
||||
"application/x-yaml",
|
||||
"application/yaml",
|
||||
"application/toml",
|
||||
"application/x-sh",
|
||||
"application/x-python",
|
||||
"application/x-ruby",
|
||||
"application/x-perl",
|
||||
}
|
||||
|
||||
for _, ct := range textContentTypes {
|
||||
if contentType == ct || strings.Contains(contentType, ct) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
return false
|
||||
}
|
||||
|
||||
// Handle processes text content
|
||||
func (h *TextHandler) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses) (*Result, error) {
|
||||
if len(info.Data) == 0 {
|
||||
return nil, fmt.Errorf("no data to process")
|
||||
}
|
||||
|
||||
var text string
|
||||
var err error
|
||||
|
||||
// Handle different text formats
|
||||
switch {
|
||||
case info.FileType == FileTypeCSV || strings.Contains(info.ContentType, "csv"):
|
||||
// Format CSV as readable text (for now, just return as-is, can enhance later)
|
||||
text = string(info.Data)
|
||||
|
||||
case info.FileType == FileTypeJSON ||
|
||||
info.ContentType == "application/json" ||
|
||||
strings.Contains(info.ContentType, "json"):
|
||||
// Pretty print JSON
|
||||
text, err = formatJSONAsText(info.Data)
|
||||
if err != nil {
|
||||
// If JSON parsing fails, return raw text
|
||||
text = string(info.Data)
|
||||
}
|
||||
|
||||
case info.ContentType == "application/xml" ||
|
||||
strings.Contains(info.ContentType, "xml"):
|
||||
// For now, return XML as-is (can enhance formatting later)
|
||||
text = string(info.Data)
|
||||
|
||||
default:
|
||||
// Plain text, code files, markdown, etc.
|
||||
text, err = readTextContent(info.Data, info.ContentType)
|
||||
if err != nil {
|
||||
return nil, fmt.Errorf("failed to read text content: %w", err)
|
||||
}
|
||||
}
|
||||
|
||||
return &Result{
|
||||
Text: text,
|
||||
}, nil
|
||||
}
|
||||
|
||||
// readTextContent reads text content from data
|
||||
func readTextContent(data []byte, contentType string) (string, error) {
|
||||
// For now, assume UTF-8 encoding
|
||||
// TODO: Add encoding detection if needed (e.g., using golang.org/x/text/encoding)
|
||||
return string(data), nil
|
||||
}
|
||||
|
||||
// formatCSVAsText formats CSV data as readable text
|
||||
func formatCSVAsText(data []byte) (string, error) {
|
||||
// TODO: Parse CSV and format as readable table
|
||||
// Consider using encoding/csv package
|
||||
// For now, just return as-is
|
||||
return string(data), nil
|
||||
}
|
||||
|
||||
// formatJSONAsText formats JSON data as readable text
|
||||
func formatJSONAsText(data []byte) (string, error) {
|
||||
// Pretty print JSON with indentation
|
||||
var obj interface{}
|
||||
if err := json.Unmarshal(data, &obj); err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
pretty, err := json.MarshalIndent(obj, "", " ")
|
||||
if err != nil {
|
||||
return "", err
|
||||
}
|
||||
|
||||
return string(pretty), nil
|
||||
}
|
||||
152
agent/content/text_test.go
Normal file
152
agent/content/text_test.go
Normal file
|
|
@ -0,0 +1,152 @@
|
|||
package content
|
||||
|
||||
import (
|
||||
"testing"
|
||||
)
|
||||
|
||||
func TestTextHandler_CanHandle(t *testing.T) {
|
||||
handler := &TextHandler{}
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
contentType string
|
||||
fileType FileType
|
||||
want bool
|
||||
}{
|
||||
{"Plain text", "text/plain", FileTypeText, true},
|
||||
{"Markdown", "text/markdown", FileTypeText, true},
|
||||
{"HTML", "text/html", FileTypeText, true},
|
||||
{"JSON", "application/json", FileTypeJSON, true},
|
||||
{"JavaScript", "application/javascript", FileTypeText, true},
|
||||
{"TypeScript", "application/typescript", FileTypeText, true},
|
||||
{"YAML", "application/yaml", FileTypeText, true},
|
||||
{"CSV", "text/csv", FileTypeCSV, true},
|
||||
{"XML", "application/xml", FileTypeText, true},
|
||||
{"PDF (should not handle)", "application/pdf", FileTypePDF, false},
|
||||
{"Image (should not handle)", "image/png", FileTypeImage, false},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := handler.CanHandle(tt.contentType, tt.fileType)
|
||||
if got != tt.want {
|
||||
t.Errorf("CanHandle(%q, %q) = %v, want %v", tt.contentType, tt.fileType, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestTextHandler_Handle(t *testing.T) {
|
||||
handler := &TextHandler{}
|
||||
|
||||
tests := []struct {
|
||||
name string
|
||||
info *Info
|
||||
wantErr bool
|
||||
checkResult func(*testing.T, *Result)
|
||||
}{
|
||||
{
|
||||
name: "Plain text",
|
||||
info: &Info{
|
||||
FileType: FileTypeText,
|
||||
ContentType: "text/plain",
|
||||
Data: []byte("Hello, World!"),
|
||||
},
|
||||
wantErr: false,
|
||||
checkResult: func(t *testing.T, r *Result) {
|
||||
if r.Text != "Hello, World!" {
|
||||
t.Errorf("Expected 'Hello, World!', got %q", r.Text)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "JSON with pretty print",
|
||||
info: &Info{
|
||||
FileType: FileTypeJSON,
|
||||
ContentType: "application/json",
|
||||
Data: []byte(`{"name":"test","value":123}`),
|
||||
},
|
||||
wantErr: false,
|
||||
checkResult: func(t *testing.T, r *Result) {
|
||||
// Should be pretty printed
|
||||
if len(r.Text) <= len(`{"name":"test","value":123}`) {
|
||||
t.Errorf("JSON should be pretty printed, got: %q", r.Text)
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "Code file (Go)",
|
||||
info: &Info{
|
||||
FileType: FileTypeText,
|
||||
ContentType: "text/plain",
|
||||
Data: []byte("package main\n\nfunc main() {\n\tprintln(\"Hello\")\n}"),
|
||||
},
|
||||
wantErr: false,
|
||||
checkResult: func(t *testing.T, r *Result) {
|
||||
if r.Text == "" {
|
||||
t.Error("Expected non-empty text for Go code")
|
||||
}
|
||||
},
|
||||
},
|
||||
{
|
||||
name: "Empty data",
|
||||
info: &Info{
|
||||
FileType: FileTypeText,
|
||||
ContentType: "text/plain",
|
||||
Data: []byte{},
|
||||
},
|
||||
wantErr: true,
|
||||
},
|
||||
}
|
||||
|
||||
// Create test context
|
||||
testCtx := newTestContext(nil)
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
result, err := handler.Handle(testCtx, tt.info, nil, nil)
|
||||
if (err != nil) != tt.wantErr {
|
||||
t.Errorf("Handle() error = %v, wantErr %v", err, tt.wantErr)
|
||||
return
|
||||
}
|
||||
if !tt.wantErr && tt.checkResult != nil {
|
||||
tt.checkResult(t, result)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestDetectFileType(t *testing.T) {
|
||||
tests := []struct {
|
||||
name string
|
||||
contentType string
|
||||
filename string
|
||||
want FileType
|
||||
}{
|
||||
{"Go file", "text/plain", "main.go", FileTypeText},
|
||||
{"Python file", "text/plain", "script.py", FileTypeText},
|
||||
{"JavaScript file", "application/javascript", "app.js", FileTypeText},
|
||||
{"TypeScript file", "text/plain", "index.ts", FileTypeText},
|
||||
{"Markdown file", "text/markdown", "README.md", FileTypeText},
|
||||
{"JSON file", "application/json", "config.json", FileTypeJSON},
|
||||
{"YAML file", "text/plain", "config.yml", FileTypeText},
|
||||
{"PDF file", "application/pdf", "document.pdf", FileTypePDF},
|
||||
{"Image file", "image/png", "photo.png", FileTypeImage},
|
||||
{"CSV file", "text/csv", "data.csv", FileTypeCSV},
|
||||
{"XML file", "application/xml", "config.xml", FileTypeXML},
|
||||
{"Shell script", "text/plain", "script.sh", FileTypeText},
|
||||
{"Dockerfile", "text/plain", "Dockerfile", FileTypeText},
|
||||
{"gitignore", "text/plain", ".gitignore", FileTypeText},
|
||||
{"HTML", "text/html", "index.html", FileTypeText},
|
||||
{"CSS", "text/css", "styles.css", FileTypeText},
|
||||
}
|
||||
|
||||
for _, tt := range tests {
|
||||
t.Run(tt.name, func(t *testing.T) {
|
||||
got := DetectFileType(tt.contentType, tt.filename)
|
||||
if got != tt.want {
|
||||
t.Errorf("DetectFileType(%q, %q) = %v, want %v", tt.contentType, tt.filename, got, tt.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
190
agent/content/tools.go
Normal file
190
agent/content/tools.go
Normal file
|
|
@ -0,0 +1,190 @@
|
|||
package content
|
||||
|
||||
import (
|
||||
"context"
|
||||
"encoding/base64"
|
||||
"fmt"
|
||||
|
||||
jsoniter "github.com/json-iterator/go"
|
||||
"github.com/yaoapp/gou/mcp"
|
||||
"github.com/yaoapp/kun/log"
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
)
|
||||
|
||||
// AgentCaller interface for calling agents (to avoid circular dependency)
|
||||
type AgentCaller interface {
|
||||
Stream(ctx *agentContext.Context, messages []agentContext.Message) (interface{}, error)
|
||||
}
|
||||
|
||||
// AgentGetterFunc is a function type that gets an agent by ID
|
||||
var AgentGetterFunc func(agentID string) (AgentCaller, error)
|
||||
|
||||
// CallAgent calls an agent to process content (vision, audio, etc.)
|
||||
// This is a generic function that can be used by any handler
|
||||
func CallAgent(ctx *agentContext.Context, agentID string, message agentContext.Message) (string, error) {
|
||||
if AgentGetterFunc == nil {
|
||||
return "", fmt.Errorf("AgentGetterFunc not initialized")
|
||||
}
|
||||
|
||||
// Load the agent by ID using the injected function
|
||||
agent, err := AgentGetterFunc(agentID)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("failed to load agent %s: %w", agentID, err)
|
||||
}
|
||||
|
||||
// Call the agent with the message
|
||||
messages := []agentContext.Message{message}
|
||||
|
||||
connectorBackup := ctx.Connector
|
||||
ctx.Connector = ""
|
||||
defer func() {
|
||||
ctx.Connector = connectorBackup
|
||||
}()
|
||||
response, err := agent.Stream(ctx, messages)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("failed to call agent %s: %w", agentID, err)
|
||||
}
|
||||
|
||||
// Extract text from agent response
|
||||
// Two formats are supported:
|
||||
// 1. Custom Hook response (from Next hook)
|
||||
// 2. Standard Agent Stream response (LLM completion)
|
||||
|
||||
return extractTextFromAgentResponse(response)
|
||||
}
|
||||
|
||||
// extractTextFromAgentResponse extracts text from agent response
|
||||
// Handles two response formats:
|
||||
// 1. Custom Hook response: if it's a string, return directly; otherwise JSON stringify
|
||||
// 2. Standard response: extract from completion.content
|
||||
func extractTextFromAgentResponse(response interface{}) (string, error) {
|
||||
if response == nil {
|
||||
return "", fmt.Errorf("agent returned nil response")
|
||||
}
|
||||
|
||||
// Try to parse as standard response format (has "completion" field with LLM result)
|
||||
if responseMap, ok := response.(map[string]interface{}); ok {
|
||||
// Check for completion field (standard LLM response)
|
||||
if completion, hasCompletion := responseMap["completion"]; hasCompletion {
|
||||
if completionMap, ok := completion.(map[string]interface{}); ok {
|
||||
// Extract content from completion
|
||||
if content, hasContent := completionMap["content"]; hasContent {
|
||||
// Content can be string or structured
|
||||
switch v := content.(type) {
|
||||
case string:
|
||||
return v, nil
|
||||
case []interface{}:
|
||||
// Handle multimodal content array
|
||||
var text string
|
||||
for _, part := range v {
|
||||
if partMap, ok := part.(map[string]interface{}); ok {
|
||||
if partType, _ := partMap["type"].(string); partType == "text" {
|
||||
if textContent, ok := partMap["text"].(string); ok {
|
||||
text += textContent
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if text != "" {
|
||||
return text, nil
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Check for data field (custom hook response with data wrapper)
|
||||
if data, hasData := responseMap["data"]; hasData {
|
||||
// If data is a string, return directly
|
||||
if dataStr, ok := data.(string); ok {
|
||||
return dataStr, nil
|
||||
}
|
||||
// Otherwise, JSON stringify
|
||||
jsonBytes, err := jsoniter.Marshal(data)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("failed to serialize hook data response: %w", err)
|
||||
}
|
||||
return string(jsonBytes), nil
|
||||
}
|
||||
|
||||
// If the map itself looks like content, try to extract
|
||||
// This handles cases where the response is the content directly
|
||||
if content, hasContent := responseMap["content"]; hasContent {
|
||||
if contentStr, ok := content.(string); ok {
|
||||
return contentStr, nil
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Custom Hook response: if it's a plain string, return directly
|
||||
if responseStr, ok := response.(string); ok {
|
||||
return responseStr, nil
|
||||
}
|
||||
|
||||
// Otherwise, JSON stringify the response
|
||||
jsonBytes, err := jsoniter.Marshal(response)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("failed to serialize agent response: %w", err)
|
||||
}
|
||||
return string(jsonBytes), nil
|
||||
}
|
||||
|
||||
// CallMCPTool calls an MCP tool to process content
|
||||
// This is a generic function that can be used by any handler
|
||||
func CallMCPTool(ctx *agentContext.Context, serverID string, toolName string, arguments map[string]interface{}) (string, error) {
|
||||
// Get MCP context for cancellation/timeout control
|
||||
mcpCtx := ctx.Context
|
||||
if mcpCtx == nil {
|
||||
mcpCtx = context.Background()
|
||||
}
|
||||
|
||||
// Get MCP client
|
||||
client, err := mcp.Select(serverID)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("failed to select MCP client '%s': %w", serverID, err)
|
||||
}
|
||||
|
||||
// Call the tool
|
||||
log.Trace("[Content] Calling MCP tool: %s (server: %s)", toolName, serverID)
|
||||
callResult, err := client.CallTool(mcpCtx, toolName, arguments)
|
||||
if err != nil {
|
||||
return "", fmt.Errorf("MCP tool call failed: %w", err)
|
||||
}
|
||||
|
||||
// Check if result is an error
|
||||
if callResult.IsError {
|
||||
return "", fmt.Errorf("MCP tool returned error: %v", callResult.Content)
|
||||
}
|
||||
|
||||
// Extract text content from result
|
||||
// callResult.Content is []ToolContent
|
||||
var text string
|
||||
for _, content := range callResult.Content {
|
||||
if content.Type == "text" {
|
||||
text += content.Text
|
||||
}
|
||||
// Can also handle other types like image, resource if needed
|
||||
}
|
||||
|
||||
if text == "" {
|
||||
// If no text content found, return error
|
||||
return "", fmt.Errorf("MCP tool returned no text content")
|
||||
}
|
||||
|
||||
return text, nil
|
||||
}
|
||||
|
||||
// EncodeToBase64DataURI encodes data to base64 with data URI prefix
|
||||
// This is useful for encoding images, audio, or other binary data
|
||||
func EncodeToBase64DataURI(data []byte, contentType string) string {
|
||||
// Ensure we have a valid content type
|
||||
if contentType == "" {
|
||||
contentType = "application/octet-stream" // default
|
||||
}
|
||||
|
||||
// Encode to base64
|
||||
encoded := base64.StdEncoding.EncodeToString(data)
|
||||
|
||||
// Return data URI format
|
||||
return fmt.Sprintf("data:%s;base64,%s", contentType, encoded)
|
||||
}
|
||||
261
agent/content/types.go
Normal file
261
agent/content/types.go
Normal file
|
|
@ -0,0 +1,261 @@
|
|||
package content
|
||||
|
||||
import "github.com/yaoapp/yao/agent/context"
|
||||
|
||||
// FileType represents the type of file content
|
||||
type FileType string
|
||||
|
||||
const (
|
||||
// Image types
|
||||
FileTypeImage FileType = "image"
|
||||
|
||||
// Audio types
|
||||
FileTypeAudio FileType = "audio"
|
||||
|
||||
// Document types
|
||||
FileTypeText FileType = "text"
|
||||
FileTypePDF FileType = "pdf"
|
||||
FileTypeWord FileType = "word"
|
||||
FileTypeExcel FileType = "excel"
|
||||
FileTypePPT FileType = "ppt"
|
||||
FileTypeCSV FileType = "csv"
|
||||
|
||||
// Data types
|
||||
FileTypeJSON FileType = "json"
|
||||
FileTypeXML FileType = "xml"
|
||||
|
||||
// Binary
|
||||
FileTypeBinary FileType = "binary"
|
||||
|
||||
// Other
|
||||
FileTypeUnknown FileType = "unknown"
|
||||
)
|
||||
|
||||
// Source represents where the content comes from
|
||||
type Source string
|
||||
|
||||
const (
|
||||
SourceHTTP Source = "http" // HTTP(S) URL
|
||||
SourceUploader Source = "uploader" // Uploader wrapper: __uploader://fileid
|
||||
SourceBase64 Source = "base64" // Base64 encoded data
|
||||
SourceLocal Source = "local" // Local file path
|
||||
)
|
||||
|
||||
// Result represents the result of content handling
|
||||
type Result struct {
|
||||
Text string // Extracted text content
|
||||
ContentPart *context.ContentPart // Processed ContentPart (for model input)
|
||||
Metadata map[string]interface{} // Additional metadata
|
||||
Error error // Error if handling failed
|
||||
}
|
||||
|
||||
// Info holds information about a content part to be handled
|
||||
type Info struct {
|
||||
Source Source // Where the content comes from
|
||||
FileType FileType // Type of the file
|
||||
ContentType string // MIME content type
|
||||
URL string // Original URL or file ID
|
||||
Data []byte // File data (if already fetched)
|
||||
|
||||
// For uploader wrapper
|
||||
UploaderName string // Uploader name from wrapper
|
||||
FileID string // File ID from wrapper
|
||||
}
|
||||
|
||||
// DetectFileType detects file type from content type, filename, and file extension
|
||||
func DetectFileType(contentType, filename string) FileType {
|
||||
// Check by content type first
|
||||
switch {
|
||||
case contentType == "application/pdf":
|
||||
return FileTypePDF
|
||||
case contentType == "application/json":
|
||||
return FileTypeJSON
|
||||
case contentType == "application/xml" || contentType == "text/xml":
|
||||
return FileTypeXML
|
||||
case contentType == "text/csv":
|
||||
return FileTypeCSV
|
||||
case contentType == "application/vnd.openxmlformats-officedocument.wordprocessingml.document",
|
||||
contentType == "application/msword":
|
||||
return FileTypeWord
|
||||
case contentType == "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
||||
contentType == "application/vnd.ms-excel":
|
||||
return FileTypeExcel
|
||||
case contentType == "application/vnd.openxmlformats-officedocument.presentationml.presentation",
|
||||
contentType == "application/vnd.ms-powerpoint":
|
||||
return FileTypePPT
|
||||
}
|
||||
|
||||
// Check image types
|
||||
if isImageContentType(contentType) {
|
||||
return FileTypeImage
|
||||
}
|
||||
|
||||
// Check audio types
|
||||
if isAudioContentType(contentType) {
|
||||
return FileTypeAudio
|
||||
}
|
||||
|
||||
// Check text types
|
||||
if isTextContentType(contentType) {
|
||||
return FileTypeText
|
||||
}
|
||||
|
||||
// If content type doesn't help, check file extension
|
||||
if filename != "" {
|
||||
if ext := getFileExtension(filename); ext != "" {
|
||||
return detectTypeByExtension(ext)
|
||||
}
|
||||
}
|
||||
|
||||
return FileTypeUnknown
|
||||
}
|
||||
|
||||
// isImageContentType checks if content type is an image
|
||||
func isImageContentType(contentType string) bool {
|
||||
return contentType != "" &&
|
||||
(contentType == "image/png" ||
|
||||
contentType == "image/jpeg" ||
|
||||
contentType == "image/jpg" ||
|
||||
contentType == "image/gif" ||
|
||||
contentType == "image/webp" ||
|
||||
contentType == "image/svg+xml" ||
|
||||
contentType == "image/bmp")
|
||||
}
|
||||
|
||||
// isAudioContentType checks if content type is audio
|
||||
func isAudioContentType(contentType string) bool {
|
||||
return contentType != "" &&
|
||||
(contentType == "audio/mpeg" ||
|
||||
contentType == "audio/mp3" ||
|
||||
contentType == "audio/wav" ||
|
||||
contentType == "audio/ogg" ||
|
||||
contentType == "audio/flac" ||
|
||||
contentType == "audio/aac")
|
||||
}
|
||||
|
||||
// isTextContentType checks if content type is text-based
|
||||
func isTextContentType(contentType string) bool {
|
||||
if contentType == "" {
|
||||
return false
|
||||
}
|
||||
|
||||
// Common text MIME types
|
||||
textTypes := []string{
|
||||
"text/plain",
|
||||
"text/html",
|
||||
"text/css",
|
||||
"text/javascript",
|
||||
"text/markdown",
|
||||
"text/x-markdown",
|
||||
"application/javascript",
|
||||
"application/typescript",
|
||||
"application/x-yaml",
|
||||
"application/yaml",
|
||||
"application/toml",
|
||||
"application/x-sh",
|
||||
"application/x-python",
|
||||
"application/x-ruby",
|
||||
"application/x-perl",
|
||||
"application/x-php",
|
||||
"application/x-go",
|
||||
}
|
||||
|
||||
for _, t := range textTypes {
|
||||
if contentType == t {
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
return false
|
||||
}
|
||||
|
||||
// getFileExtension extracts file extension from filename (without dot)
|
||||
func getFileExtension(filename string) string {
|
||||
for i := len(filename) - 1; i >= 0; i-- {
|
||||
if filename[i] == '.' {
|
||||
return filename[i+1:]
|
||||
}
|
||||
if filename[i] == '/' || filename[i] == '\\' {
|
||||
break
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// detectTypeByExtension detects file type by file extension
|
||||
func detectTypeByExtension(ext string) FileType {
|
||||
// Normalize to lowercase
|
||||
ext = toLower(ext)
|
||||
|
||||
// Image extensions
|
||||
imageExts := []string{"png", "jpg", "jpeg", "gif", "webp", "svg", "bmp", "ico"}
|
||||
for _, e := range imageExts {
|
||||
if ext == e {
|
||||
return FileTypeImage
|
||||
}
|
||||
}
|
||||
|
||||
// Audio extensions
|
||||
audioExts := []string{"mp3", "wav", "ogg", "flac", "aac", "m4a"}
|
||||
for _, e := range audioExts {
|
||||
if ext == e {
|
||||
return FileTypeAudio
|
||||
}
|
||||
}
|
||||
|
||||
// Document extensions
|
||||
switch ext {
|
||||
case "pdf":
|
||||
return FileTypePDF
|
||||
case "doc", "docx":
|
||||
return FileTypeWord
|
||||
case "xls", "xlsx":
|
||||
return FileTypeExcel
|
||||
case "ppt", "pptx":
|
||||
return FileTypePPT
|
||||
case "csv":
|
||||
return FileTypeCSV
|
||||
case "json":
|
||||
return FileTypeJSON
|
||||
case "xml":
|
||||
return FileTypeXML
|
||||
}
|
||||
|
||||
// Code and text file extensions (very comprehensive list)
|
||||
textExts := []string{
|
||||
"txt", "text", "md", "markdown", "rst",
|
||||
// Programming languages
|
||||
"go", "py", "js", "ts", "jsx", "tsx", "java", "c", "cpp", "h", "hpp",
|
||||
"cs", "rb", "php", "pl", "swift", "kt", "rs", "scala", "clj",
|
||||
// Web
|
||||
"html", "htm", "css", "scss", "sass", "less",
|
||||
// Config
|
||||
"yaml", "yml", "toml", "ini", "conf", "config",
|
||||
// Shell
|
||||
"sh", "bash", "zsh", "fish",
|
||||
// Data
|
||||
"sql", "graphql", "proto",
|
||||
// Others
|
||||
"log", "gitignore", "env", "dockerfile",
|
||||
}
|
||||
for _, e := range textExts {
|
||||
if ext == e {
|
||||
return FileTypeText
|
||||
}
|
||||
}
|
||||
|
||||
return FileTypeUnknown
|
||||
}
|
||||
|
||||
// toLower converts ASCII string to lowercase (simple version)
|
||||
func toLower(s string) string {
|
||||
result := make([]byte, len(s))
|
||||
for i := 0; i < len(s); i++ {
|
||||
c := s[i]
|
||||
if c >= 'A' && c <= 'Z' {
|
||||
c += 'a' - 'A'
|
||||
}
|
||||
result[i] = c
|
||||
}
|
||||
return string(result)
|
||||
}
|
||||
39
agent/content/word.go
Normal file
39
agent/content/word.go
Normal file
|
|
@ -0,0 +1,39 @@
|
|||
package content
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
|
||||
"github.com/yaoapp/gou/connector/openai"
|
||||
agentContext "github.com/yaoapp/yao/agent/context"
|
||||
)
|
||||
|
||||
// WordHandler handles Microsoft Word documents
|
||||
type WordHandler struct{}
|
||||
|
||||
// CanHandle checks if this handler can handle the content type
|
||||
func (h *WordHandler) CanHandle(contentType string, fileType FileType) bool {
|
||||
return fileType == FileTypeWord ||
|
||||
contentType == "application/vnd.openxmlformats-officedocument.wordprocessingml.document" ||
|
||||
contentType == "application/msword" ||
|
||||
strings.Contains(contentType, "word")
|
||||
}
|
||||
|
||||
// Handle processes Word document content
|
||||
func (h *WordHandler) Handle(ctx *agentContext.Context, info *Info, capabilities *openai.Capabilities, uses *agentContext.Uses) (*Result, error) {
|
||||
// TODO: Implement Word document handling
|
||||
// 1. Extract text from .docx or .doc file
|
||||
// 2. Preserve formatting information if needed
|
||||
// 3. Return Result with extracted text
|
||||
return nil, fmt.Errorf("not implemented")
|
||||
}
|
||||
|
||||
// extractWordText extracts text from Word document
|
||||
func extractWordText(data []byte, contentType string) (string, error) {
|
||||
// TODO: Implement Word text extraction
|
||||
// Handle both .doc (old format) and .docx (new format)
|
||||
// Consider using libraries like:
|
||||
// - github.com/unidoc/unioffice for .docx
|
||||
// - Other libraries for .doc
|
||||
return "", fmt.Errorf("not implemented")
|
||||
}
|
||||
|
|
@ -23,6 +23,24 @@ func (ctx *Context) Send(msg *message.Message) error {
|
|||
// Skip lifecycle events for event-type messages (prevent recursion)
|
||||
isEventMessage := msg.Type == message.TypeEvent
|
||||
|
||||
// === Handle message_start event: record metadata for future delta chunks ===
|
||||
if isEventMessage && msg.Props != nil {
|
||||
if event, ok := msg.Props["event"].(string); ok && event == message.EventMessageStart {
|
||||
if data, ok := msg.Props["data"].(message.EventMessageStartData); ok {
|
||||
// Record metadata from message_start event
|
||||
if data.MessageID != "" && ctx.messageMetadata != nil {
|
||||
ctx.messageMetadata.setMessage(data.MessageID, &MessageMetadata{
|
||||
MessageID: data.MessageID,
|
||||
ThreadID: data.ThreadID,
|
||||
Type: data.Type,
|
||||
StartTime: time.Now(),
|
||||
ChunkCount: 0, // Will be incremented by delta chunks
|
||||
})
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// === Delta operations: Auto-inherit and update metadata ===
|
||||
if msg.Delta && msg.MessageID != "" && ctx.messageMetadata != nil {
|
||||
if metadata := ctx.getMessageMetadata(msg.MessageID); metadata != nil {
|
||||
|
|
@ -103,6 +121,7 @@ func (ctx *Context) Send(msg *message.Message) error {
|
|||
MessageID: msg.MessageID,
|
||||
Type: msg.Type,
|
||||
Timestamp: time.Now().UnixMilli(),
|
||||
ThreadID: msg.ThreadID, // Include ThreadID for concurrent stream identification
|
||||
}
|
||||
messageStartEvent := output.NewEventMessage(message.EventMessageStart, "Message started", messageStartData)
|
||||
if err := ctx.sendRaw(messageStartEvent); err != nil {
|
||||
|
|
@ -147,6 +166,7 @@ func (ctx *Context) Send(msg *message.Message) error {
|
|||
MessageID: msg.MessageID,
|
||||
Type: msg.Type,
|
||||
Timestamp: time.Now().UnixMilli(),
|
||||
ThreadID: metadata.ThreadID, // Include ThreadID for concurrent stream identification
|
||||
DurationMs: durationMs,
|
||||
ChunkCount: metadata.ChunkCount,
|
||||
Status: "completed",
|
||||
|
|
@ -191,6 +211,7 @@ func (ctx *Context) EndMessage(messageID string, content interface{}) error {
|
|||
MessageID: messageID,
|
||||
Type: metadata.Type,
|
||||
Timestamp: time.Now().UnixMilli(),
|
||||
ThreadID: metadata.ThreadID, // Include ThreadID for concurrent stream identification
|
||||
DurationMs: durationMs,
|
||||
ChunkCount: metadata.ChunkCount,
|
||||
Status: "completed",
|
||||
|
|
|
|||
|
|
@ -464,6 +464,8 @@ const (
|
|||
ContentText ContentPartType = "text" // Text content
|
||||
ContentImageURL ContentPartType = "image_url" // Image URL content (Vision)
|
||||
ContentInputAudio ContentPartType = "input_audio" // Input audio content (Audio)
|
||||
ContentFile ContentPartType = "file" // File attachment (documents, etc.)
|
||||
ContentData ContentPartType = "data" // Generic data content (base64, binary, etc.)
|
||||
)
|
||||
|
||||
// ContentPart represents a part of the message content (for multimodal messages)
|
||||
|
|
@ -473,6 +475,8 @@ type ContentPart struct {
|
|||
Text string `json:"text,omitempty"` // For type="text": the text content
|
||||
ImageURL *ImageURL `json:"image_url,omitempty"` // For type="image_url": the image URL
|
||||
InputAudio *InputAudio `json:"input_audio,omitempty"` // For type="input_audio": the input audio data
|
||||
File *FileAttachment `json:"file,omitempty"` // For type="file": file attachment
|
||||
Data *DataContent `json:"data,omitempty"` // For type="data": generic data content
|
||||
}
|
||||
|
||||
// ImageDetailLevel represents the detail level for image processing
|
||||
|
|
@ -497,6 +501,41 @@ type InputAudio struct {
|
|||
Format string `json:"format"` // Required: Audio format (e.g., "wav", "mp3")
|
||||
}
|
||||
|
||||
// FileAttachment represents a file attachment in the message content
|
||||
// Compatible with frontend InputArea format: { type: 'file', file: { url, filename } }
|
||||
type FileAttachment struct {
|
||||
URL string `json:"url"` // Required: URL of the file (http:// or __uploader://fileid wrapper)
|
||||
Filename string `json:"filename,omitempty"` // Optional: original filename
|
||||
}
|
||||
|
||||
// DataSourceType represents the type of data source
|
||||
type DataSourceType string
|
||||
|
||||
// Data source type constants
|
||||
const (
|
||||
DataSourceModel DataSourceType = "model" // Data model
|
||||
DataSourceKBCollection DataSourceType = "kb_collection" // Knowledge base collection
|
||||
DataSourceKBDocument DataSourceType = "kb_document" // Knowledge base document/file
|
||||
DataSourceTable DataSourceType = "table" // Database table
|
||||
DataSourceAPI DataSourceType = "api" // API endpoint
|
||||
DataSourceMCPResource DataSourceType = "mcp_resource" // MCP (Model Context Protocol) resource
|
||||
)
|
||||
|
||||
// DataSource represents a single data source reference
|
||||
type DataSource struct {
|
||||
Type DataSourceType `json:"type"` // Required: type of data source
|
||||
Name string `json:"name"` // Required: name/identifier of the data source
|
||||
ID string `json:"id,omitempty"` // Optional: specific ID (e.g., document ID, record ID)
|
||||
Filters map[string]interface{} `json:"filters,omitempty"` // Optional: filters to apply
|
||||
Metadata map[string]interface{} `json:"metadata,omitempty"` // Optional: additional metadata
|
||||
}
|
||||
|
||||
// DataContent represents data source references in the message
|
||||
// Used to reference data models, knowledge base collections, KB documents, etc.
|
||||
type DataContent struct {
|
||||
Sources []DataSource `json:"sources"` // Required: array of data source references
|
||||
}
|
||||
|
||||
// ToolCallType represents the type of tool call
|
||||
type ToolCallType string
|
||||
|
||||
|
|
|
|||
|
|
@ -318,6 +318,7 @@ type EventMessageStartData struct {
|
|||
MessageID string `json:"message_id"` // Message ID (M1, M2, M3...)
|
||||
Type string `json:"type"` // Message type: "text" | "thinking" | "tool_call" | "refusal"
|
||||
Timestamp int64 `json:"timestamp"` // Unix timestamp when message started
|
||||
ThreadID string `json:"thread_id,omitempty"` // Thread ID (optional; for concurrent streams)
|
||||
ToolCall *EventToolCallInfo `json:"tool_call,omitempty"` // Tool call metadata (if type is "tool_call")
|
||||
Extra map[string]interface{} `json:"extra,omitempty"` // Additional metadata (for custom providers or future extensions)
|
||||
}
|
||||
|
|
@ -329,6 +330,7 @@ type EventMessageEndData struct {
|
|||
MessageID string `json:"message_id"` // Message ID (M1, M2, M3...)
|
||||
Type string `json:"type"` // Message type (same as in message_start)
|
||||
Timestamp int64 `json:"timestamp"` // Unix timestamp when message ended
|
||||
ThreadID string `json:"thread_id,omitempty"` // Thread ID (optional; for concurrent streams)
|
||||
DurationMs int64 `json:"duration_ms"` // Duration of this message in milliseconds
|
||||
ChunkCount int `json:"chunk_count"` // Number of data chunks in this message
|
||||
Status string `json:"status"` // "completed" | "partial" | "error"
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue