feat(providers): Add research API and image generation/editing support; extend providers and tests

2025-10-03 13:43:29 +00:00
parent fe8540c8ba
commit f0556e89f3
13 changed files with 612 additions and 194 deletions
--- a/ts/abstract.classes.multimodal.ts
+++ b/ts/abstract.classes.multimodal.ts
@@ -50,6 +50,60 @@ export interface ResearchResponse {
  metadata?: any;
 }

+/**
+ * Options for image generation
+ */
+export interface ImageGenerateOptions {
+  prompt: string;
+  model?: 'gpt-image-1' | 'dall-e-3' | 'dall-e-2';
+  quality?: 'low' | 'medium' | 'high' | 'standard' | 'hd' | 'auto';
+  size?: '256x256' | '512x512' | '1024x1024' | '1536x1024' | '1024x1536' | '1792x1024' | '1024x1792' | 'auto';
+  style?: 'vivid' | 'natural';
+  background?: 'transparent' | 'opaque' | 'auto';
+  outputFormat?: 'png' | 'jpeg' | 'webp';
+  outputCompression?: number; // 0-100 for webp/jpeg
+  moderation?: 'low' | 'auto';
+  n?: number; // Number of images to generate
+  stream?: boolean;
+  partialImages?: number; // 0-3 for streaming
+}
+
+/**
+ * Options for image editing
+ */
+export interface ImageEditOptions {
+  image: Buffer;
+  prompt: string;
+  mask?: Buffer;
+  model?: 'gpt-image-1' | 'dall-e-2';
+  quality?: 'low' | 'medium' | 'high' | 'standard' | 'auto';
+  size?: '256x256' | '512x512' | '1024x1024' | '1536x1024' | '1024x1536' | 'auto';
+  background?: 'transparent' | 'opaque' | 'auto';
+  outputFormat?: 'png' | 'jpeg' | 'webp';
+  outputCompression?: number;
+  n?: number;
+  stream?: boolean;
+  partialImages?: number;
+}
+
+/**
+ * Response format for image operations
+ */
+export interface ImageResponse {
+  images: Array<{
+    b64_json?: string;
+    url?: string;
+    revisedPrompt?: string;
+  }>;
+  metadata?: {
+    model: string;
+    quality?: string;
+    size?: string;
+    outputFormat?: string;
+    tokensUsed?: number;
+  };
+}
+
 /**
 * Abstract base class for multi-modal AI models.
 * Provides a common interface for different AI providers (OpenAI, Anthropic, Perplexity, Ollama)
@@ -131,4 +185,20 @@ export abstract class MultiModalModel {
   * @throws Error if the provider doesn't support research capabilities
   */
  public abstract research(optionsArg: ResearchOptions): Promise<ResearchResponse>;
+
+  /**
+   * Image generation from text prompts
+   * @param optionsArg Options containing the prompt and generation parameters
+   * @returns Promise resolving to the generated image(s)
+   * @throws Error if the provider doesn't support image generation
+   */
+  public abstract imageGenerate(optionsArg: ImageGenerateOptions): Promise<ImageResponse>;
+
+  /**
+   * Image editing and inpainting
+   * @param optionsArg Options containing the image, prompt, and editing parameters
+   * @returns Promise resolving to the edited image(s)
+   * @throws Error if the provider doesn't support image editing
+   */
+  public abstract imageEdit(optionsArg: ImageEditOptions): Promise<ImageResponse>;
 }