feat(providers): Add vision and document processing capabilities to providers

2025-02-03 15:26:00 +01:00
parent e82c510094
commit eda8ce36df
9 changed files with 212 additions and 6 deletions
--- a/ts/provider.ollama.ts
+++ b/ts/provider.ollama.ts
@@ -6,18 +6,21 @@ import type { ChatOptions, ChatResponse, ChatMessage } from './abstract.classes.
 export interface IOllamaProviderOptions {
  baseUrl?: string;
  model?: string;
+  visionModel?: string; // Model to use for vision tasks (e.g. 'llava')
 }

 export class OllamaProvider extends MultiModalModel {
  private options: IOllamaProviderOptions;
  private baseUrl: string;
  private model: string;
+  private visionModel: string;

  constructor(optionsArg: IOllamaProviderOptions = {}) {
    super();
    this.options = optionsArg;
    this.baseUrl = optionsArg.baseUrl || 'http://localhost:11434';
    this.model = optionsArg.model || 'llama2';
+    this.visionModel = optionsArg.visionModel || 'llava';
  }

  async start() {
@@ -167,4 +170,83 @@ export class OllamaProvider extends MultiModalModel {
  public async audio(optionsArg: { message: string }): Promise<NodeJS.ReadableStream> {
    throw new Error('Audio generation is not supported by Ollama.');
  }
+
+  public async vision(optionsArg: { image: Buffer; prompt: string }): Promise<string> {
+    const base64Image = optionsArg.image.toString('base64');
+    
+    const response = await fetch(`${this.baseUrl}/api/chat`, {
+      method: 'POST',
+      headers: {
+        'Content-Type': 'application/json',
+      },
+      body: JSON.stringify({
+        model: this.visionModel,
+        messages: [{
+          role: 'user',
+          content: optionsArg.prompt,
+          images: [base64Image]
+        }],
+        stream: false
+      }),
+    });
+
+    if (!response.ok) {
+      throw new Error(`Ollama API error: ${response.statusText}`);
+    }
+
+    const result = await response.json();
+    return result.message.content;
+  }
+
+  public async document(optionsArg: {
+    systemMessage: string;
+    userMessage: string;
+    pdfDocuments: Uint8Array[];
+    messageHistory: ChatMessage[];
+  }): Promise<{ message: any }> {
+    // Convert PDF documents to images using SmartPDF
+    const smartpdfInstance = new plugins.smartpdf.SmartPdf();
+    let documentImageBytesArray: Uint8Array[] = [];
+
+    for (const pdfDocument of optionsArg.pdfDocuments) {
+      const documentImageArray = await smartpdfInstance.convertPDFToPngBytes(pdfDocument);
+      documentImageBytesArray = documentImageBytesArray.concat(documentImageArray);
+    }
+
+    // Convert images to base64
+    const base64Images = documentImageBytesArray.map(bytes => Buffer.from(bytes).toString('base64'));
+
+    // Send request to Ollama with images
+    const response = await fetch(`${this.baseUrl}/api/chat`, {
+      method: 'POST',
+      headers: {
+        'Content-Type': 'application/json',
+      },
+      body: JSON.stringify({
+        model: this.visionModel,
+        messages: [
+          { role: 'system', content: optionsArg.systemMessage },
+          ...optionsArg.messageHistory,
+          {
+            role: 'user',
+            content: optionsArg.userMessage,
+            images: base64Images
+          }
+        ],
+        stream: false
+      }),
+    });
+
+    if (!response.ok) {
+      throw new Error(`Ollama API error: ${response.statusText}`);
+    }
+
+    const result = await response.json();
+    return {
+      message: {
+        role: 'assistant',
+        content: result.message.content
+      }
+    };
+  }
 }