diff --git a/README.md b/README.md index a0fce0537..1b9450a0b 100644 --- a/README.md +++ b/README.md @@ -38,7 +38,9 @@ It supports extraction, crawling, scraping, and search — designed to scale fro 2. **[Scrape](https://docs.maxun.dev/robot/scrape/scrape-robots)** – Convert full webpages into clean Markdown or HTML and capture screenshots. 3. **[Crawl](https://docs.maxun.dev/robot/crawl/crawl-introduction)** – Crawl entire websites and extract content from every relevant page, with full control over scope and discovery. 4. **[Search](https://docs.maxun.dev/robot/search/search-introduction)** – Run automated web searches to discover or scrape results, with support for time-based filters. -5. **[SDK](https://docs.maxun.dev/sdk/sdk-overview)** – A complete developer toolkit for scraping, extraction, scheduling, and end-to-end data automation. +5. **[SDK](https://docs.maxun.dev/category/sdk)** – A complete developer toolkit for scraping, extraction, scheduling, and end-to-end data automation. +6. **[CLI](https://docs.maxun.dev/category/cli)** – Create robots, trigger runs, and retrieve extracted data from your terminal. + ## How Does It Work? diff --git a/maxun-core/src/interpret.ts b/maxun-core/src/interpret.ts index 521f8e798..d5406a878 100644 --- a/maxun-core/src/interpret.ts +++ b/maxun-core/src/interpret.ts @@ -2524,7 +2524,12 @@ export default class Interpreter extends EventEmitter { console.log("REPEAT COUNT", repeatCount); if (this.options.maxRepeats && repeatCount > this.options.maxRepeats) { - return; + + const failedAction = (action?.what?.find((w: any) => w?.action !== 'flag')?.action) || 'unknown'; + const maxRepeats = this.options.maxRepeats; + this.log(`Action ${String(failedAction)} exceeded max retries (${maxRepeats})`, Level.ERROR); + cleanup(); + throw new Error(`Action ${String(failedAction)} exceeded max retries (${maxRepeats})`); } lastAction = action; diff --git a/maxun-core/src/utils/concurrency.ts b/maxun-core/src/utils/concurrency.ts index 41fc10475..53f3a617e 100644 --- a/maxun-core/src/utils/concurrency.ts +++ b/maxun-core/src/utils/concurrency.ts @@ -18,9 +18,14 @@ export default class Concurrency { private jobQueue: Function[] = []; /** - * "Resolve" callbacks of the waitForCompletion() promises. + * Resolve/reject callbacks of the waitForCompletion() promises. */ - private waiting: Function[] = []; + private waiting: Array<{ resolve: () => void; reject: (error: Error) => void }> = []; + + /** + * First worker error captured during current execution wave. + */ + private firstError: Error | null = null; /** * Constructs a new instance of concurrency manager. @@ -42,7 +47,11 @@ export default class Concurrency { // console.debug("Job finished, running the next waiting job..."); this.runNextJob(); }).catch((error) => { - console.error(`Job failed with error: ${error.message}`); + const normalizedError = error instanceof Error ? error : new Error(String(error)); + console.error(`Job failed with error: ${normalizedError.message}`); + if (!this.firstError) { + this.firstError = normalizedError; + } // Continue processing other jobs even if one fails this.runNextJob(); }); @@ -51,7 +60,18 @@ export default class Concurrency { this.activeWorkers -= 1; if (this.activeWorkers === 0) { // console.debug("This concurrency manager is idle!"); - this.waiting.forEach((x) => x()); + const pending = [...this.waiting]; + this.waiting = []; + const pendingError = this.firstError; + this.firstError = null; + + pending.forEach(({ resolve, reject }) => { + if (pendingError) { + reject(pendingError); + } else { + resolve(); + } + }); } } } @@ -82,8 +102,8 @@ export default class Concurrency { * @returns Promise, resolved after there is no running/waiting worker. */ waitForCompletion(): Promise { - return new Promise((res) => { - this.waiting.push(res); + return new Promise((resolve, reject) => { + this.waiting.push({ resolve, reject }); }); } } diff --git a/package.json b/package.json index bbebd635c..392ffb6e9 100644 --- a/package.json +++ b/package.json @@ -1,6 +1,6 @@ { "name": "maxun", - "version": "0.0.35", + "version": "0.0.37", "author": "Maxun", "license": "AGPL-3.0-or-later", "dependencies": { @@ -88,7 +88,7 @@ "start": "npm run build:server && concurrently -k \"npm run server\" \"npm run client\"", "server": "cross-env NODE_OPTIONS='--max-old-space-size=4096' node server/dist/server/src/server.js", "start:dev": "concurrently -k \"npm run server:dev\" \"npm run client\"", - "server:dev": "cross-env NODE_OPTIONS='--max-old-space-size=2048' nodemon server/src/server.ts", + "server:dev": "cross-env NODE_OPTIONS='--max-old-space-size=4096' nodemon server/src/server.ts", "client": "vite", "build": "vite build", "build:server": "tsc -p server/tsconfig.json", diff --git a/public/locales/de.json b/public/locales/de.json index 12dba1047..f75cdb672 100644 --- a/public/locales/de.json +++ b/public/locales/de.json @@ -279,6 +279,7 @@ "errors": { "user_not_logged": "Benutzer nicht angemeldet. Aufnahme kann nicht gespeichert werden.", "exists_warning": "Ein Roboter mit diesem Namen existiert bereits, bitte bestätigen Sie das Überschreiben des Roboters.", + "name_exists": "Ein Roboter mit diesem Namen existiert bereits. Bitte wählen Sie einen anderen Namen.", "no_actions_performed": "Roboter kann nicht gespeichert werden. Bitte führen Sie mindestens eine Erfassungsaktion durch, bevor Sie speichern." }, "tooltips": { @@ -294,7 +295,8 @@ "notifications": { "terminated": "Aktuelle Aufnahme wurde beendet", "environment_reset": "Browser-Umgebung wurde zurückgesetzt", - "reset_successful": "Alle Aufnahmen erfolgreich zurückgesetzt und zum Ausgangszustand zurückgekehrt" + "reset_successful": "Alle Aufnahmen erfolgreich zurückgesetzt und zum Ausgangszustand zurückgekehrt", + "timeout_discarded": "Die Aufnahmesitzung ist nach 10 Minuten abgelaufen und wurde automatisch verworfen." } }, "interpretation_log": { @@ -440,7 +442,8 @@ "warning": "⚠️ Stellen Sie sicher, dass die neue Seite die gleiche Struktur wie die Originalseite hat." }, "fields": { - "target_url": "Roboter Ziel-URL" + "target_url": "Roboter Ziel-URL", + "new_name": "Roboter-Name" }, "buttons": { "duplicate": "Roboter duplizieren", @@ -449,6 +452,7 @@ "notifications": { "robot_not_found": "Roboterdetails konnten nicht gefunden werden. Bitte versuchen Sie es erneut.", "url_required": "Ziel-URL ist erforderlich.", + "name_required": "Roboter-Name ist erforderlich.", "duplicate_success": "Roboter erfolgreich dupliziert.", "duplicate_error": "Fehler beim Aktualisieren der Ziel-URL. Bitte versuchen Sie es erneut.", "unknown_error": "Beim Aktualisieren der Ziel-URL ist ein Fehler aufgetreten." @@ -551,6 +555,9 @@ "manual_run": "Manuelle Ausführung", "scheduled_run": "Geplante Ausführung", "api": "API", + "sdk": "SDK", + "mcp": "MCP", + "cli": "CLI", "unknown_run_type": "Unbekannter Ausführungstyp" }, "run_status_chips": { diff --git a/public/locales/en.json b/public/locales/en.json index 84f46bcd0..ff42d7b07 100644 --- a/public/locales/en.json +++ b/public/locales/en.json @@ -279,6 +279,7 @@ "errors": { "user_not_logged": "User not logged in. Cannot save recording.", "exists_warning": "Robot with this name already exists, please confirm the Robot's overwrite.", + "name_exists": "A robot with this name already exists. Please choose a different name.", "no_actions_performed": "Cannot save robot. Please perform at least one capture action before saving." }, "tooltips": { @@ -294,7 +295,8 @@ "notifications": { "terminated": "Current Recording was terminated", "environment_reset": "Browser environment has been reset", - "reset_successful": "Successfully reset all captures and returned to initial state" + "reset_successful": "Successfully reset all captures and returned to initial state", + "timeout_discarded": "Recording session timed out after 10 minutes and was automatically discarded." } }, "interpretation_log": { @@ -440,7 +442,8 @@ "warning": "⚠️ Ensure the new page has the same structure as the original page." }, "fields": { - "target_url": "Target URL" + "target_url": "Target URL", + "new_name": "Robot Name" }, "buttons": { "duplicate": "Duplicate Robot", @@ -449,6 +452,7 @@ "notifications": { "robot_not_found": "Could not find robot details. Please try again.", "url_required": "Target URL is required.", + "name_required": "Robot name is required.", "duplicate_success": "Robot duplicated successfully.", "duplicate_error": "Failed to update the Target URL. Please try again.", "unknown_error": "An error occurred while updating the Target URL." @@ -552,6 +556,8 @@ "scheduled_run": "Scheduled", "api": "API", "sdk": "SDK", + "mcp": "MCP", + "cli": "CLI", "unknown_run_type": "Unknown Run Type" }, "run_status_chips": { diff --git a/public/locales/es.json b/public/locales/es.json index 47ce46e3a..5421856aa 100644 --- a/public/locales/es.json +++ b/public/locales/es.json @@ -279,6 +279,7 @@ "errors": { "user_not_logged": "Usuario no conectado. No se puede guardar la grabación.", "exists_warning": "Ya existe un robot con este nombre, por favor confirme la sobrescritura del robot.", + "name_exists": "Ya existe un robot con este nombre. Por favor elige un nombre diferente.", "no_actions_performed": "No se puede guardar el robot. Por favor realice al menos una acción de captura antes de guardar." }, "tooltips": { @@ -294,7 +295,8 @@ "notifications": { "terminated": "La grabación actual fue terminada", "environment_reset": "El entorno del navegador ha sido reiniciado", - "reset_successful": "Se reiniciaron correctamente todas las capturas y se volvió al estado inicial" + "reset_successful": "Se reiniciaron correctamente todas las capturas y se volvió al estado inicial", + "timeout_discarded": "La sesión de grabación expiró después de 10 minutos y fue descartada automáticamente." } }, "interpretation_buttons": { @@ -440,7 +442,8 @@ "warning": "⚠️ Asegúrate de que la nueva página tenga la misma estructura que la página original." }, "fields": { - "target_url": "URL Destino del Robot" + "target_url": "URL Destino del Robot", + "new_name": "Nombre del Robot" }, "buttons": { "duplicate": "Duplicar Robot", @@ -449,6 +452,7 @@ "notifications": { "robot_not_found": "No se pudieron encontrar los detalles del robot. Por favor, inténtalo de nuevo.", "url_required": "Se requiere la URL de destino.", + "name_required": "El nombre del robot es obligatorio.", "duplicate_success": "Robot duplicado con éxito.", "duplicate_error": "Error al actualizar la URL de destino. Por favor, inténtalo de nuevo.", "unknown_error": "Ocurrió un error al actualizar la URL de destino." @@ -551,6 +555,9 @@ "manual_run": "Ejecución Manual", "scheduled_run": "Ejecución Programada", "api": "API", + "sdk": "SDK", + "mcp": "MCP", + "cli": "CLI", "unknown_run_type": "Tipo de Ejecución Desconocido" }, "run_status_chips": { diff --git a/public/locales/ja.json b/public/locales/ja.json index 69aa522f3..1c0486fbd 100644 --- a/public/locales/ja.json +++ b/public/locales/ja.json @@ -279,6 +279,7 @@ "errors": { "user_not_logged": "ユーザーがログインしていません。録画を保存できません。", "exists_warning": "この名前のロボットは既に存在します。ロボットの上書きを確認してください。", + "name_exists": "この名前のロボットは既に存在します。別の名前を選んでください。", "no_actions_performed": "ロボットを保存できません。保存する前に少なくとも1つのキャプチャアクションを実行してください。" }, "tooltips": { @@ -294,7 +295,8 @@ "notifications": { "terminated": "現在の録画は終了しました", "environment_reset": "ブラウザー環境がリセットされました", - "reset_successful": "すべてのキャプチャーを正常にリセットし、初期状態に戻りました" + "reset_successful": "すべてのキャプチャーを正常にリセットし、初期状態に戻りました", + "timeout_discarded": "録画セッションが10分後にタイムアウトし、自動的に破棄されました。" } }, "interpretation_log": { @@ -440,7 +442,8 @@ "warning": "⚠️ 新しいページが元のページと同じ構造であることを確認してください。" }, "fields": { - "target_url": "ロボットのターゲットURL" + "target_url": "ロボットのターゲットURL", + "new_name": "ロボット名" }, "buttons": { "duplicate": "ロボットを複製", @@ -449,6 +452,7 @@ "notifications": { "robot_not_found": "ロボットの詳細が見つかりません。もう一度お試しください。", "url_required": "ターゲットURLが必要です。", + "name_required": "ロボット名は必須です。", "duplicate_success": "ロボットが正常に複製されました。", "duplicate_error": "ターゲットURLの更新に失敗しました。もう一度お試しください。", "unknown_error": "ターゲットURLの更新中にエラーが発生しました。" @@ -457,6 +461,7 @@ "robot_settings": { "title": "ロボット設定", "target_url": "ロボットのターゲットURL", + "new_name": "ロボット名", "robot_id": "ロボットID", "robot_limit": "ロボットの制限", "created_by_user": "作成したユーザー", @@ -551,6 +556,9 @@ "manual_run": "手動実行", "scheduled_run": "スケジュール実行", "api": "API", + "sdk": "SDK", + "mcp": "MCP", + "cli": "CLI", "unknown_run_type": "不明な実行タイプ" }, "run_status_chips": { diff --git a/public/locales/tr.json b/public/locales/tr.json index b329075d1..b7209bea9 100644 --- a/public/locales/tr.json +++ b/public/locales/tr.json @@ -260,6 +260,7 @@ "confirm_all_list_fields": "Lütfen devam etmeden önce tüm liste alanlarını onaylayın." }, "tooltips": { + "capture_list_first": "Önce bir listenin üzerine gelin ve içindeki metin alanlarını seçin", "confirm_all_list_fields": "Lütfen bir sonraki adıma geçmeden önce tüm liste alanlarını onaylayın" } }, @@ -278,6 +279,7 @@ "errors": { "user_not_logged": "Kullanıcı girişi yok. Kaydedilemedi.", "exists_warning": "Bu isimde robot zaten var; üzerine yazmayı onaylayın.", + "name_exists": "Bu isimde bir robot zaten mevcut. Lütfen farklı bir isim seçin.", "no_actions_performed": "Robot kaydedilemez. Lütfen kaydetmeden önce en az bir yakalama eylemi gerçekleştirin." }, "tooltips": { @@ -293,7 +295,8 @@ "notifications": { "terminated": "Kayıt sonlandırıldı", "environment_reset": "Tarayıcı ortamı sıfırlandı", - "reset_successful": "Yakalamalar sıfırlandı ve başlangıç durumuna dönüldü" + "reset_successful": "Yakalamalar sıfırlandı ve başlangıç durumuna dönüldü", + "timeout_discarded": "Kayıt oturumu 10 dakika sonra zaman aşımına uğradı ve otomatik olarak atıldı." } }, "interpretation_log": { @@ -439,7 +442,8 @@ "warning": "⚠️ Yeni sayfanın yapısının aynı olduğundan emin olun." }, "fields": { - "target_url": "Robot Hedef URL" + "target_url": "Robot Hedef URL", + "new_name": "Robot Adı" }, "buttons": { "duplicate": "Robotu Çoğalt", @@ -448,6 +452,7 @@ "notifications": { "robot_not_found": "Robot bulunamadı. Tekrar deneyin.", "url_required": "Hedef URL gerekli.", + "name_required": "Robot adı gereklidir.", "duplicate_success": "Robot çoğaltıldı", "duplicate_error": "Hedef URL güncellenemedi. Tekrar deneyin.", "unknown_error": "Hedef URL güncellenirken hata oluştu" @@ -456,6 +461,7 @@ "robot_settings": { "title": "Robot Ayarları", "target_url": "Robot Hedef URL", + "new_name": "Robot Adı", "robot_id": "Robot ID", "robot_limit": "Robot Limiti", "created_by_user": "Oluşturan", @@ -550,6 +556,9 @@ "manual_run": "Manuel", "scheduled_run": "Zamanlanmış", "api": "API", + "sdk": "SDK", + "mcp": "MCP", + "cli": "CLI", "unknown_run_type": "Bilinmeyen" }, "run_status_chips": { diff --git a/public/locales/zh.json b/public/locales/zh.json index 20769b0a0..9eea18c64 100644 --- a/public/locales/zh.json +++ b/public/locales/zh.json @@ -279,6 +279,7 @@ "errors": { "user_not_logged": "用户未登录。无法保存录制。", "exists_warning": "已存在同名机器人,请确认是否覆盖机器人。", + "name_exists": "已存在同名机器人,请选择不同的名称。", "no_actions_performed": "无法保存机器人。请在保存之前至少执行一次捕获操作。" }, "tooltips": { @@ -294,7 +295,8 @@ "notifications": { "terminated": "当前录制已终止", "environment_reset": "浏览器环境已重置", - "reset_successful": "已成功重置所有捕获并返回初始状态" + "reset_successful": "已成功重置所有捕获并返回初始状态", + "timeout_discarded": "录制会话在10分钟后超时并自动丢弃。" } }, "interpretation_log": { @@ -440,7 +442,8 @@ "warning": "⚠️ 确保新页面与原始页面具有相同的结构。" }, "fields": { - "target_url": "机器人目标URL" + "target_url": "机器人目标URL", + "new_name": "机器人名称" }, "buttons": { "duplicate": "复制机器人", @@ -449,6 +452,7 @@ "notifications": { "robot_not_found": "找不到机器人详细信息。请重试。", "url_required": "需要目标URL。", + "name_required": "机器人名称为必填项。", "duplicate_success": "机器人复制成功。", "duplicate_error": "更新目标URL失败。请重试。", "unknown_error": "更新目标URL时发生错误。" @@ -457,6 +461,7 @@ "robot_settings": { "title": "机器人设置", "target_url": "机器人目标URL", + "new_name": "机器人名称", "robot_id": "机器人ID", "robot_limit": "机器人限制", "created_by_user": "由用户创建", @@ -551,6 +556,9 @@ "manual_run": "手动运行", "scheduled_run": "计划运行", "api": "API", + "sdk": "SDK", + "mcp": "MCP", + "cli": "CLI", "unknown_run_type": "未知运行类型" }, "run_status_chips": { diff --git a/public/svg/claude.svg b/public/svg/claude.svg new file mode 100644 index 000000000..62dc0db12 --- /dev/null +++ b/public/svg/claude.svg @@ -0,0 +1 @@ +Claude \ No newline at end of file diff --git a/public/svg/langgraph.svg b/public/svg/langgraph.svg new file mode 100644 index 000000000..fd7404004 --- /dev/null +++ b/public/svg/langgraph.svg @@ -0,0 +1 @@ +LangGraph \ No newline at end of file diff --git a/public/svg/openclaw.svg b/public/svg/openclaw.svg new file mode 100644 index 000000000..bcbc1e10c --- /dev/null +++ b/public/svg/openclaw.svg @@ -0,0 +1,22 @@ + + + + + + + + + + + + + + + + + + + + + + diff --git a/server/src/api/record.ts b/server/src/api/record.ts index fe4f7f9ca..f39eb4bca 100644 --- a/server/src/api/record.ts +++ b/server/src/api/record.ts @@ -340,6 +340,8 @@ function formatRunResponse(run: any) { runByScheduleId: run.runByScheduleId, runByAPI: run.runByAPI, runBySDK: run.runBySDK, + runByMCP: run.runByMCP, + runByCLI: run.runByCLI, data: { textData: {}, listData: {}, @@ -477,7 +479,7 @@ router.get("/robots/:id/runs/:runId", requireAPIKey, async (req: Request, res: R } }); -async function createWorkflowAndStoreMetadata(id: string, userId: string, isSDK: boolean) { +async function createWorkflowAndStoreMetadata(id: string, userId: string, runSource: 'api' | 'sdk' | 'mcp' | 'cli') { try { const recording = await Robot.findOne({ where: { @@ -522,8 +524,10 @@ async function createWorkflowAndStoreMetadata(id: string, userId: string, isSDK: log: '', runId, runByUserId: userId, - runByAPI: !isSDK, - runBySDK: isSDK, + runByAPI: runSource === 'api', + runBySDK: runSource === 'sdk', + runByMCP: runSource === 'mcp', + runByCLI: runSource === 'cli', serializableOutput: {}, binaryOutput: {}, retryCount: 0 @@ -695,7 +699,9 @@ async function executeRun(id: string, userId: string, requestedFormats?: string[ throw new Error('Could not create a new page'); } - if (recording.recording_meta.type === 'scrape') { + const robotType = (recording.recording_meta as any)?.type || (recording.recording_meta as any)?.robotType || 'extract'; + + if (robotType === 'scrape') { logger.log('info', `Executing scrape robot for API run ${id}`); let formats = recording.recording_meta.formats || ['markdown']; @@ -1192,11 +1198,11 @@ async function executeRun(id: string, userId: string, requestedFormats?: string[ } } -export async function handleRunRecording(id: string, userId: string, isSDK: boolean = false, requestedFormats?: string[]) { +export async function handleRunRecording(id: string, userId: string, runSource: 'api' | 'sdk' | 'mcp' | 'cli' = 'api', requestedFormats?: string[]) { let socket: Socket | null = null; try { - const result = await createWorkflowAndStoreMetadata(id, userId, isSDK); + const result = await createWorkflowAndStoreMetadata(id, userId, runSource); const { browserId, runId: newRunId } = result; if (!browserId || !newRunId || !userId) { @@ -1376,7 +1382,8 @@ router.post("/robots/:id/runs", requireAPIKey, async (req: AuthenticatedRequest, } const requestedFormats = req.body?.formats; - const runId = await handleRunRecording(req.params.id, req.user.id, false, requestedFormats); + const runSource = req.headers['x-run-source'] === 'mcp' ? 'mcp' : 'api'; + const runId = await handleRunRecording(req.params.id, req.user.id, runSource, requestedFormats); if (!runId) { throw new Error('Run ID is undefined'); diff --git a/server/src/api/sdk.ts b/server/src/api/sdk.ts index 7a67c3103..5c521572c 100644 --- a/server/src/api/sdk.ts +++ b/server/src/api/sdk.ts @@ -17,6 +17,12 @@ import { WorkflowEnricher } from "../sdk/workflowEnricher"; import { cancelScheduledWorkflow, scheduleWorkflow } from '../storage/schedule'; import { computeNextRun } from "../utils/schedule"; import moment from 'moment-timezone'; +import { + DEFAULT_OUTPUT_FORMATS, + parseOutputFormats, + OutputFormat, + SCRAPE_OUTPUT_FORMAT_OPTIONS, +} from '../constants/output-formats'; const router = Router(); @@ -24,6 +30,27 @@ interface AuthenticatedRequest extends Request { user?: any; } +/** + * Get the status of the authenticated user + * GET /api/sdk/status + */ +router.get("/sdk/status", requireAPIKey, async (req: AuthenticatedRequest, res: Response) => { + try { + const user = req.user; + return res.status(200).json({ + email: user.email, + plan: 'OSS', + credits: 999999 + }); + } catch (error: any) { + logger.error("Error getting status:", error); + return res.status(500).json({ + error: "Failed to get status", + message: error.message + }); + } +}); + /** * Create a new robot programmatically * POST /api/sdk/robots @@ -45,7 +72,7 @@ router.post("/sdk/robots", requireAPIKey, async (req: AuthenticatedRequest, res: }); } - const type = (workflowFile.meta as any).type || 'extract'; + const type = (workflowFile.meta as any).type || (workflowFile.meta as any).robotType || 'extract'; let enrichedWorkflow: any[] = []; let extractedUrl: string | undefined; @@ -75,6 +102,38 @@ router.post("/sdk/robots", requireAPIKey, async (req: AuthenticatedRequest, res: extractedUrl = enrichResult.url; } + const rawFormats = (workflowFile.meta as any).formats; + const { validFormats, invalidFormats, wasProvided } = parseOutputFormats( + rawFormats, + type === 'scrape' ? SCRAPE_OUTPUT_FORMAT_OPTIONS : undefined + ); + + if (invalidFormats.length > 0) { + return res.status(400).json({ + error: `Invalid formats: ${invalidFormats.map(String).join(', ')}` + }); + } + + let normalizedFormats: OutputFormat[] = validFormats; + + if (type === 'search') { + const searchAction = enrichedWorkflow + .flatMap((pair: any) => pair.what || []) + .find((action: any) => action?.action === 'search'); + const searchMode = searchAction?.args?.[0]?.mode; + + if (searchMode === 'discover') { + + normalizedFormats = validFormats.length > 0 ? validFormats : []; + } else { + + normalizedFormats = validFormats.length > 0 ? validFormats : [...DEFAULT_OUTPUT_FORMATS]; + } + } else if (type === 'crawl' || type === 'scrape') { + + normalizedFormats = validFormats.length > 0 ? validFormats : [...DEFAULT_OUTPUT_FORMATS]; + } + const robotId = uuid(); const metaId = uuid(); @@ -87,7 +146,7 @@ router.post("/sdk/robots", requireAPIKey, async (req: AuthenticatedRequest, res: params: [], type, url: extractedUrl, - formats: (workflowFile.meta as any).formats || [], + formats: normalizedFormats, isLLM: (workflowFile.meta as any).isLLM, }; @@ -430,7 +489,8 @@ router.post("/sdk/robots/:id/execute", requireAPIKey, async (req: AuthenticatedR logger.info(`[SDK] Starting execution for robot ${robotId}`); - const runId = await handleRunRecording(robotId, user.id.toString(), true); + const runSource = req.headers['x-run-source'] === 'cli' ? 'cli' : 'sdk'; + const runId = await handleRunRecording(robotId, user.id.toString(), runSource); if (!runId) { throw new Error('Failed to start robot execution'); } @@ -475,6 +535,31 @@ router.post("/sdk/robots/:id/execute", requireAPIKey, async (req: AuthenticatedR searchData = run.serializableOutput.search; } + let text: string | undefined = undefined; + if (run.serializableOutput?.text && Array.isArray(run.serializableOutput.text)) { + text = run.serializableOutput.text[0]?.content || undefined; + } + + const scrapeOutput = run.serializableOutput?.scrape as Record | undefined; + if (!text && scrapeOutput?.text && Array.isArray(scrapeOutput.text)) { + text = scrapeOutput.text[0]?.content || undefined; + } + + let markdown: string | undefined = undefined; + let html: string | undefined = undefined; + if (run.serializableOutput?.markdown && Array.isArray(run.serializableOutput.markdown)) { + markdown = run.serializableOutput.markdown[0]?.content || undefined; + } + if (!markdown && scrapeOutput?.markdown && Array.isArray(scrapeOutput.markdown)) { + markdown = scrapeOutput.markdown[0]?.content || undefined; + } + if (run.serializableOutput?.html && Array.isArray(run.serializableOutput.html)) { + html = run.serializableOutput.html[0]?.content || undefined; + } + if (!html && scrapeOutput?.html && Array.isArray(scrapeOutput.html)) { + html = scrapeOutput.html[0]?.content || undefined; + } + return res.status(200).json({ data: { runId: run.runId, @@ -483,7 +568,10 @@ router.post("/sdk/robots/:id/execute", requireAPIKey, async (req: AuthenticatedR textData: run.serializableOutput?.scrapeSchema || {}, listData: listData, crawlData: crawlData, - searchData: searchData + searchData: searchData, + text: text, + markdown: markdown, + html: html }, screenshots: Object.values(run.binaryOutput || {}) } @@ -667,6 +755,121 @@ router.post("/sdk/robots/:id/runs/:runId/abort", requireAPIKey, async (req: Auth } }); +/** + * Duplicate a robot with a new target URL + * POST /api/sdk/robots/:id/duplicate + */ +router.post("/sdk/robots/:id/duplicate", requireAPIKey, async (req: AuthenticatedRequest, res: Response) => { + try { + const robotId = req.params.id; + const { targetUrl } = req.body; + + if (!targetUrl) { + return res.status(400).json({ + error: "The \"targetUrl\" field is required." + }); + } + + try { + const parsed = new URL(targetUrl); + if (!['http:', 'https:'].includes(parsed.protocol)) { + return res.status(400).json({ + error: "The \"targetUrl\" must use http or https protocol." + }); + } + } catch { + return res.status(400).json({ + error: "The \"targetUrl\" must be a valid URL." + }); + } + + const originalRobot = await Robot.findOne({ + where: { 'recording_meta.id': robotId } + }); + + if (!originalRobot) { + return res.status(404).json({ + error: `Robot with ID "${robotId}" not found.` + }); + } + + const lastWord = targetUrl.split('/').filter(Boolean).pop() || 'Unnamed'; + + const steps: any[] = originalRobot.recording.workflow; + const entryStep = steps.findLast((step: any) => step.where?.url === 'about:blank'); + const originalEntryUrl: string | null = entryStep?.what?.find( + (action: any) => action.action === 'goto' && action.args?.length + )?.args?.[0] ?? null; + + let gotoUpdated = false; + let whereUpdateStopped = false; + + const workflow = [...steps].reverse().map((step: any) => { + let updatedWhere = step.where; + + if (originalEntryUrl && step.where?.url !== 'about:blank' && !whereUpdateStopped) { + if (step.where?.url === originalEntryUrl) { + updatedWhere = { ...step.where, url: targetUrl }; + } else { + whereUpdateStopped = true; + } + } + + const updatedWhat = step.what.map((action: any) => { + if (!gotoUpdated && action.action === 'goto' && action.args?.[0] === originalEntryUrl) { + gotoUpdated = true; + return { ...action, args: [targetUrl, ...action.args.slice(1)] }; + } + return action; + }); + + return { ...step, where: updatedWhere, what: updatedWhat }; + }).reverse(); + + const currentTimestamp = new Date().toISOString(); + + const newRobot = await Robot.create({ + id: uuid(), + userId: originalRobot.userId, + recording_meta: { + ...originalRobot.recording_meta, + id: uuid(), + name: `${originalRobot.recording_meta.name} (${lastWord})`, + url: targetUrl, + createdAt: currentTimestamp, + updatedAt: currentTimestamp, + }, + recording: { ...originalRobot.recording, workflow }, + google_sheet_email: null, + google_sheet_name: null, + google_sheet_id: null, + google_access_token: null, + google_refresh_token: null, + airtable_base_id: null, + airtable_base_name: null, + airtable_table_name: null, + airtable_table_id: null, + airtable_access_token: null, + airtable_refresh_token: null, + webhooks: null, + schedule: null, + }); + + logger.info(`[SDK] Robot ${robotId} duplicated as ${newRobot.recording_meta.id}`); + + return res.status(201).json({ + data: newRobot, + message: "Robot duplicated successfully" + }); + } catch (error: any) { + logger.error("[SDK] Error duplicating robot:", error); + return res.status(500).json({ + error: "Failed to duplicate robot", + message: error.message + }); + } +}); + /** * Create a crawl robot programmatically * POST /api/sdk/crawl @@ -674,7 +877,7 @@ router.post("/sdk/robots/:id/runs/:runId/abort", requireAPIKey, async (req: Auth router.post("/sdk/crawl", requireAPIKey, async (req: AuthenticatedRequest, res: Response) => { try { const user = req.user; - const { url, name, crawlConfig } = req.body; + const { url, name, crawlConfig, formats } = req.body; if (!url || !crawlConfig) { return res.status(400).json({ @@ -696,6 +899,18 @@ router.post("/sdk/crawl", requireAPIKey, async (req: AuthenticatedRequest, res: }); } + const { validFormats: requestedFormats, invalidFormats, wasProvided } = parseOutputFormats(formats); + if (invalidFormats.length > 0) { + return res.status(400).json({ + error: `Invalid formats: ${invalidFormats.map(String).join(', ')}` + }); + } + + // Crawl always needs formats; use defaults even if explicit empty array is provided + const crawlFormats: OutputFormat[] = requestedFormats.length > 0 + ? requestedFormats + : [...DEFAULT_OUTPUT_FORMATS]; + const robotName = name || `Crawl Robot - ${new URL(url).hostname}`; const robotId = uuid(); const metaId = uuid(); @@ -712,13 +927,13 @@ router.post("/sdk/crawl", requireAPIKey, async (req: AuthenticatedRequest, res: params: [], type: 'crawl', url: url, + formats: crawlFormats, }, recording: { workflow: [ { where: { url }, what: [ - { action: 'flag', args: ['generated'] }, { action: 'crawl', args: [crawlConfig], @@ -778,7 +993,7 @@ router.post("/sdk/crawl", requireAPIKey, async (req: AuthenticatedRequest, res: router.post("/sdk/search", requireAPIKey, async (req: AuthenticatedRequest, res: Response) => { try { const user = req.user; - const { name, searchConfig } = req.body; + const { name, searchConfig, formats } = req.body; if (!searchConfig) { return res.status(400).json({ @@ -804,8 +1019,23 @@ router.post("/sdk/search", requireAPIKey, async (req: AuthenticatedRequest, res: }); } + const { validFormats: requestedFormats, invalidFormats, wasProvided } = parseOutputFormats(formats); + if (invalidFormats.length > 0) { + return res.status(400).json({ + error: `Invalid formats: ${invalidFormats.map(String).join(', ')}` + }); + } + + const searchFormats: OutputFormat[] = searchConfig.mode === 'discover' + ? (requestedFormats.length > 0 ? requestedFormats : []) + : (requestedFormats.length > 0 ? requestedFormats : [...DEFAULT_OUTPUT_FORMATS]); + searchConfig.provider = 'duckduckgo'; + if (searchConfig.outputFormats && Array.isArray(searchConfig.outputFormats) && searchConfig.outputFormats.length > 0) { + searchConfig.mode = 'scrape'; + } + const robotName = name || `Search Robot - ${searchConfig.query}`; const robotId = uuid(); const metaId = uuid(); @@ -821,6 +1051,7 @@ router.post("/sdk/search", requireAPIKey, async (req: AuthenticatedRequest, res: pairs: 1, params: [], type: 'search', + formats: searchFormats, }, recording: { workflow: [ diff --git a/server/src/browser-management/controller.ts b/server/src/browser-management/controller.ts index 47814c555..79379ee67 100644 --- a/server/src/browser-management/controller.ts +++ b/server/src/browser-management/controller.ts @@ -11,6 +11,9 @@ import { RemoteBrowser } from "./classes/RemoteBrowser"; import { RemoteBrowserOptions } from "../types"; import logger from "../logger"; +const RECORDING_TIMEOUT_MS = 10 * 60 * 1000; // 10 minutes +const recordingTimeouts = new Map(); + /** * Starts and initializes a {@link RemoteBrowser} instance. * Creates a new socket connection over a dedicated namespace @@ -41,7 +44,31 @@ export const initializeRemoteBrowserForRecording = (userId: string, mode: string logger.info('DOM streaming started for remote browser in recording mode'); - browserPool.addRemoteBrowser(id, browserSession, userId, false, "recording"); + const added = browserPool.addRemoteBrowser(id, browserSession, userId, false, "recording"); + if (!added) { + logger.error(`Failed to add recording browser ${id} to pool; cleaning up session`); + socket.emit('dom-mode-error', { + userId, + error: 'Failed to start the browser, please try again in some time.' + }); + await browserSession.switchOff(); + return id; + } + + const timeoutHandle = setTimeout(async () => { + recordingTimeouts.delete(id); + logger.warn(`Recording session ${id} timed out, auto-discarding`); + try { + io.of(id).emit('recording-timeout'); + } catch (e) { + logger.warn(`Failed to emit recording-timeout event for session ${id}: ${e}`); + } + // Wait for the frontend to receive the event, post BroadcastChannel message, + // and close the recording tab before we tear down the socket connection. + await new Promise(resolve => setTimeout(resolve, 1000)); + await destroyRemoteBrowser(id, userId); + }, RECORDING_TIMEOUT_MS); + recordingTimeouts.set(id, timeoutHandle); } catch (initError: any) { logger.error(`Failed to initialize browser for recording: ${initError.message}`); logger.info('Sending browser failure notification to frontend'); @@ -116,7 +143,18 @@ export const createRemoteBrowserForRun = (userId: string): string => { * @returns {Promise} * @category BrowserManagement-Controller */ +export const clearRecordingTimeout = (id: string): void => { + const existingTimeout = recordingTimeouts.get(id); + if (existingTimeout) { + clearTimeout(existingTimeout); + recordingTimeouts.delete(id); + logger.log('debug', `Recording timeout cancelled for session ${id}`); + } +}; + export const destroyRemoteBrowser = async (id: string, userId: string): Promise => { + clearRecordingTimeout(id); + const DESTROY_TIMEOUT = 30000; const destroyPromise = (async () => { diff --git a/server/src/constants/output-formats.ts b/server/src/constants/output-formats.ts new file mode 100644 index 000000000..b414959f7 --- /dev/null +++ b/server/src/constants/output-formats.ts @@ -0,0 +1,57 @@ +export const OUTPUT_FORMAT_OPTIONS = [ + 'markdown', + 'html', + 'text', + 'screenshot-visible', + 'screenshot-fullpage', +] as const; + +export type OutputFormat = (typeof OUTPUT_FORMAT_OPTIONS)[number]; + +export const SCRAPE_OUTPUT_FORMAT_OPTIONS: OutputFormat[] = [ + 'markdown', + 'html', + 'screenshot-visible', + 'screenshot-fullpage', +]; + +export const SEARCH_SCRAPE_OUTPUT_FORMAT_OPTIONS: OutputFormat[] = [ + 'markdown', + 'html', + 'text', + 'screenshot-visible', + 'screenshot-fullpage', +]; + +const OUTPUT_FORMAT_SET = new Set(OUTPUT_FORMAT_OPTIONS as readonly string[]); + +export const DEFAULT_OUTPUT_FORMATS: OutputFormat[] = ['markdown']; + +export function isOutputFormat(value: unknown): value is OutputFormat { + return typeof value === 'string' && OUTPUT_FORMAT_SET.has(value); +} + +export function parseOutputFormats( + formats: unknown, + allowedFormats: readonly OutputFormat[] = OUTPUT_FORMAT_OPTIONS +): { + validFormats: OutputFormat[]; + invalidFormats: unknown[]; + wasProvided: boolean; +} { + const wasProvided = formats !== undefined; + const requestedFormats = Array.isArray(formats) ? formats : []; + const validFormats: OutputFormat[] = []; + const invalidFormats: unknown[] = []; + const allowedSet = new Set(allowedFormats as readonly string[]); + + requestedFormats.forEach((format) => { + if (isOutputFormat(format) && allowedSet.has(format)) { + validFormats.push(format); + } else { + invalidFormats.push(format); + } + }); + + return { validFormats, invalidFormats, wasProvided }; +} diff --git a/server/src/db/migrations/20250612000000-add-robot-name-unique-index.js b/server/src/db/migrations/20250612000000-add-robot-name-unique-index.js new file mode 100644 index 000000000..344ddf6a1 --- /dev/null +++ b/server/src/db/migrations/20250612000000-add-robot-name-unique-index.js @@ -0,0 +1,19 @@ +'use strict'; + +module.exports = { + async up(queryInterface) { + await queryInterface.sequelize.query(` + CREATE UNIQUE INDEX IF NOT EXISTS robot_user_name_unique + ON robot ( + "userId", + lower(trim(recording_meta->>'name')) + ); + `); + }, + + async down(queryInterface) { + await queryInterface.sequelize.query(` + DROP INDEX IF EXISTS robot_user_name_unique; + `); + } +}; diff --git a/server/src/markdownify/scrape.ts b/server/src/markdownify/scrape.ts index 70c06084e..075df7ae2 100644 --- a/server/src/markdownify/scrape.ts +++ b/server/src/markdownify/scrape.ts @@ -135,7 +135,10 @@ export async function convertPageToScreenshot(url: string, page: Page, fullPage: const screenshotType = fullPage ? 'full page' : 'visible viewport'; logger.log('info', `[Scrape] Taking ${screenshotType} screenshot of ${url}`); - await gotoWithFallback(page, url); + const currentUrl = page.url(); + if (currentUrl !== url) { + await gotoWithFallback(page, url); + } const screenshot = await page.screenshot({ type: 'png', diff --git a/server/src/mcp-worker.ts b/server/src/mcp-worker.ts index 259ef2fda..34462ce2c 100644 --- a/server/src/mcp-worker.ts +++ b/server/src/mcp-worker.ts @@ -38,6 +38,7 @@ class MaxunMCPWorker { const headers = { 'Content-Type': 'application/json', 'x-api-key': this.apiKey, + 'x-run-source': 'mcp', ...options.headers }; diff --git a/server/src/middlewares/api.ts b/server/src/middlewares/api.ts index 5374af9bd..6e5cf1edc 100644 --- a/server/src/middlewares/api.ts +++ b/server/src/middlewares/api.ts @@ -3,14 +3,21 @@ import User from "../models/User"; import { AuthenticatedRequest } from "../routes/record" export const requireAPIKey = async (req: AuthenticatedRequest, res: Response, next: any) => { - const apiKey = req.headers['x-api-key']; - if (!apiKey) { - return res.status(401).json({ error: "API key is missing" }); - } - const user = await User.findOne({ where: { api_key: apiKey } }); - if (!user) { - return res.status(403).json({ error: "Invalid API key" }); + try { + const apiKey = req.headers['x-api-key']; + if (!apiKey) { + return res.status(401).json({ error: "API key is missing" }); + } + + const user = await User.findOne({ where: { api_key: apiKey } }); + if (!user) { + return res.status(403).json({ error: "Invalid API key" }); + } + + req.user = user; + next(); + } catch (error) { + console.error("API key authentication failed:", error); + return res.status(503).json({ error: "Authentication service temporarily unavailable" }); } - req.user = user; - next(); }; diff --git a/server/src/models/Robot.ts b/server/src/models/Robot.ts index d45ac502d..637bcb9cb 100644 --- a/server/src/models/Robot.ts +++ b/server/src/models/Robot.ts @@ -1,6 +1,7 @@ import { Model, DataTypes, Optional } from 'sequelize'; import sequelize from '../storage/db'; import { WhereWhatPair } from 'maxun-core'; +import { OutputFormat } from '../constants/output-formats'; interface RobotMeta { name: string; @@ -11,7 +12,7 @@ interface RobotMeta { params: any[]; type?: 'extract' | 'scrape' | 'crawl' | 'search'; url?: string; - formats?: ('markdown' | 'html' | 'screenshot-visible' | 'screenshot-fullpage')[]; + formats?: OutputFormat[]; isLLM?: boolean; } diff --git a/server/src/models/Run.ts b/server/src/models/Run.ts index 5c45f2a8c..9437cc6df 100644 --- a/server/src/models/Run.ts +++ b/server/src/models/Run.ts @@ -24,6 +24,8 @@ interface RunAttributes { runByScheduleId?: string; runByAPI?: boolean; runBySDK?: boolean; + runByMCP?: boolean; + runByCLI?: boolean; serializableOutput: Record; binaryOutput: Record; retryCount?: number; @@ -47,6 +49,8 @@ class Run extends Model implements RunAttr public runByScheduleId!: string; public runByAPI!: boolean; public runBySDK!: boolean; + public runByMCP!: boolean; + public runByCLI!: boolean; public serializableOutput!: Record; public binaryOutput!: Record; public retryCount!: number; @@ -119,6 +123,14 @@ Run.init( type: DataTypes.BOOLEAN, allowNull: true, }, + runByMCP: { + type: DataTypes.BOOLEAN, + allowNull: true, + }, + runByCLI: { + type: DataTypes.BOOLEAN, + allowNull: true, + }, serializableOutput: { type: DataTypes.JSONB, allowNull: true, diff --git a/server/src/pgboss-worker.ts b/server/src/pgboss-worker.ts index 1dc63307c..c350f7c6a 100644 --- a/server/src/pgboss-worker.ts +++ b/server/src/pgboss-worker.ts @@ -21,6 +21,8 @@ import { io as serverIo } from "./server"; import { sendWebhook } from './routes/webhook'; import { BinaryOutputService } from './storage/mino'; import { convertPageToMarkdown, convertPageToHTML, convertPageToScreenshot } from './markdownify/scrape'; +import { processRobotOutputFormats } from './utils/output-post-processor'; +import { getInterpretationFailureReason, hasExpectedRobotOutput } from './utils/output-validation'; if (!process.env.DB_USER || !process.env.DB_PASSWORD || !process.env.DB_HOST || !process.env.DB_PORT || !process.env.DB_NAME) { throw new Error('Failed to start pgboss worker: one or more required environment variables are missing.'); @@ -125,7 +127,7 @@ async function triggerIntegrationUpdates(runId: string, robotMetaId: string): Pr * Modified processRunExecution function - only add browser reset */ async function processRunExecution(job: Job) { - const BROWSER_INIT_TIMEOUT = 30000; + const BROWSER_INIT_TIMEOUT = 60000; const BROWSER_PAGE_TIMEOUT = 15000; const data = job.data; @@ -161,7 +163,7 @@ async function processRunExecution(job: Job) { const browserWaitStart = Date.now(); let lastLogTime = 0; let pollAttempts = 0; - const MAX_POLL_ATTEMPTS = 15; + const MAX_POLL_ATTEMPTS = Math.ceil(BROWSER_INIT_TIMEOUT / 2000); while (!browser && (Date.now() - browserWaitStart) < BROWSER_INIT_TIMEOUT && pollAttempts < MAX_POLL_ATTEMPTS) { const currentTime = Date.now(); @@ -467,12 +469,6 @@ async function processRunExecution(job: Job) { logger.log('info', `Workflow execution completed for run ${data.runId}`); - const binaryOutputService = new BinaryOutputService('maxun-run-screenshots'); - const uploadedBinaryOutput = await binaryOutputService.uploadAndStoreBinaryOutput( - run, - interpretationInfo.binaryOutput - ); - const finalRun = await Run.findByPk(run.id); const categorizedOutput = { scrapeSchema: finalRun?.serializableOutput?.scrapeSchema || {}, @@ -480,6 +476,46 @@ async function processRunExecution(job: Job) { crawl: finalRun?.serializableOutput?.crawl || {}, search: finalRun?.serializableOutput?.search || {} }; + + let binaryOutput: Record = { + ...(interpretationInfo.binaryOutput || {}) + }; + + // Post-process crawl/search output according to selected output formats + const robotType = recording.recording_meta.type; + const outputFormats = (recording.recording_meta as any).formats as string[] | undefined; + if (robotType === 'crawl' || robotType === 'search') { + const processedOutput = await processRobotOutputFormats({ + robotType, + outputFormats, + categorizedOutput: { + crawl: categorizedOutput.crawl as Record, + search: categorizedOutput.search as Record, + }, + currentPage, + initialBinaryOutput: binaryOutput, + }); + + categorizedOutput.crawl = processedOutput.categorizedOutput.crawl; + categorizedOutput.search = processedOutput.categorizedOutput.search; + binaryOutput = processedOutput.binaryOutput; + + const hasOutput = hasExpectedRobotOutput(robotType, { + crawl: categorizedOutput.crawl as Record, + search: categorizedOutput.search as Record + }, outputFormats, binaryOutput); + + if (!hasOutput) { + const humanRobotType = robotType.charAt(0).toUpperCase() + robotType.slice(1); + const fallbackReason = `${humanRobotType} run completed without producing output data`; + throw new Error(getInterpretationFailureReason(interpretationInfo.log, fallbackReason)); + } + } + + const binaryOutputService = new BinaryOutputService('maxun-run-screenshots'); + const uploadedBinaryOutput = Object.keys(binaryOutput).length > 0 + ? await binaryOutputService.uploadAndStoreBinaryOutput(run, binaryOutput) + : {}; if (await isRunAborted()) { logger.log('info', `Run ${data.runId} was aborted while processing results, not updating status`); @@ -491,11 +527,16 @@ async function processRunExecution(job: Job) { finishedAt: new Date().toLocaleString(), log: interpretationInfo.log.join('\n'), binaryOutput: uploadedBinaryOutput, + serializableOutput: { + ...(finalRun?.serializableOutput || {}), + crawl: categorizedOutput.crawl, + search: categorizedOutput.search, + } }); let totalSchemaItemsExtracted = 0; let totalListItemsExtracted = 0; - let extractedScreenshotsCount = 0; + let extractedScreenshotsCount = Object.keys(uploadedBinaryOutput).length; if (categorizedOutput) { if (categorizedOutput.scrapeSchema) { @@ -516,9 +557,6 @@ async function processRunExecution(job: Job) { }); } - if (run.binaryOutput) { - extractedScreenshotsCount = Object.keys(run.binaryOutput).length; - } } const totalRowsExtracted = totalSchemaItemsExtracted + totalListItemsExtracted; @@ -609,7 +647,9 @@ async function processRunExecution(job: Job) { try { const hasData = (run.serializableOutput && ((run.serializableOutput.scrapeSchema && run.serializableOutput.scrapeSchema.length > 0) || - (run.serializableOutput.scrapeList && run.serializableOutput.scrapeList.length > 0))) || + (run.serializableOutput.scrapeList && run.serializableOutput.scrapeList.length > 0) || + (run.serializableOutput.crawl && Object.keys(run.serializableOutput.crawl).length > 0) || + (run.serializableOutput.search && Object.keys(run.serializableOutput.search).length > 0))) || (run.binaryOutput && Object.keys(run.binaryOutput).length > 0); if (hasData) { diff --git a/server/src/routes/record.ts b/server/src/routes/record.ts index 8992b1e34..ad7684fee 100644 --- a/server/src/routes/record.ts +++ b/server/src/routes/record.ts @@ -12,6 +12,7 @@ import { getActiveBrowserIdByState, destroyRemoteBrowser, canCreateBrowserInState, + clearRecordingTimeout, } from '../browser-management/controller'; import logger from "../logger"; import { requireSignIn } from '../middlewares/auth'; @@ -135,6 +136,10 @@ router.get('/stop/:browserId', requireSignIn, async (req: AuthenticatedRequest, return res.status(401).send('User not authenticated'); } + // Cancel the recording timeout synchronously before any async work + // to prevent the timer firing during the pgBoss job queue window. + clearRecordingTimeout(req.params.browserId); + try { await pgBossClient.createQueue('destroy-browser'); diff --git a/server/src/routes/storage.ts b/server/src/routes/storage.ts index 6b57bc6b9..ea710bb8b 100644 --- a/server/src/routes/storage.ts +++ b/server/src/routes/storage.ts @@ -16,9 +16,36 @@ import { WorkflowFile } from 'maxun-core'; import { cancelScheduledWorkflow, scheduleWorkflow } from '../storage/schedule'; import { pgBossClient } from '../storage/pgboss'; import { WorkflowEnricher } from '../sdk/workflowEnricher'; +import sequelizeInstance from '../storage/db'; +import { Op } from 'sequelize'; +import { + DEFAULT_OUTPUT_FORMATS, + parseOutputFormats, + SEARCH_SCRAPE_OUTPUT_FORMAT_OPTIONS, + SCRAPE_OUTPUT_FORMAT_OPTIONS, + OutputFormat, +} from '../constants/output-formats'; export const router = Router(); +async function isRobotNameTaken(name: string, userId: number, excludeId?: string): Promise { + const normalised = name.trim().toLowerCase(); + const robots = await Robot.findAll({ + where: { + userId, + [Op.and]: sequelizeInstance.where( + sequelizeInstance.fn('lower', sequelizeInstance.fn('trim', sequelizeInstance.literal("recording_meta->>'name'"))), + normalised + ), + } as any, + }); + if (robots.length === 0) return false; + if (excludeId) { + return robots.some((r: any) => r.recording_meta.id !== excludeId); + } + return true; +} + export const processWorkflowActions = async (workflow: any[], checkLimit: boolean = false): Promise => { const processedWorkflow = JSON.parse(JSON.stringify(workflow)); @@ -251,10 +278,10 @@ function handleWorkflowActions(workflow: any[], credentials: Credentials) { router.put('/recordings/:id', requireSignIn, async (req: AuthenticatedRequest, res) => { try { const { id } = req.params; - const { name, limits, credentials, targetUrl, workflow: incomingWorkflow } = req.body; + const { name, limits, credentials, targetUrl, workflow: incomingWorkflow, formats } = req.body; - if (!name && !limits && !credentials && !targetUrl && !incomingWorkflow) { - return res.status(400).json({ error: 'Either "name", "limits", "credentials" or "target_url" must be provided.' }); + if (!name && !limits && !credentials && !targetUrl && !incomingWorkflow && formats === undefined) { + return res.status(400).json({ error: 'Either "name", "limits", "credentials", "target_url", "workflow" or "formats" must be provided.' }); } const robot = await Robot.findOne({ where: { 'recording_meta.id': id } }); @@ -338,9 +365,68 @@ router.put('/recordings/:id', requireSignIn, async (req: AuthenticatedRequest, r } } + let normalizedFormats: OutputFormat[] | undefined; + + let searchMode: string | undefined; + if (robot.recording_meta?.type === 'search') { + const searchAction = workflow + .flatMap((pair: any) => pair.what || []) + .find((action: any) => action?.action === 'search'); + searchMode = searchAction?.args?.[0]?.mode; + } + + if (formats !== undefined || (robot.recording_meta?.type === 'search' && searchMode === 'discover')) { + let allowedFormats: readonly OutputFormat[] | undefined; + if (robot.recording_meta?.type === 'scrape') { + allowedFormats = SCRAPE_OUTPUT_FORMAT_OPTIONS; + } else if (robot.recording_meta?.type === 'search' && searchMode === 'scrape') { + allowedFormats = SEARCH_SCRAPE_OUTPUT_FORMAT_OPTIONS; + } + + const { validFormats, invalidFormats } = parseOutputFormats(formats, allowedFormats); + + if (invalidFormats.length > 0) { + return res.status(400).json({ + error: `Invalid formats: ${invalidFormats.map(String).join(', ')}`, + }); + } + + if (robot.recording_meta?.type === 'crawl') { + normalizedFormats = validFormats.length > 0 ? validFormats : [...DEFAULT_OUTPUT_FORMATS]; + } else if (robot.recording_meta?.type === 'scrape') { + normalizedFormats = validFormats.length > 0 ? validFormats : [...DEFAULT_OUTPUT_FORMATS]; + } else if (robot.recording_meta?.type === 'search') { + if (searchMode === 'discover') { + normalizedFormats = []; + } else { + normalizedFormats = validFormats.length > 0 ? validFormats : [...DEFAULT_OUTPUT_FORMATS]; + } + } else { + normalizedFormats = validFormats; + } + } + + let trimmedName: string | undefined; + if (name !== undefined) { + if (typeof name !== 'string') { + return res.status(400).json({ error: 'Robot name must be a string.' }); + } + trimmedName = name.trim(); + if (!trimmedName) { + return res.status(400).json({ error: 'Robot name cannot be empty.' }); + } + if (trimmedName.toLowerCase() !== robot.recording_meta.name.trim().toLowerCase()) { + const nameTaken = await isRobotNameTaken(trimmedName, robot.userId as number, id); + if (nameTaken) { + return res.status(409).json({ error: `A robot with the name "${trimmedName}" already exists.` }); + } + } + } + let updatedMeta = { ...robot.recording_meta }; - if (name) updatedMeta.name = name; + if (trimmedName) updatedMeta.name = trimmedName; if (targetUrl) updatedMeta.url = targetUrl; + if (normalizedFormats !== undefined) updatedMeta.formats = normalizedFormats; const updates: any = { recording: { ...robot.recording, workflow }, @@ -354,7 +440,10 @@ router.put('/recordings/:id', requireSignIn, async (req: AuthenticatedRequest, r logger.log('info', `Robot with ID ${id} was updated successfully.`); return res.status(200).json({ message: 'Robot updated successfully', robot }); - } catch (error) { + } catch (error: any) { + if (error.name === 'SequelizeUniqueConstraintError' || error.parent?.code === '23505') { + return res.status(409).json({ error: 'A robot with this name already exists.' }); + } // Safely handle the error type if (error instanceof Error) { logger.log('error', `Error updating robot with ID ${req.params.id}: ${error.message}`); @@ -388,19 +477,32 @@ router.post('/recordings/scrape', requireSignIn, async (req: AuthenticatedReques return res.status(400).json({ error: 'Invalid URL format' }); } - // Validate format - const validFormats = ['markdown', 'html', 'screenshot-visible', 'screenshot-fullpage']; + const { validFormats: scrapeFormats, invalidFormats } = parseOutputFormats( + formats, + SCRAPE_OUTPUT_FORMAT_OPTIONS + ); - if (!Array.isArray(formats) || formats.length === 0) { - return res.status(400).json({ error: 'At least one output format must be selected.' }); + if (invalidFormats.length > 0) { + return res.status(400).json({ + error: `Invalid formats: ${invalidFormats.map(String).join(', ')}`, + }); } - const invalid = formats.filter(f => !validFormats.includes(f)); - if (invalid.length > 0) { - return res.status(400).json({ error: `Invalid formats: ${invalid.join(', ')}` }); + const finalFormats = scrapeFormats.length > 0 ? scrapeFormats : DEFAULT_OUTPUT_FORMATS; + + const robotName = (typeof name === 'string' ? name.trim() : '') || `Markdown Robot - ${new URL(url).hostname}`; + if (!robotName) { + return res.status(400).json({ error: 'Robot name cannot be empty.' }); + } + + if (await isRobotNameTaken(robotName, req.user.id)) { + return res.status(409).json({ error: `A robot with the name "${robotName}" already exists.` }); + } + + if (scrapeFormats.length === 0 && formats !== undefined) { + return res.status(400).json({ error: 'At least one output format must be selected.' }); } - const robotName = name || `Markdown Robot - ${new URL(url).hostname}`; const currentTimestamp = new Date().toLocaleString(); const robotId = uuid(); @@ -416,7 +518,7 @@ router.post('/recordings/scrape', requireSignIn, async (req: AuthenticatedReques params: [], type: 'scrape', url: url, - formats: formats, + formats: finalFormats, }, recording: { workflow: [] }, google_sheet_email: null, @@ -440,7 +542,10 @@ router.post('/recordings/scrape', requireSignIn, async (req: AuthenticatedReques message: 'Markdown robot created successfully.', robot: newRobot, }); - } catch (error) { + } catch (error: any) { + if (error.name === 'SequelizeUniqueConstraintError' || error.parent?.code === '23505') { + return res.status(409).json({ error: 'A robot with this name already exists.' }); + } if (error instanceof Error) { logger.log('error', `Error creating markdown robot: ${error.message}`); return res.status(500).json({ error: error.message }); @@ -476,6 +581,15 @@ router.post('/recordings/llm', requireSignIn, async (req: AuthenticatedRequest, } } + const finalRobotName = (typeof robotName === 'string' ? robotName.trim() : '') || `LLM Extract: ${prompt.substring(0, 50)}`; + if (!finalRobotName) { + return res.status(400).json({ error: 'Robot name cannot be empty.' }); + } + + if (await isRobotNameTaken(finalRobotName, req.user.id)) { + return res.status(409).json({ error: `A robot with the name "${finalRobotName}" already exists.` }); + } + let workflowResult: any; let finalUrl: string; @@ -509,7 +623,6 @@ router.post('/recordings/llm', requireSignIn, async (req: AuthenticatedRequest, const robotId = uuid(); const currentTimestamp = new Date().toISOString(); - const finalRobotName = robotName || `LLM Extract: ${prompt.substring(0, 50)}`; const newRobot = await Robot.create({ id: uuid(), @@ -547,7 +660,10 @@ router.post('/recordings/llm', requireSignIn, async (req: AuthenticatedRequest, message: 'LLM robot created successfully.', robot: newRobot, }); - } catch (error) { + } catch (error: any) { + if (error.name === 'SequelizeUniqueConstraintError' || error.parent?.code === '23505') { + return res.status(409).json({ error: 'A robot with this name already exists.' }); + } if (error instanceof Error) { logger.log('error', `Error creating LLM robot: ${error.message}`); return res.status(500).json({ error: error.message }); @@ -591,7 +707,7 @@ router.delete('/recordings/:id', requireSignIn, async (req: AuthenticatedRequest router.post('/recordings/:id/duplicate', requireSignIn, async (req: AuthenticatedRequest, res) => { try { const { id } = req.params; - const { targetUrl } = req.body; + const { targetUrl, newName } = req.body; if (!targetUrl) { return res.status(400).json({ error: 'The "targetUrl" field is required.' }); @@ -615,6 +731,11 @@ router.post('/recordings/:id/duplicate', requireSignIn, async (req: Authenticate } const lastWord = targetUrl.split('/').filter(Boolean).pop() || 'Unnamed'; + const duplicateName = (newName?.trim() || `${originalRobot.recording_meta.name} (${lastWord})`).trim(); + + if (await isRobotNameTaken(duplicateName, originalRobot.userId as number)) { + return res.status(409).json({ error: `A robot with the name "${duplicateName}" already exists.` }); + } const steps: any[] = originalRobot.recording.workflow; const entryStep = steps.findLast((step: any) => step.where?.url === 'about:blank'); @@ -655,7 +776,7 @@ router.post('/recordings/:id/duplicate', requireSignIn, async (req: Authenticate recording_meta: { ...originalRobot.recording_meta, id: uuid(), - name: `${originalRobot.recording_meta.name} (${lastWord})`, + name: duplicateName, url: targetUrl, createdAt: currentTimestamp, updatedAt: currentTimestamp, @@ -1390,7 +1511,7 @@ export async function recoverOrphanedRuns() { */ router.post('/recordings/crawl', requireSignIn, async (req: AuthenticatedRequest, res) => { try { - const { url, name, crawlConfig } = req.body; + const { url, name, crawlConfig, formats } = req.body; if (!url || !crawlConfig) { return res.status(400).json({ error: 'URL and crawl configuration are required.' }); @@ -1406,7 +1527,27 @@ router.post('/recordings/crawl', requireSignIn, async (req: AuthenticatedRequest return res.status(400).json({ error: 'Invalid URL format' }); } - const robotName = name || `Crawl Robot - ${new URL(url).hostname}`; + const robotName = (typeof name === 'string' ? name.trim() : '') || `Crawl Robot - ${new URL(url).hostname}`; + if (!robotName) { + return res.status(400).json({ error: 'Robot name cannot be empty.' }); + } + + if (await isRobotNameTaken(robotName, req.user.id)) { + return res.status(409).json({ error: `A robot with the name "${robotName}" already exists.` }); + } + + const { validFormats: requestedFormats, invalidFormats } = parseOutputFormats(formats); + if (invalidFormats.length > 0) { + return res.status(400).json({ + error: `Invalid formats: ${invalidFormats.map(String).join(', ')}`, + }); + } + + // Crawl always needs formats; use defaults even if explicit empty array is provided + const crawlFormats: OutputFormat[] = requestedFormats.length > 0 + ? requestedFormats + : [...DEFAULT_OUTPUT_FORMATS]; + const currentTimestamp = new Date().toLocaleString('en-US'); const robotId = uuid(); @@ -1422,6 +1563,7 @@ router.post('/recordings/crawl', requireSignIn, async (req: AuthenticatedRequest params: [], type: 'crawl', url: url, + formats: crawlFormats, }, recording: { workflow: [ @@ -1482,7 +1624,10 @@ router.post('/recordings/crawl', requireSignIn, async (req: AuthenticatedRequest message: 'Crawl robot created successfully.', robot: newRobot, }); - } catch (error) { + } catch (error: any) { + if (error.name === 'SequelizeUniqueConstraintError' || error.parent?.code === '23505') { + return res.status(409).json({ error: 'A robot with this name already exists.' }); + } if (error instanceof Error) { logger.log('error', `Error creating crawl robot: ${error.message}`); return res.status(500).json({ error: error.message }); @@ -1500,7 +1645,7 @@ router.post('/recordings/crawl', requireSignIn, async (req: AuthenticatedRequest */ router.post('/recordings/search', requireSignIn, async (req: AuthenticatedRequest, res) => { try { - const { searchConfig, name } = req.body; + const { searchConfig, name, formats } = req.body; if (!searchConfig || !searchConfig.query) { return res.status(400).json({ error: 'Search configuration with query is required.' }); @@ -1510,7 +1655,34 @@ router.post('/recordings/search', requireSignIn, async (req: AuthenticatedReques return res.status(401).send({ error: 'Unauthorized' }); } - const robotName = name || `Search Robot - ${searchConfig.query.substring(0, 50)}`; + const robotName = (typeof name === 'string' ? name.trim() : '') || `Search Robot - ${searchConfig.query.substring(0, 50)}`; + if (!robotName) { + return res.status(400).json({ error: 'Robot name cannot be empty.' }); + } + + if (await isRobotNameTaken(robotName, req.user.id)) { + return res.status(409).json({ error: `A robot with the name "${robotName}" already exists.` }); + } + + const { validFormats: requestedFormats, invalidFormats } = parseOutputFormats( + formats, + searchConfig.mode === 'scrape' ? SEARCH_SCRAPE_OUTPUT_FORMAT_OPTIONS : undefined + ); + if (invalidFormats.length > 0) { + return res.status(400).json({ + error: `Invalid formats: ${invalidFormats.map(String).join(', ')}`, + }); + } + + let searchFormats: OutputFormat[]; + if (searchConfig.mode === 'discover') { + // Discover-mode: always empty, ignore caller input + searchFormats = []; + } else { + // Scrape-mode: apply defaults if empty + searchFormats = requestedFormats.length > 0 ? requestedFormats : [...DEFAULT_OUTPUT_FORMATS]; + } + const currentTimestamp = new Date().toLocaleString('en-US'); const robotId = uuid(); @@ -1525,6 +1697,7 @@ router.post('/recordings/search', requireSignIn, async (req: AuthenticatedReques pairs: 1, params: [], type: 'search', + formats: searchFormats, }, recording: { workflow: [ @@ -1570,7 +1743,10 @@ router.post('/recordings/search', requireSignIn, async (req: AuthenticatedReques message: 'Search robot created successfully.', robot: newRobot, }); - } catch (error) { + } catch (error: any) { + if (error.name === 'SequelizeUniqueConstraintError' || error.parent?.code === '23505') { + return res.status(409).json({ error: 'A robot with this name already exists.' }); + } if (error instanceof Error) { logger.log('error', `Error creating search robot: ${error.message}`); return res.status(500).json({ error: error.message }); diff --git a/server/src/server.ts b/server/src/server.ts index 316bf3dec..d5b5e5b70 100644 --- a/server/src/server.ts +++ b/server/src/server.ts @@ -24,11 +24,25 @@ import { startWorkers } from './pgboss-worker'; import { stopPgBossClient, startPgBossClient } from './storage/pgboss' import Run from './models/Run'; -const app = express(); -app.use(cors({ - origin: process.env.PUBLIC_URL ? process.env.PUBLIC_URL : 'http://localhost:5173', +const normalizeOrigin = (urlString?: string): string => { + if (!urlString) return 'http://localhost:5173'; + try { + const url = new URL(urlString); + return `${url.protocol}//${url.host}`; + } catch { + return 'http://localhost:5173'; + } +}; + +const CORS_CONFIG = { + origin: normalizeOrigin(process.env.PUBLIC_URL), credentials: true, -})); + methods: ['GET', 'POST', 'PUT', 'DELETE', 'OPTIONS'], + allowedHeaders: ['Content-Type', 'Authorization'], +}; + +const app = express(); +app.use(cors(CORS_CONFIG)); app.use(express.json()); const { Pool } = pg; @@ -90,7 +104,7 @@ export let io = new Server(server, { pingInterval: 25000, maxHttpBufferSize: 1e8, transports: ['websocket', 'polling'], - allowEIO3: true + cors: CORS_CONFIG }); /** @@ -136,17 +150,6 @@ app.get('/', function (req, res) { return res.send('Maxun server started 🚀'); }); -app.use((req, res, next) => { - res.header('Access-Control-Allow-Origin', process.env.PUBLIC_URL || 'http://localhost:5173'); - res.header('Access-Control-Allow-Methods', 'GET,PUT,POST,DELETE,OPTIONS'); - res.header('Access-Control-Allow-Headers', 'Content-Type, Authorization'); - res.header('Access-Control-Allow-Credentials', 'true'); - if (req.method === 'OPTIONS') { - return res.sendStatus(200); - } - next(); -}); - if (require.main === module) { const serverIntervals: NodeJS.Timeout[] = []; diff --git a/server/src/utils/output-post-processor.ts b/server/src/utils/output-post-processor.ts new file mode 100644 index 000000000..37b67df1e --- /dev/null +++ b/server/src/utils/output-post-processor.ts @@ -0,0 +1,193 @@ +import { Page } from 'playwright-core'; +import logger from '../logger'; +import { parseMarkdown } from '../markdownify/markdown'; +import { convertPageToScreenshot } from '../markdownify/scrape'; +import { DEFAULT_OUTPUT_FORMATS } from '../constants/output-formats'; + +interface CategorizedOutput { + crawl: Record; + search: Record; +} + +interface ProcessRobotOutputParams { + robotType: string | undefined; + outputFormats?: string[]; + categorizedOutput: CategorizedOutput; + currentPage: Page; + initialBinaryOutput?: Record; +} + +interface ProcessRobotOutputResult { + categorizedOutput: CategorizedOutput; + binaryOutput: Record; +} + +export async function processRobotOutputFormats( + params: ProcessRobotOutputParams +): Promise { + const { + robotType, + outputFormats, + categorizedOutput, + currentPage, + initialBinaryOutput, + } = params; + + const binaryOutput: Record = { + ...(initialBinaryOutput || {}), + }; + + const effectiveFormats = Array.isArray(outputFormats) + ? (outputFormats.length > 0 + ? outputFormats + : robotType === 'crawl' + ? DEFAULT_OUTPUT_FORMATS + : outputFormats) + : DEFAULT_OUTPUT_FORMATS; + + if (robotType !== 'crawl' && robotType !== 'search') { + return { categorizedOutput, binaryOutput }; + } + + if (robotType === 'crawl' && Array.isArray((categorizedOutput.crawl as any)?.['Crawl Results'])) { + const crawlResults: any[] = (categorizedOutput.crawl as any)['Crawl Results']; + const includeVisibleScreenshot = effectiveFormats.includes('screenshot-visible'); + const includeFullpageScreenshot = effectiveFormats.includes('screenshot-fullpage'); + + for (let pageIndex = 0; pageIndex < crawlResults.length; pageIndex++) { + const pageResult = crawlResults[pageIndex]; + if (!pageResult.error) { + let markdownConversionSucceeded = false; + if (effectiveFormats.includes('markdown') && pageResult.html) { + try { + pageResult.markdown = await parseMarkdown(pageResult.html, pageResult.metadata?.url); + markdownConversionSucceeded = true; + } catch (e: any) { + logger.log('warn', `Failed to convert crawl page to markdown: ${e.message}`); + } + } + + if (!effectiveFormats.includes('html') && markdownConversionSucceeded) { + delete pageResult.html; + } + if (!effectiveFormats.includes('text')) delete pageResult.text; + + const pageUrl = pageResult.metadata?.url || pageResult.url; + const hasPageUrl = typeof pageUrl === 'string' && pageUrl.trim() !== ''; + + if (!includeVisibleScreenshot) { + delete pageResult.screenshotVisible; + } + + if (!includeFullpageScreenshot) { + delete pageResult.screenshotFullpage; + } + + if ((includeVisibleScreenshot || includeFullpageScreenshot) && hasPageUrl) { + if (includeVisibleScreenshot) { + try { + const screenshotBuffer = await convertPageToScreenshot(pageUrl, currentPage, false); + const screenshotKey = `crawl-${pageIndex + 1}-screenshot-visible`; + binaryOutput[screenshotKey] = { + data: screenshotBuffer.toString('base64'), + mimeType: 'image/png', + }; + pageResult.screenshotVisible = screenshotKey; + } catch (e: any) { + logger.log('warn', `Failed to capture visible crawl screenshot for ${pageUrl}: ${e.message}`); + } + } + + if (includeFullpageScreenshot) { + try { + const screenshotBuffer = await convertPageToScreenshot(pageUrl, currentPage, true); + const screenshotKey = `crawl-${pageIndex + 1}-screenshot-fullpage`; + binaryOutput[screenshotKey] = { + data: screenshotBuffer.toString('base64'), + mimeType: 'image/png', + }; + pageResult.screenshotFullpage = screenshotKey; + } catch (e: any) { + logger.log('warn', `Failed to capture fullpage crawl screenshot for ${pageUrl}: ${e.message}`); + } + } + } else if (includeVisibleScreenshot || includeFullpageScreenshot) { + logger.log('warn', `Skipping crawl screenshot capture for page index ${pageIndex} because URL is missing`); + } + } + } + } + + if (robotType === 'search') { + const searchResultGroup = (categorizedOutput.search as any)?.['Search Results']; + if (searchResultGroup?.mode === 'scrape' && Array.isArray(searchResultGroup?.results)) { + const includeVisibleScreenshot = effectiveFormats.includes('screenshot-visible'); + const includeFullpageScreenshot = effectiveFormats.includes('screenshot-fullpage'); + + for (let resultIndex = 0; resultIndex < searchResultGroup.results.length; resultIndex++) { + const result = searchResultGroup.results[resultIndex]; + if (!result.error) { + let markdownConversionSucceeded = false; + if (effectiveFormats.includes('markdown') && result.html) { + try { + result.markdown = await parseMarkdown(result.html, result.metadata?.url); + markdownConversionSucceeded = true; + } catch (e: any) { + logger.log('warn', `Failed to convert search result to markdown: ${e.message}`); + } + } + + if (!effectiveFormats.includes('html') && markdownConversionSucceeded) { + delete result.html; + } + if (!effectiveFormats.includes('text')) delete result.text; + + const resultUrl = result.metadata?.url || result.url; + const hasResultUrl = typeof resultUrl === 'string' && resultUrl.trim() !== ''; + + if (!includeVisibleScreenshot) { + delete result.screenshotVisible; + } + + if (!includeFullpageScreenshot) { + delete result.screenshotFullpage; + } + + if ((includeVisibleScreenshot || includeFullpageScreenshot) && hasResultUrl) { + if (includeVisibleScreenshot) { + try { + const screenshotBuffer = await convertPageToScreenshot(resultUrl, currentPage, false); + const screenshotKey = `search-${resultIndex + 1}-screenshot-visible`; + binaryOutput[screenshotKey] = { + data: screenshotBuffer.toString('base64'), + mimeType: 'image/png', + }; + result.screenshotVisible = screenshotKey; + } catch (e: any) { + logger.log('warn', `Failed to capture visible search screenshot for ${resultUrl}: ${e.message}`); + } + } + + if (includeFullpageScreenshot) { + try { + const screenshotBuffer = await convertPageToScreenshot(resultUrl, currentPage, true); + const screenshotKey = `search-${resultIndex + 1}-screenshot-fullpage`; + binaryOutput[screenshotKey] = { + data: screenshotBuffer.toString('base64'), + mimeType: 'image/png', + }; + result.screenshotFullpage = screenshotKey; + } catch (e: any) { + logger.log('warn', `Failed to capture fullpage search screenshot for ${resultUrl}: ${e.message}`); + } + } + } else if (includeVisibleScreenshot || includeFullpageScreenshot) { + logger.log('warn', `Skipping search screenshot capture for result index ${resultIndex} because URL is missing`); + } + } + } + } + } + + return { categorizedOutput, binaryOutput }; +} diff --git a/server/src/utils/output-validation.ts b/server/src/utils/output-validation.ts new file mode 100644 index 000000000..6b942fed9 --- /dev/null +++ b/server/src/utils/output-validation.ts @@ -0,0 +1,94 @@ +const SEARCH_OR_CRAWL_ERROR_MARKERS = [ + 'Search execution error:', + 'Search action failed:', + 'Crawl execution error:', + 'Crawl action failed:', +]; + +export function getInterpretationFailureReason(logLines: unknown, fallbackMessage: string): string { + if (!Array.isArray(logLines)) { + return fallbackMessage; + } + + const matchedLine = logLines.find( + (line) => + typeof line === 'string' && + SEARCH_OR_CRAWL_ERROR_MARKERS.some((marker) => line.includes(marker)) + ); + + return typeof matchedLine === 'string' && matchedLine.trim().length > 0 + ? matchedLine.trim() + : fallbackMessage; +} + +export function hasExpectedRobotOutput( + robotType: string, + categorizedOutput: { + crawl?: Record; + search?: Record; + }, + selectedFormats?: string[], + binaryOutput?: Record +): boolean { + const formatToFieldMap: Record = { + 'markdown': 'markdown', + 'html': 'html', + 'text': 'text', + 'screenshot-visible': 'screenshotVisible', + 'screenshot-fullpage': 'screenshotFullpage', + }; + + let requestedFields = new Set(); + if (selectedFormats && selectedFormats.length > 0) { + selectedFormats.forEach(format => { + const field = formatToFieldMap[format]; + if (field) requestedFields.add(field); + }); + } + + if (binaryOutput && Object.keys(binaryOutput).length > 0) { + requestedFields.add('binary'); + } + + if (requestedFields.size === 0) { + if (robotType === 'search') { + return Array.isArray((categorizedOutput.search as any)?.['Search Results']?.results) && + (categorizedOutput.search as any)?.['Search Results']?.results.length > 0; + } + if (robotType === 'crawl') { + return Array.isArray((categorizedOutput.crawl as any)?.['Crawl Results']) && + (categorizedOutput.crawl as any)?.['Crawl Results'].length > 0; + } + return true; + } + + if (robotType === 'search') { + const results = (categorizedOutput.search as any)?.['Search Results']?.results; + if (!Array.isArray(results) || results.length === 0) return false; + + return results.some((result: any) => { + if (result.error) return false; + for (const field of requestedFields) { + if (field === 'binary') continue; + if (result[field]) return true; + } + return requestedFields.has('binary'); + }); + } + + if (robotType === 'crawl') { + const results = (categorizedOutput.crawl as any)?.['Crawl Results']; + if (!Array.isArray(results) || results.length === 0) return false; + + return results.some((result: any) => { + if (result.error) return false; + for (const field of requestedFields) { + if (field === 'binary') continue; + if (result[field]) return true; + } + return requestedFields.has('binary'); + }); + } + + return true; +} diff --git a/server/src/workflow-management/classes/Generator.ts b/server/src/workflow-management/classes/Generator.ts index 1fad86a25..c59c626fb 100644 --- a/server/src/workflow-management/classes/Generator.ts +++ b/server/src/workflow-management/classes/Generator.ts @@ -1004,8 +1004,24 @@ export class WorkflowGenerator { logger.log('info', `Robot retrained with id: ${robot.id}`); } } else { + const trimmedFileName = fileName.trim(); + if (!trimmedFileName) { + this.socket.emit('fileSaved', { actionType: 'error' }); + return; + } + const allUserRobots = await Robot.findAll({ where: { userId } as any }); + const normalised = trimmedFileName.toLowerCase(); + const nameConflict = allUserRobots.some( + (r: any) => typeof r.recording_meta?.name === 'string' && + r.recording_meta.name.trim().toLowerCase() === normalised + ); + if (nameConflict) { + this.socket.emit('fileSaved', { actionType: 'nameExists' }); + return; + } + this.recordingMeta = { - name: fileName, + name: trimmedFileName, id: uuid(), createdAt: this.recordingMeta.createdAt || new Date().toLocaleString(), pairs: recording.workflow.length, @@ -1031,8 +1047,11 @@ export class WorkflowGenerator { logger.log('info', `Robot saved with id: ${robot.id}`); } } - catch (e) { - const { message } = e as Error; + catch (e: any) { + if (e.name === 'SequelizeUniqueConstraintError' || e.parent?.code === '23505') { + this.socket.emit('fileSaved', { actionType: 'nameExists' }); + return; + } logger.log('warn', `Cannot save the file to the local file system ${e}`) actionType = 'error'; } diff --git a/server/src/workflow-management/scheduler/index.ts b/server/src/workflow-management/scheduler/index.ts index 8534c2dd4..8ef7f4638 100644 --- a/server/src/workflow-management/scheduler/index.ts +++ b/server/src/workflow-management/scheduler/index.ts @@ -14,6 +14,8 @@ import { Page } from "playwright-core"; import { sendWebhook } from "../../routes/webhook"; import { addAirtableUpdateTask, airtableUpdateTasks, processAirtableUpdates } from "../integrations/airtable"; import { convertPageToMarkdown, convertPageToHTML, convertPageToScreenshot } from "../../markdownify/scrape"; +import { processRobotOutputFormats } from "../../utils/output-post-processor"; +import { getInterpretationFailureReason, hasExpectedRobotOutput } from "../../utils/output-validation"; async function createWorkflowAndStoreMetadata(id: string, userId: string) { try { @@ -479,9 +481,6 @@ async function executeRun(id: string, userId: string) { const interpretationInfo = await Promise.race([interpretationPromise, timeoutPromise]); - const binaryOutputService = new BinaryOutputService('maxun-run-screenshots'); - const uploadedBinaryOutput = await binaryOutputService.uploadAndStoreBinaryOutput(run, interpretationInfo.binaryOutput); - const finalRun = await Run.findByPk(run.id); const categorizedOutput = { scrapeSchema: finalRun?.serializableOutput?.scrapeSchema || {}, @@ -490,6 +489,54 @@ async function executeRun(id: string, userId: string) { search: finalRun?.serializableOutput?.search || {} }; + let binaryOutput: Record = { + ...(interpretationInfo.binaryOutput || {}) + }; + + const robotType = recording.recording_meta.type; + const outputFormats = (recording.recording_meta as any).formats as string[] | undefined; + + if (robotType === 'crawl' || robotType === 'search') { + const processedOutput = await processRobotOutputFormats({ + robotType, + outputFormats, + categorizedOutput: { + crawl: categorizedOutput.crawl as Record, + search: categorizedOutput.search as Record, + }, + currentPage, + initialBinaryOutput: binaryOutput, + }); + + categorizedOutput.crawl = processedOutput.categorizedOutput.crawl; + categorizedOutput.search = processedOutput.categorizedOutput.search; + binaryOutput = processedOutput.binaryOutput; + + await run.update({ + serializableOutput: { + ...(finalRun?.serializableOutput || {}), + crawl: categorizedOutput.crawl, + search: categorizedOutput.search, + } + }); + + const hasOutput = hasExpectedRobotOutput(robotType, { + crawl: categorizedOutput.crawl as Record, + search: categorizedOutput.search as Record + }); + + if (!hasOutput) { + const humanRobotType = robotType.charAt(0).toUpperCase() + robotType.slice(1); + const fallbackReason = `${humanRobotType} run completed without producing output data`; + throw new Error(getInterpretationFailureReason(interpretationInfo.log, fallbackReason)); + } + } + + const binaryOutputService = new BinaryOutputService('maxun-run-screenshots'); + const uploadedBinaryOutput = Object.keys(binaryOutput).length > 0 + ? await binaryOutputService.uploadAndStoreBinaryOutput(run, binaryOutput) + : {}; + await destroyRemoteBrowser(plainRun.browserId, userId); await run.update({ @@ -502,7 +549,7 @@ async function executeRun(id: string, userId: string) { // Get metrics from persisted data for analytics and webhooks let totalSchemaItemsExtracted = 0; let totalListItemsExtracted = 0; - let extractedScreenshotsCount = 0; + let extractedScreenshotsCount = Object.keys(uploadedBinaryOutput).length; if (categorizedOutput) { if (categorizedOutput.scrapeSchema) { @@ -524,10 +571,6 @@ async function executeRun(id: string, userId: string) { } } - if (run.binaryOutput) { - extractedScreenshotsCount = Object.keys(run.binaryOutput).length; - } - const totalRowsExtracted = totalSchemaItemsExtracted + totalListItemsExtracted; capture( diff --git a/src/api/storage.ts b/src/api/storage.ts index b8c751111..d07913323 100644 --- a/src/api/storage.ts +++ b/src/api/storage.ts @@ -53,6 +53,8 @@ export const createScrapeRobot = async ( throw new Error('Failed to create markdown robot'); } } catch (error: any) { + const message = error.response?.data?.error; + if (message) throw new Error(message); console.error('Error creating markdown robot:', error); return null; } @@ -92,6 +94,8 @@ export const createLLMRobot = async ( throw new Error('Failed to create LLM robot'); } } catch (error: any) { + const message = error.response?.data?.error; + if (message) throw new Error(message); console.error('Error creating LLM robot:', error); return null; } @@ -103,6 +107,7 @@ export const updateRecording = async (id: string, data: { credentials?: Credentials; targetUrl?: string; workflow?: any[]; + formats?: string[]; }): Promise => { try { const response = await axios.put(`${apiUrl}/storage/recordings/${id}`, data); @@ -112,6 +117,13 @@ export const updateRecording = async (id: string, data: { throw new Error(`Couldn't update recording with id ${id}`); } } catch (error: any) { + const message = error.response?.data?.error; + const status = error.response?.status; + if (message) { + const err = new Error(message) as any; + err.isDuplicate = status === 409; + throw err; + } console.error(`Error updating recording: ${error.message}`); return false; } @@ -131,10 +143,11 @@ export const getStoredRuns = async (): Promise => { } }; -export const duplicateRecording = async (id: string, targetUrl: string): Promise => { +export const duplicateRecording = async (id: string, targetUrl: string, newName?: string): Promise => { try { const response = await axios.post(`${apiUrl}/storage/recordings/${id}/duplicate`, { targetUrl, + newName, }, { withCredentials: true }); if (response.status === 201) { return response.data; @@ -142,6 +155,8 @@ export const duplicateRecording = async (id: string, targetUrl: string): Promise throw new Error(`Couldn't duplicate recording with id ${id}`); } } catch (error: any) { + const message = error.response?.data?.error; + if (message) throw new Error(message); console.error(`Error duplicating recording: ${error.message}`); return null; } @@ -349,7 +364,8 @@ export const createCrawlRobot = async ( useSitemap: boolean; followLinks: boolean; respectRobots: boolean; - } + }, + formats: string[] = ['markdown'] ): Promise => { try { const response = await axios.post( @@ -358,6 +374,7 @@ export const createCrawlRobot = async ( url, name, crawlConfig, + formats, }, { headers: { 'Content-Type': 'application/json' }, @@ -371,6 +388,8 @@ export const createCrawlRobot = async ( throw new Error('Failed to create crawl robot'); } } catch (error: any) { + const message = error.response?.data?.error; + if (message) throw new Error(message); console.error('Error creating crawl robot:', error); return null; } @@ -388,7 +407,8 @@ export const createSearchRobot = async ( lang?: string; }; mode: 'discover' | 'scrape'; - } + }, + formats?: string[] ): Promise => { try { const response = await axios.post( @@ -396,6 +416,7 @@ export const createSearchRobot = async ( { name, searchConfig, + formats: formats || [], }, { headers: { 'Content-Type': 'application/json' }, @@ -409,6 +430,8 @@ export const createSearchRobot = async ( throw new Error('Failed to create search robot'); } } catch (error: any) { + const message = error.response?.data?.error; + if (message) throw new Error(message); console.error('Error creating search robot:', error); return null; } diff --git a/src/components/api/ApiKey.tsx b/src/components/api/ApiKey.tsx index f75eb2812..8183fd3c4 100644 --- a/src/components/api/ApiKey.tsx +++ b/src/components/api/ApiKey.tsx @@ -94,11 +94,34 @@ const ApiKeyManager = () => { }; const copyToClipboard = () => { - if (apiKey) { - navigator.clipboard.writeText(apiKey); - setCopySuccess(true); - setTimeout(() => setCopySuccess(false), 2000); - notify('info', t('apikey.notifications.copy_success')); + if (!apiKey) return; + + if (navigator.clipboard) { + navigator.clipboard.writeText(apiKey).then(() => { + setCopySuccess(true); + setTimeout(() => setCopySuccess(false), 2000); + notify('info', t('apikey.notifications.copy_success')); + }).catch(() => { + notify('error', t('apikey.notifications.copy_error')); + }); + } else { + const textarea = document.createElement('textarea'); + textarea.value = apiKey; + textarea.style.position = 'fixed'; + textarea.style.opacity = '0'; + document.body.appendChild(textarea); + textarea.focus(); + textarea.select(); + try { + document.execCommand('copy'); + setCopySuccess(true); + setTimeout(() => setCopySuccess(false), 2000); + notify('info', t('apikey.notifications.copy_success')); + } catch { + notify('error', t('apikey.notifications.copy_error')); + } finally { + document.body.removeChild(textarea); + } } }; diff --git a/src/components/browser/BrowserRecordingSave.tsx b/src/components/browser/BrowserRecordingSave.tsx index 5beed5d03..7172dbf95 100644 --- a/src/components/browser/BrowserRecordingSave.tsx +++ b/src/components/browser/BrowserRecordingSave.tsx @@ -1,5 +1,5 @@ import React, { useState } from 'react' -import { Grid, Button, Box, Typography, IconButton, Menu, MenuItem, ListItemText } from '@mui/material'; +import { Grid, Button, Box, Typography, IconButton, Menu, MenuItem, ListItemText, Dialog, DialogTitle, DialogActions, } from '@mui/material'; import { SaveRecording } from "../recorder/SaveRecording"; import { useGlobalInfoStore } from '../../context/globalInfo'; import { useActionContext } from '../../context/browserActions'; @@ -20,9 +20,9 @@ const BrowserRecordingSave = () => { const { socket } = useSocketStore(); - const { - stopGetText, - stopGetList, + const { + stopGetText, + stopGetList, stopGetScreenshot, stopPaginationMode, stopLimitMode, @@ -45,7 +45,7 @@ const BrowserRecordingSave = () => { timestamp: Date.now() }; window.sessionStorage.setItem('pendingNotification', JSON.stringify(notificationData)); - + if (window.opener) { window.opener.postMessage({ type: 'recording-notification', @@ -57,9 +57,9 @@ const BrowserRecordingSave = () => { timestamp: Date.now() }, '*'); } - + setBrowserId(null); - + window.close(); stopRecording(browserId).catch((error) => { @@ -74,7 +74,7 @@ const BrowserRecordingSave = () => { stopGetScreenshot(); stopPaginationMode(); stopLimitMode(); - + setShowLimitOptions(false); setShowPaginationOptions(false); setCaptureStage('initial'); @@ -97,7 +97,7 @@ const BrowserRecordingSave = () => { browserSteps.forEach(step => { deleteBrowserStep(step.id); }); - + if (socket) { socket?.emit('new-recording'); socket.emit('input:url', initialUrl); @@ -119,7 +119,7 @@ const BrowserRecordingSave = () => { }; const handleClick = (event: any) => { - setAnchorEl(event.currentTarget); + setAnchorEl(event.currentTarget); }; const handleClose = () => { @@ -183,19 +183,39 @@ const BrowserRecordingSave = () => { - setOpenDiscardModal(false)} modalStyle={modalStyle}> - - {t('browser_recording.modal.confirm_discard')} - - - - - - + setOpenDiscardModal(false)} + maxWidth="xs" + fullWidth + PaperProps={{ + sx: { + p: 0, + borderRadius: 2, + border: "none" + } + }} + > + + {t('browser_recording.modal.confirm_discard')} + + + + + + + setOpenResetModal(false)} modalStyle={modalStyle}> @@ -204,9 +224,9 @@ const BrowserRecordingSave = () => { {t('browser_recording.modal.reset_warning')} - - setDocModalOpen(false)}> - + setDocModalOpen(false)} + maxWidth="xs" + fullWidth + PaperProps={{ + sx: { + borderRadius: 2, + width: 400 + } + }} + > + + + + @@ -224,20 +243,20 @@ export const MainMenu = ({ value = 'robots', handleChangeContent }: MainMenuProp rel="noopener noreferrer" sx={starButtonStyles} startIcon={ - } > - Star On GitHub + Star On GitHub {isLoading ? ( - - setSponsorModalOpen(false)}> - - - Support Maxun Open Source - + setSponsorModalOpen(false)} + maxWidth="sm" + fullWidth + PaperProps={{ + sx: { + borderRadius: 2, + width: 600 + } + }} + > + + Support Maxun Open Source + + + - Maxun is built by a small, full-time team. Your donations directly contribute to making it better. + Maxun is built by a small, full-time team. Your donations directly + contribute to making it better.
Thank you for your support! 🩷
+ - -
-
+ + ); }; \ No newline at end of file diff --git a/src/components/recorder/SaveRecording.tsx b/src/components/recorder/SaveRecording.tsx index af877503e..a6a43ffc6 100644 --- a/src/components/recorder/SaveRecording.tsx +++ b/src/components/recorder/SaveRecording.tsx @@ -1,13 +1,10 @@ import React, { useCallback, useEffect, useState, useContext } from 'react'; -import { Button, Box, LinearProgress, Tooltip } from "@mui/material"; -import { GenericModal } from "../ui/GenericModal"; +import { Button, Box, LinearProgress, Tooltip, Dialog, DialogTitle, DialogContent } from "@mui/material"; import { stopRecording } from "../../api/recording"; import { useGlobalInfoStore } from "../../context/globalInfo"; import { AuthContext } from '../../context/auth'; import { useSocketStore } from "../../context/socket"; -import { TextField, Typography } from "@mui/material"; -import { WarningText } from "../ui/texts"; -import NotificationImportantIcon from "@mui/icons-material/NotificationImportant"; +import { TextField } from "@mui/material"; import { useNavigate } from 'react-router-dom'; import { useTranslation } from 'react-i18next'; @@ -18,7 +15,6 @@ interface SaveRecordingProps { export const SaveRecording = ({ fileName }: SaveRecordingProps) => { const { t } = useTranslation(); const [openModal, setOpenModal] = useState(false); - const [needConfirm, setNeedConfirm] = useState(false); const [saveRecordingName, setSaveRecordingName] = useState(fileName); const [waitingForSave, setWaitingForSave] = useState(false); @@ -35,27 +31,23 @@ export const SaveRecording = ({ fileName }: SaveRecordingProps) => { }, [recordingName]); const handleChangeOfTitle = (event: React.ChangeEvent) => { - const { value } = event.target; - if (needConfirm) { - setNeedConfirm(false); - } - setSaveRecordingName(value); + setSaveRecordingName(event.target.value); } const handleSaveRecording = async (event: React.SyntheticEvent) => { event.preventDefault(); - if (recordings.includes(saveRecordingName)) { - if (needConfirm) { return; } - setNeedConfirm(true); - } else { - await saveRecording(); + const trimmedName = saveRecordingName.trim(); + if (!retrainRobotId && recordings.some(r => r.trim().toLowerCase() === trimmedName.toLowerCase())) { + notify('error', t('save_recording.errors.name_exists')); + return; } + await saveRecording(); }; const handleFinishClick = () => { const { hasScrapeListAction, hasScreenshotAction, hasScrapeSchemaAction } = currentWorkflowActionsState; const hasAnyAction = hasScrapeListAction || hasScreenshotAction || hasScrapeSchemaAction; - + if (!hasAnyAction) { notify('warning', t('save_recording.errors.no_actions_performed')); return; @@ -70,7 +62,7 @@ export const SaveRecording = ({ fileName }: SaveRecordingProps) => { const exitRecording = useCallback(async (data?: { actionType: string }) => { let successMessage = t('save_recording.notifications.save_success'); - + if (data && data.actionType) { if (data.actionType === 'retrained') { successMessage = t('save_recording.notifications.retrain_success'); @@ -80,31 +72,31 @@ export const SaveRecording = ({ fileName }: SaveRecordingProps) => { successMessage = t('save_recording.notifications.save_error'); } } - + const notificationData = { type: data?.actionType === 'error' ? 'error' : 'success', message: successMessage, timestamp: Date.now() }; window.sessionStorage.setItem('pendingNotification', JSON.stringify(notificationData)); - + if (window.opener) { window.opener.postMessage({ type: 'recording-notification', notification: notificationData }, '*'); - + window.opener.postMessage({ type: 'session-data-clear', timestamp: Date.now() }, '*'); } - + if (browserId) { await stopRecording(browserId); } setBrowserId(null); - + window.close(); }, [setBrowserId, browserId, t]); @@ -114,15 +106,15 @@ export const SaveRecording = ({ fileName }: SaveRecordingProps) => { if (user) { const { hasScrapeListAction, hasScreenshotAction, hasScrapeSchemaAction } = currentWorkflowActionsState; const hasAnyAction = hasScrapeListAction || hasScreenshotAction || hasScrapeSchemaAction; - + if (!hasAnyAction) { notify('warning', t('save_recording.errors.no_actions_performed')); return; } - const payload = { - fileName: saveRecordingName || recordingName, - userId: user.id, + const payload = { + fileName: (saveRecordingName || recordingName).trim(), + userId: user.id, isLogin: isLogin, robotId: retrainRobotId, }; @@ -134,12 +126,21 @@ export const SaveRecording = ({ fileName }: SaveRecordingProps) => { } }; + const handleFileSaved = useCallback(async (data?: { actionType: string }) => { + if (data?.actionType === 'nameExists') { + setWaitingForSave(false); + notify('error', t('save_recording.errors.name_exists')); + return; + } + await exitRecording(data); + }, [exitRecording, notify, t]); + useEffect(() => { - socket?.on('fileSaved', exitRecording); + socket?.on('fileSaved', handleFileSaved); return () => { - socket?.off('fileSaved', exitRecording); + socket?.off('fileSaved', handleFileSaved); } - }, [socket, exitRecording]); + }, [socket, handleFileSaved]); return (
@@ -158,42 +159,54 @@ export const SaveRecording = ({ fileName }: SaveRecordingProps) => { {t('right_panel.buttons.finish')} - setOpenModal(false)} modalStyle={modalStyle}> -
- {t('save_recording.title')} - - {needConfirm - ? - ( - - - - {t('save_recording.errors.exists_warning')} - - ) - : - } - {waitingForSave && - - - - - - } - -
+ + {waitingForSave && ( + + + + + + )} + + +
); } diff --git a/src/components/robot/RecordingsTable.tsx b/src/components/robot/RecordingsTable.tsx index f1e9bbe02..dfb69de00 100644 --- a/src/components/robot/RecordingsTable.tsx +++ b/src/components/robot/RecordingsTable.tsx @@ -23,6 +23,11 @@ import { CircularProgress, FormControlLabel, Checkbox, + Dialog, + DialogTitle, + DialogContent, + DialogContentText, + DialogActions, } from "@mui/material"; import { Schedule, @@ -94,7 +99,7 @@ const LoadingRobotRow = memo(({ row, columns }: any) => { } else if (column.id === 'interpret') { return ( - - + - ); } else { @@ -181,7 +186,7 @@ export const RecordingsTable = ({ handleSettingsRecording, handleEditRobot, handleDuplicateRobot, - }: RecordingsTableProps) => { +}: RecordingsTableProps) => { const { t } = useTranslation(); const theme = useTheme(); const [page, setPage] = React.useState(0); @@ -227,11 +232,11 @@ export const RecordingsTable = ({ const notificationData = event.data.notification; if (notificationData) { notify(notificationData.type, notificationData.message); - - if ((notificationData.type === 'success' && - (notificationData.message.includes('saved') || notificationData.message.includes('retrained'))) || - (notificationData.type === 'warning' && - notificationData.message.includes('terminated'))) { + + if ((notificationData.type === 'success' && + (notificationData.message.includes('saved') || notificationData.message.includes('retrained'))) || + (notificationData.type === 'warning' && + notificationData.message.includes('terminated'))) { setRerenderRobots(true); } } @@ -248,9 +253,9 @@ export const RecordingsTable = ({ window.sessionStorage.removeItem('initialUrl'); } }; - + window.addEventListener('message', handleMessage); - + return () => { window.removeEventListener('message', handleMessage); }; @@ -325,7 +330,7 @@ export const RecordingsTable = ({ timestamp: Date.now() }; window.sessionStorage.setItem('recordingTabCloseMessage', JSON.stringify(closeMessage)); - + if (window.openedRecordingWindow && !window.openedRecordingWindow.closed) { try { window.openedRecordingWindow.close(); @@ -339,10 +344,10 @@ export const RecordingsTable = ({ if (activeBrowserId) { await stopRecording(activeBrowserId); notify('warning', t('browser_recording.notifications.terminated')); - + notifyRecordingTabsToClose(activeBrowserId); } - + setWarningModalOpen(false); setModalOpen(true); }; @@ -350,31 +355,31 @@ export const RecordingsTable = ({ const handleRetrainRobot = useCallback(async (id: string, name: string) => { const robot = rows.find(row => row.id === id); let targetUrl; - + if (robot?.content?.workflow && robot.content.workflow.length > 0) { const lastPair = robot.content.workflow[robot.content.workflow.length - 1]; - + if (lastPair?.what) { if (Array.isArray(lastPair.what)) { - const gotoAction = lastPair.what.find((action: any) => + const gotoAction = lastPair.what.find((action: any) => action && typeof action === 'object' && 'action' in action && action.action === "goto" ) as any; - + if (gotoAction?.args?.[0]) { targetUrl = gotoAction.args[0]; } } } } - + if (targetUrl) { setInitialUrl(targetUrl); setRecordingUrl(targetUrl); window.sessionStorage.setItem('initialUrl', targetUrl); } - + const canCreateRecording = await canCreateBrowserInState("recording"); - + if (!canCreateRecording) { const activeBrowserId = await getActiveBrowserId(); if (activeBrowserId) { @@ -384,45 +389,45 @@ export const RecordingsTable = ({ notify('warning', t('recordingtable.notifications.browser_limit_warning')); } } else { - startRetrainRecording(id, name, targetUrl); + startRetrainRecording(id, name, targetUrl); } }, [rows, setInitialUrl, setRecordingUrl]); const startRetrainRecording = (id: string, name: string, url?: string) => { setBrowserId('new-recording'); - setRecordingName(name); - setRecordingId(id); - + setRecordingName(name); + setRecordingId(id); + window.sessionStorage.setItem('browserId', 'new-recording'); window.sessionStorage.setItem('robotToRetrain', id); window.sessionStorage.setItem('robotName', name); - + window.sessionStorage.setItem('recordingUrl', url || recordingUrl); - + const sessionId = Date.now().toString(); window.sessionStorage.setItem('recordingSessionId', sessionId); - + window.openedRecordingWindow = window.open(`/recording-setup?session=${sessionId}`, '_blank'); - + window.sessionStorage.setItem('nextTabIsRecording', 'true'); }; const startRecording = () => { setModalOpen(false); - + // Set local state setBrowserId('new-recording'); setRecordingName(''); setRecordingId(''); - + window.sessionStorage.setItem('browserId', 'new-recording'); - + const sessionId = Date.now().toString(); window.sessionStorage.setItem('recordingSessionId', sessionId); window.sessionStorage.setItem('recordingUrl', recordingUrl); - + window.openedRecordingWindow = window.open(`/recording-setup?session=${sessionId}`, '_blank'); - + window.sessionStorage.setItem('nextTabIsRecording', 'true'); }; @@ -442,17 +447,17 @@ export const RecordingsTable = ({ function useDebounce(value: T, delay: number): T { const [debouncedValue, setDebouncedValue] = React.useState(value); - + useEffect(() => { const handler = setTimeout(() => { setDebouncedValue(value); }, delay); - + return () => { clearTimeout(handler); }; }, [value, delay]); - + return debouncedValue; } @@ -472,9 +477,9 @@ export const RecordingsTable = ({ }, [filteredRows, page, rowsPerPage]); const openDeleteConfirm = React.useCallback((id: string) => { - setPendingDeleteId(String(id)); - setDeleteConfirmOpen(true); - }, []); + setPendingDeleteId(String(id)); + setDeleteConfirmOpen(true); + }, []); const confirmDeleteRecording = React.useCallback(async () => { if (!pendingDeleteId) return; @@ -548,13 +553,13 @@ export const RecordingsTable = ({
- + {isFetching ? ( {debouncedSearchTerm ? t('recordingtable.placeholder.search') : t('recordingtable.placeholder.title')} - {debouncedSearchTerm + {debouncedSearchTerm ? t('recordingtable.search_criteria') : t('recordingtable.placeholder.body') } @@ -587,16 +592,16 @@ export const RecordingsTable = ({ <> - - {columns.map((column) => ( - - {column.label} - - ))} - + + {columns.map((column) => ( + + {column.label} + + ))} + {visibleRows.map((row) => ( )} - setWarningModalOpen(false)} modalStyle={modalStyle}> -
- {t('recordingtable.warning_modal.title')} - + setWarningModalOpen(false)} + maxWidth="xs" + fullWidth + PaperProps={{ + sx: { + p: 0, + borderRadius: 2 + } + }} + > + + {t('recordingtable.warning_modal.title')} + + + + {t('recordingtable.warning_modal.message')} - - - - - -
-
+ + + + + + + setModalOpen(false)} modalStyle={modalStyle}>
{t('recordingtable.modal.title')} @@ -679,33 +698,50 @@ export const RecordingsTable = ({
- { setDeleteConfirmOpen(false); setPendingDeleteId(null); }} - modalStyle={{ ...modalStyle, padding: 0, backgroundColor: 'transparent', width: 'auto', maxWidth: '520px' }} + { + setDeleteConfirmOpen(false); + setPendingDeleteId(null); + }} + maxWidth="xs" + fullWidth > + + {t('recordingtable.delete_confirm.title', { + name: pendingRow?.name, + defaultValue: 'Delete {{name}}?' + })} + + + + + {t('recordingtable.delete_confirm.message', { + name: pendingRow?.name, + defaultValue: 'Are you sure you want to delete the robot "{{name}}"?' + })} + + + + + - - - {t('recordingtable.delete_confirm.title', { name: pendingRow?.name, defaultValue: 'Delete {{name}}?' })} - - - {t('recordingtable.delete_confirm.message', { - name: pendingRow?.name, - defaultValue: 'Are you sure you want to delete the robot "{{name}}"?' - })} - - - - - - - - + + + ); } @@ -838,7 +874,6 @@ const OptionsButton = ({ handleRetrain, handleEdit, handleDuplicate, handleDelet const MemoizedTableCell = memo(TableCell); -// Memoized action buttons const MemoizedInterpretButton = memo(InterpretButton); const MemoizedScheduleButton = memo(ScheduleButton); const MemoizedIntegrateButton = memo(IntegrateButton); diff --git a/src/components/robot/pages/RobotCreate.tsx b/src/components/robot/pages/RobotCreate.tsx index acb003b22..f0a8825e1 100644 --- a/src/components/robot/pages/RobotCreate.tsx +++ b/src/components/robot/pages/RobotCreate.tsx @@ -18,6 +18,10 @@ import { Select, MenuItem, InputLabel, + Dialog, + DialogTitle, + DialogContent, + DialogActions, Collapse, FormControlLabel } from '@mui/material'; @@ -27,6 +31,7 @@ import { canCreateBrowserInState, getActiveBrowserId, stopRecording } from '../. import { createScrapeRobot, createLLMRobot, createAndRunRecording, createCrawlRobot, createSearchRobot } from "../../../api/storage"; import { AuthContext } from '../../../context/auth'; import { GenericModal } from '../../ui/GenericModal'; +import { DEFAULT_OUTPUT_FORMATS, OUTPUT_FORMAT_LABELS, OUTPUT_FORMAT_OPTIONS, OutputFormat } from '../../../constants/outputFormats'; interface TabPanelProps { @@ -54,7 +59,7 @@ function TabPanel(props: TabPanelProps) { const RobotCreate: React.FC = () => { const { t } = useTranslation(); const navigate = useNavigate(); - const { setBrowserId, setRecordingUrl, notify, setRecordingId, setRerenderRobots } = useGlobalInfoStore(); + const { setBrowserId, setRecordingUrl, notify, setRecordingId, setRerenderRobots, recordings } = useGlobalInfoStore(); const [tabValue, setTabValue] = useState(0); const [url, setUrl] = useState(''); @@ -93,9 +98,12 @@ const RobotCreate: React.FC = () => { const [searchMode, setSearchMode] = useState<'discover' | 'scrape'>('discover'); const [searchTimeRange, setSearchTimeRange] = useState<'day' | 'week' | 'month' | 'year' | ''>(''); + const [crawlOutputFormats, setCrawlOutputFormats] = useState(DEFAULT_OUTPUT_FORMATS); + const [searchOutputFormats, setSearchOutputFormats] = useState(DEFAULT_OUTPUT_FORMATS); + const { state } = React.useContext(AuthContext); const { user } = state; - const { addOptimisticRobot, removeOptimisticRobot, invalidateRecordings, invalidateRuns, addOptimisticRun } = useCacheInvalidation(); + const { addOptimisticRobot, removeOptimisticRobot, invalidateRecordings, invalidateRuns, updateOptimisticRun } = useCacheInvalidation(); const handleTabChange = (event: React.SyntheticEvent, newValue: number) => { setTabValue(newValue); @@ -135,6 +143,7 @@ const RobotCreate: React.FC = () => { const sessionId = Date.now().toString(); window.sessionStorage.setItem('recordingSessionId', sessionId); + window.sessionStorage.setItem('recordingOriginPage', window.location.pathname + window.location.search); window.open(`/recording-setup?session=${sessionId}`, '_blank'); window.sessionStorage.setItem('nextTabIsRecording', 'true'); @@ -169,6 +178,7 @@ const RobotCreate: React.FC = () => { const sessionId = Date.now().toString(); window.sessionStorage.setItem('recordingSessionId', sessionId); + window.sessionStorage.setItem('recordingOriginPage', window.location.pathname + window.location.search); window.open(`/recording-setup?session=${sessionId}`, '_blank'); window.sessionStorage.setItem('nextTabIsRecording', 'true'); @@ -185,30 +195,39 @@ const RobotCreate: React.FC = () => { notify('error', 'Please enter a robot name'); return; } + if (crawlOutputFormats.length === 0) { + notify('error', 'Please select at least one output format'); + return; + } setIsLoading(true); - const result = await createCrawlRobot( - crawlUrl, - crawlRobotName, - { - mode: crawlMode, - limit: crawlLimit, - maxDepth: crawlMaxDepth, - includePaths: crawlIncludePaths ? crawlIncludePaths.split(',').map(p => p.trim()) : [], - excludePaths: crawlExcludePaths ? crawlExcludePaths.split(',').map(p => p.trim()) : [], - useSitemap: crawlUseSitemap, - followLinks: crawlFollowLinks, - respectRobots: crawlRespectRobots + try { + const result = await createCrawlRobot( + crawlUrl, + crawlRobotName, + { + mode: crawlMode, + limit: crawlLimit, + maxDepth: crawlMaxDepth, + includePaths: crawlIncludePaths ? crawlIncludePaths.split(',').map(p => p.trim()) : [], + excludePaths: crawlExcludePaths ? crawlExcludePaths.split(',').map(p => p.trim()) : [], + useSitemap: crawlUseSitemap, + followLinks: crawlFollowLinks, + respectRobots: crawlRespectRobots + }, + crawlOutputFormats + ); + setIsLoading(false); + if (result) { + invalidateRecordings(); + notify('success', `${crawlRobotName} created successfully!`); + navigate('/robots'); + } else { + notify('error', 'Failed to create crawl robot'); } - ); - setIsLoading(false); - - if (result) { - invalidateRecordings(); - notify('success', `${crawlRobotName} created successfully!`); - navigate('/robots'); - } else { - notify('error', 'Failed to create crawl robot'); + } catch (error: any) { + setIsLoading(false); + notify('error', error.message || 'Failed to create crawl robot'); } }; @@ -221,28 +240,39 @@ const RobotCreate: React.FC = () => { notify('error', 'Please enter a robot name'); return; } + if (searchMode === 'scrape' && searchOutputFormats.length === 0) { + notify('error', 'Please select at least one output format'); + return; + } setIsLoading(true); - const result = await createSearchRobot( - searchRobotName, - { - query: searchQuery, - limit: searchLimit, - provider: searchProvider, - filters: { - timeRange: searchTimeRange ? searchTimeRange as 'day' | 'week' | 'month' | 'year' : undefined + try { + const formatsForRequest = searchMode === 'discover' ? [] : searchOutputFormats; + + const result = await createSearchRobot( + searchRobotName, + { + query: searchQuery, + limit: searchLimit, + provider: searchProvider, + filters: { + timeRange: searchTimeRange ? searchTimeRange as 'day' | 'week' | 'month' | 'year' : undefined + }, + mode: searchMode }, - mode: searchMode + formatsForRequest + ); + setIsLoading(false); + if (result) { + invalidateRecordings(); + notify('success', `${searchRobotName} created successfully!`); + navigate('/robots'); + } else { + notify('error', 'Failed to create search robot'); } - ); - setIsLoading(false); - - if (result) { - invalidateRecordings(); - notify('success', `${searchRobotName} created successfully!`); - navigate('/robots'); - } else { - notify('error', 'Failed to create search robot'); + } catch (error: any) { + setIsLoading(false); + notify('error', error.message || 'Failed to create search robot'); } }; @@ -342,7 +372,7 @@ const RobotCreate: React.FC = () => { } }} > - + Recorder Mode @@ -383,255 +413,259 @@ const RobotCreate: React.FC = () => { Beta - + AI Mode - Describe the task. It builds it for you. + Describe the task. Maxun builds it for you. - {generationMode === 'agent' && ( - - - setExtractRobotName(e.target.value)} - label="Name" - /> - + {generationMode === 'agent' && ( + + + setExtractRobotName(e.target.value)} + label="Name" + /> + + + setAiPrompt(e.target.value)} + label="Extraction Prompt" + /> + + + + setUrl(e.target.value)} + label="Website URL (Optional)" + /> + + + + + LLM Provider + + + + + Model + + + + + {/* API Key for non-Ollama providers */} + {llmProvider !== 'ollama' && ( setAiPrompt(e.target.value)} - label="Extraction Prompt" + type="password" + value={llmApiKey} + onChange={(e) => setLlmApiKey(e.target.value)} + label="API Key (Optional if set in .env)" /> + )} + {llmProvider === 'ollama' && ( setUrl(e.target.value)} - label="Website URL (Optional)" + value={llmBaseUrl} + onChange={(e) => setLlmBaseUrl(e.target.value)} + label="Ollama Base URL (Optional)" /> + )} - - - LLM Provider - - - - - Model - - - + - - )} + } catch (error: any) { + console.error('Error in AI robot creation:', error); + removeOptimisticRobot(tempRobotId); + invalidateRecordings(); + notify('error', error?.message || 'Failed to create and run AI robot'); + } + }} + disabled={!extractRobotName.trim() || !aiPrompt.trim() || isLoading} + sx={{ + bgcolor: '#ff00c3', + py: 1.4, + fontSize: '1rem', + textTransform: 'none', + borderRadius: 2 + }} + startIcon={isLoading ? : null} + > + {isLoading ? 'Creating & Running...' : 'Create & Run Robot'} + + + )} - {generationMode === 'recorder' && ( + {generationMode === 'recorder' && ( <> { - )} - + )} + @@ -803,15 +837,19 @@ const RobotCreate: React.FC = () => { } setIsLoading(true); - const result = await createScrapeRobot(url, scrapeRobotName, outputFormats); - setIsLoading(false); - - if (result) { - setRerenderRobots(true); - notify('success', `${scrapeRobotName} created successfully!`); - navigate('/robots'); - } else { - notify('error', 'Failed to create scrape robot'); + try { + const result = await createScrapeRobot(url, scrapeRobotName, outputFormats); + setIsLoading(false); + if (result) { + setRerenderRobots(true); + notify('success', `${scrapeRobotName} created successfully!`); + navigate('/robots'); + } else { + notify('error', 'Failed to create scrape robot'); + } + } catch (error: any) { + setIsLoading(false); + notify('error', error.message || 'Failed to create scrape robot'); } }} disabled={!url.trim() || !scrapeRobotName.trim() || outputFormats.length === 0 || isLoading} @@ -879,15 +917,42 @@ const RobotCreate: React.FC = () => { sx={{ mb: 2 }} /> + + + Output Formats * + + + + @@ -976,7 +1041,7 @@ const RobotCreate: React.FC = () => { variant="contained" fullWidth onClick={handleCreateCrawlRobot} - disabled={!crawlUrl.trim() || !crawlRobotName.trim() || isLoading} + disabled={!crawlUrl.trim() || !crawlRobotName.trim() || crawlOutputFormats.length === 0 || isLoading} sx={{ bgcolor: '#ff00c3', py: 1.4, @@ -1041,39 +1106,82 @@ const RobotCreate: React.FC = () => { - Mode - + Mode + - Time Range - + Time Range + + + {searchMode === 'scrape' ? ( + + + Output Formats * + + + + ) : ( + + + Output formats are only available in "Extract Data from Results" mode + + + )} - - - - - + + + + + ); diff --git a/src/components/robot/pages/RobotDuplicatePage.tsx b/src/components/robot/pages/RobotDuplicatePage.tsx index cff3a0c55..a8393f233 100644 --- a/src/components/robot/pages/RobotDuplicatePage.tsx +++ b/src/components/robot/pages/RobotDuplicatePage.tsx @@ -15,6 +15,7 @@ export const RobotDuplicatePage = ({ handleStart }: RobotDuplicatePageProps) => const navigate = useNavigate(); const location = useLocation(); const [targetUrl, setTargetUrl] = useState(""); + const [newName, setNewName] = useState(""); const [robot, setRobot] = useState(null); const [isLoading, setIsLoading] = useState(false); const { recordingId, notify, setRerenderRobots } = useGlobalInfoStore(); @@ -34,7 +35,11 @@ export const RobotDuplicatePage = ({ handleStart }: RobotDuplicatePageProps) => url = lastPair?.what?.find((action: any) => action.action === "goto")?.args?.[0]; } - if (url) setTargetUrl(url); + if (url) { + setTargetUrl(url); + const lastWord = url.split('/').filter(Boolean).pop() || 'Unnamed'; + setNewName(`${robot.recording_meta.name} (${lastWord})`); + } } }, [robot]); @@ -57,9 +62,14 @@ export const RobotDuplicatePage = ({ handleStart }: RobotDuplicatePageProps) => return; } + if (!newName.trim()) { + notify("error", t("robot_duplication.notifications.name_required")); + return; + } + setIsLoading(true); try { - const result = await duplicateRecording(robot.recording_meta.id, targetUrl); + const result = await duplicateRecording(robot.recording_meta.id, targetUrl, newName.trim()); if (result) { setRerenderRobots(true); @@ -69,8 +79,8 @@ export const RobotDuplicatePage = ({ handleStart }: RobotDuplicatePageProps) => } else { notify("error", t("robot_duplication.notifications.duplicate_error")); } - } catch (error) { - notify("error", t("robot_duplication.notifications.unknown_error")); + } catch (error: any) { + notify("error", error.message || t("robot_duplication.notifications.unknown_error")); console.error("Error duplicating robot:", error); } finally { setIsLoading(false); @@ -102,11 +112,25 @@ export const RobotDuplicatePage = ({ handleStart }: RobotDuplicatePageProps) => {t("robot_duplication.descriptions.warning")} + setNewName(e.target.value)} + style={{ marginBottom: "20px", marginTop: "30px" }} + fullWidth + /> setTargetUrl(e.target.value)} - style={{ marginBottom: "20px", marginTop: "30px" }} + onChange={(e) => { + setTargetUrl(e.target.value); + if (robot) { + const lastWord = e.target.value.split('/').filter(Boolean).pop() || 'Unnamed'; + setNewName(`${robot.recording_meta.name} (${lastWord})`); + } + }} + style={{ marginBottom: "20px" }} + fullWidth /> )} diff --git a/src/components/robot/pages/RobotEditPage.tsx b/src/components/robot/pages/RobotEditPage.tsx index 0f2343294..f3e37c626 100644 --- a/src/components/robot/pages/RobotEditPage.tsx +++ b/src/components/robot/pages/RobotEditPage.tsx @@ -21,6 +21,12 @@ import { getStoredRecording, updateRecording } from "../../../api/storage"; import { WhereWhatPair } from "maxun-core"; import { RobotConfigPage } from "./RobotConfigPage"; import { useNavigate, useLocation } from "react-router-dom"; +import { + DEFAULT_OUTPUT_FORMATS, + OUTPUT_FORMAT_LABELS, + OUTPUT_FORMAT_OPTIONS, + OutputFormat, +} from "../../../constants/outputFormats"; interface RobotMeta { name: string; @@ -32,7 +38,7 @@ interface RobotMeta { params: any[]; type?: 'extract' | 'scrape' | 'crawl' | 'search'; url?: string; - formats?: ('markdown' | 'html' | 'screenshot-visible' | 'screenshot-fullpage')[]; + formats?: OutputFormat[]; isLLM?: boolean; } @@ -142,6 +148,8 @@ export const RobotEditPage = ({ handleStart }: RobotSettingsProps) => { const [isLoading, setIsLoading] = useState(false); const [crawlConfig, setCrawlConfig] = useState({}); const [searchConfig, setSearchConfig] = useState({}); + const [crawlOutputFormats, setCrawlOutputFormats] = useState(DEFAULT_OUTPUT_FORMATS); + const [searchOutputFormats, setSearchOutputFormats] = useState(DEFAULT_OUTPUT_FORMATS); const [showCrawlAdvanced, setShowCrawlAdvanced] = useState(false); const isEmailPattern = (value: string): boolean => { @@ -193,6 +201,38 @@ export const RobotEditPage = ({ handleStart }: RobotSettingsProps) => { findScrapeListLimits(robot.recording.workflow); extractCrawlConfig(robot.recording.workflow); extractSearchConfig(robot.recording.workflow); + + const rawFormats = Array.isArray(robot.recording_meta?.formats) + ? robot.recording_meta.formats + : []; + + const filteredFormats = rawFormats.filter((format): format is OutputFormat => + OUTPUT_FORMAT_OPTIONS.includes(format as OutputFormat) + ); + + if (robot.recording_meta?.type === 'crawl') { + setCrawlOutputFormats( + filteredFormats.length > 0 ? filteredFormats : DEFAULT_OUTPUT_FORMATS + ); + } + + if (robot.recording_meta?.type === 'search') { + const isDiscoverMode = robot.recording?.workflow?.some((pair: any) => + (pair.what || []).some( + (action: any) => + action.action === 'search' && + action.args?.[0]?.mode === 'discover' + ) + ); + + if (isDiscoverMode) { + setSearchOutputFormats(filteredFormats); + } else { + setSearchOutputFormats( + filteredFormats.length > 0 ? filteredFormats : DEFAULT_OUTPUT_FORMATS + ); + } + } } }, [robot]); @@ -783,6 +823,31 @@ export const RobotEditPage = ({ handleStart }: RobotSettingsProps) => { return ( <> + + Output Formats * + + + { const renderSearchConfigFields = () => { if (robot?.recording_meta.type !== 'search') return null; + const currentSearchMode = searchConfig.mode || 'discover'; + return ( <> { + {currentSearchMode === 'scrape' ? ( + + Output Formats * + + + ) : ( + + Output formats are only available in "Extract Data from Results" mode + + )} + Time Range