From c3296cd322905d3f98ec524b658dfa0515912533 Mon Sep 17 00:00:00 2001 From: Steven Enamakel <31011319+senamakel@users.noreply.github.com> Date: Sun, 28 Jun 2026 16:22:32 -0700 Subject: [PATCH] feat(media-generation): image & video agents + GMI media tools (#4244) --- gitbooks/features/native-tools/image-tools.md | 35 ++ src/openhuman/about_app/catalog_data.rs | 28 ++ .../agents/image_agent/agent.toml | 29 ++ .../agent_registry/agents/image_agent/mod.rs | 1 + .../agents/image_agent/prompt.md | 38 ++ .../agents/image_agent/prompt.rs | 71 +++ src/openhuman/agent_registry/agents/loader.rs | 10 + src/openhuman/agent_registry/agents/mod.rs | 2 + .../agents/orchestrator/agent.toml | 8 + .../agents/video_agent/agent.toml | 29 ++ .../agent_registry/agents/video_agent/mod.rs | 1 + .../agents/video_agent/prompt.md | 38 ++ .../agents/video_agent/prompt.rs | 71 +++ src/openhuman/mcp_server/resources.rs | 12 + src/openhuman/media_generation/download.rs | 176 ++++++++ src/openhuman/media_generation/mod.rs | 15 + src/openhuman/media_generation/tools.rs | 417 ++++++++++++++++++ src/openhuman/media_generation/tools_tests.rs | 249 +++++++++++ src/openhuman/media_generation/types.rs | 47 ++ src/openhuman/mod.rs | 1 + src/openhuman/tools/ops.rs | 7 + 21 files changed, 1285 insertions(+) create mode 100644 src/openhuman/agent_registry/agents/image_agent/agent.toml create mode 100644 src/openhuman/agent_registry/agents/image_agent/mod.rs create mode 100644 src/openhuman/agent_registry/agents/image_agent/prompt.md create mode 100644 src/openhuman/agent_registry/agents/image_agent/prompt.rs create mode 100644 src/openhuman/agent_registry/agents/video_agent/agent.toml create mode 100644 src/openhuman/agent_registry/agents/video_agent/mod.rs create mode 100644 src/openhuman/agent_registry/agents/video_agent/prompt.md create mode 100644 src/openhuman/agent_registry/agents/video_agent/prompt.rs create mode 100644 src/openhuman/media_generation/download.rs create mode 100644 src/openhuman/media_generation/mod.rs create mode 100644 src/openhuman/media_generation/tools.rs create mode 100644 src/openhuman/media_generation/tools_tests.rs create mode 100644 src/openhuman/media_generation/types.rs diff --git a/gitbooks/features/native-tools/image-tools.md b/gitbooks/features/native-tools/image-tools.md index fafc4bd3a..17a50dc18 100644 --- a/gitbooks/features/native-tools/image-tools.md +++ b/gitbooks/features/native-tools/image-tools.md @@ -79,3 +79,38 @@ The module has focused Rust tests for: Future runtime PRs should add provider-specific execution tests next to the runtime adapter, not in the hosted contract module. + +## Media generation (GMI) — image & video tools + +Separate from the high-level `image_generation` contract above, the +`src/openhuman/media_generation/` domain ships **wired, executing** tools that +generate images and video through the OpenHuman backend's `media_generation` +provider (GMI Cloud — Seedream, SeedEdit, Seedance, Veo). + +| Tool | Purpose | Permission | Output | +| --- | --- | --- | --- | +| `media_generate_image` | Text-to-image / image-to-image via GMI. | Execute | Local file path under `generated-media/`. | +| `media_generate_video` | Text-to-video / image-to-video via GMI. | Execute | Local file path under `generated-media/`. | +| `media_list_models` | List the curated model catalog (and optionally GMI's live list). | Read-only | Model ids + pricing. | + +How it works: + +- Generation is asynchronous. The tool submits to the backend (which charges on + submit and returns a request id), then **blocks with progress**, polling until + the request reaches a terminal state. +- GMI returns expiring signed URLs; the tool downloads each artifact into the + agent's `generated-media/` directory and returns a stable local file path. +- The backend owns provider keys, billing, and rate limiting + (`/agent-integrations/media-generation/*`, see `backend/docs/media-generation.md`). + +### Image & video sub-agents + +Two specialist sub-agents wrap these tools and are reachable from the +orchestrator via delegation: + +- **`image_agent`** (`delegate_create_image`) — owns prompt craft, model + selection, and saving generated images. Rides the multimodal `vision-v1` tier + so it can inspect what it produces. +- **`video_agent`** (`delegate_create_video`) — owns text-to-video and + image-to-video; sets expectations that generation can take minutes and blocks + until the clip is saved. diff --git a/src/openhuman/about_app/catalog_data.rs b/src/openhuman/about_app/catalog_data.rs index 9f674d9ff..2a4b8ea27 100644 --- a/src/openhuman/about_app/catalog_data.rs +++ b/src/openhuman/about_app/catalog_data.rs @@ -22,6 +22,14 @@ const IMAGE_TO_BACKEND: Option = Some(CapabilityPrivacy { destinations: &["OpenHuman backend", "TinyHumans Neocortex"], }); +// Media generation sends the prompt (and any reference image URL) to GMI Cloud +// via the OpenHuman backend; generated media is downloaded back to the device. +const MEDIA_GEN_TO_BACKEND: Option = Some(CapabilityPrivacy { + leaves_device: true, + data_kind: PrivacyDataKind::Raw, + destinations: &["OpenHuman backend", "GMI Cloud"], +}); + const LOCAL_CREDENTIALS: Option = Some(CapabilityPrivacy { leaves_device: false, data_kind: PrivacyDataKind::Credentials, @@ -261,6 +269,26 @@ pub(super) const CAPABILITIES: &[Capability] = &[ status: CapabilityStatus::Beta, privacy: IMAGE_TO_BACKEND, }, + Capability { + id: "intelligence.image_generation", + name: "Image Generation", + domain: "agent", + category: CapabilityCategory::Intelligence, + description: "Delegate image creation to a dedicated image sub-agent — generate images from a text prompt, or edit/restyle reference images, using hosted GMI models (Seedream / SeedEdit). Results are saved to the workspace.", + how_to: "Ask the assistant to generate, draw, or edit an image", + status: CapabilityStatus::Beta, + privacy: MEDIA_GEN_TO_BACKEND, + }, + Capability { + id: "intelligence.video_generation", + name: "Video Generation", + domain: "agent", + category: CapabilityCategory::Intelligence, + description: "Delegate short-video creation to a dedicated video sub-agent — text-to-video or animate a reference image using hosted GMI models (Seedance / Veo). Generation is asynchronous; the finished clip is saved to the workspace.", + how_to: "Ask the assistant to generate a video or animate an image", + status: CapabilityStatus::Beta, + privacy: MEDIA_GEN_TO_BACKEND, + }, Capability { id: "conversation.label_filter", name: "Thread Label Filters", diff --git a/src/openhuman/agent_registry/agents/image_agent/agent.toml b/src/openhuman/agent_registry/agents/image_agent/agent.toml new file mode 100644 index 000000000..64e12bdb8 --- /dev/null +++ b/src/openhuman/agent_registry/agents/image_agent/agent.toml @@ -0,0 +1,29 @@ +id = "image_agent" +display_name = "Image Creator" +delegate_name = "create_image" +when_to_use = "Image-generation specialist — create new images from a text prompt, or edit/restyle reference images, using the hosted GMI models (Seedream / SeedEdit). Route here for make/generate/draw an image, create a logo/illustration/photo, or edit this image requests. It rides the multimodal `vision-v1` tier, so it can also inspect the images it produces. It saves each result to the workspace and returns the local file path." +temperature = 0.6 +max_iterations = 8 +agent_tier = "worker" +omit_identity = true +omit_memory_context = true +omit_safety_preamble = false +omit_skills_catalog = true +omit_profile = true +omit_memory_md = true + +# Multimodal tier so the agent can review the images it generates (and any +# reference images) via the image_info / inline-image path, then iterate. +[model] +hint = "vision" + +[tools] +# media_generate_image submits the generation and returns a saved local path; +# media_list_models surfaces the catalog so the agent can pick a model; the +# remaining tools let it inspect reference images and read prior outputs. +named = [ + "media_generate_image", + "media_list_models", + "image_info", + "file_read", +] diff --git a/src/openhuman/agent_registry/agents/image_agent/mod.rs b/src/openhuman/agent_registry/agents/image_agent/mod.rs new file mode 100644 index 000000000..8bf84783c --- /dev/null +++ b/src/openhuman/agent_registry/agents/image_agent/mod.rs @@ -0,0 +1 @@ +pub mod prompt; diff --git a/src/openhuman/agent_registry/agents/image_agent/prompt.md b/src/openhuman/agent_registry/agents/image_agent/prompt.md new file mode 100644 index 000000000..5036393a8 --- /dev/null +++ b/src/openhuman/agent_registry/agents/image_agent/prompt.md @@ -0,0 +1,38 @@ +# Image-generation specialist + +You are a focused **image-creation** sub-agent. You turn a delegating agent's +request into one or more finished image files using the hosted GMI image models +(Seedream for text-to-image, SeedEdit for edits). You run on a multimodal model, +so you can look at reference images and at the images you generate. + +## Your job + +- **Create** images from a text prompt (`media_generate_image`). +- **Edit / restyle** a supplied image by passing its URL(s) as `input_images`. +- **Pick the right model** when it matters — call `media_list_models` to see the + catalog (defaults are fine for most requests; `include_upstream` exposes the + full GMI list). + +## How to work + +- Write a vivid, specific prompt. Translate a terse request into concrete visual + detail — subject, composition, lighting, style, mood, colour — but stay true + to what was asked. Don't invent requirements the user didn't state. +- Default the model and size unless the task calls for something specific. Use a + `size` like `1024x1024` (square), `1536x1024` (landscape), or `1024x1536` + (portrait) when the aspect ratio matters. +- For edits, pass the source image URL(s) in `input_images` and describe the + change precisely. +- Each generation **saves the image to the workspace and returns a local file + path**. Always report that path back so the deck/answer can reference the + concrete artifact. Do not paste raw base64 or invent URLs. +- Generation is billed. Don't loop on near-identical prompts — generate, inspect + the result, and only re-run if it materially misses the brief. + +## Boundaries + +- Report results to the delegating agent — you are not talking to the end user. +- If a request is unsafe or disallowed, decline rather than attempting a + work-around. +- If generation fails or times out, say so plainly and surface the request id; + don't fabricate a path or claim success. diff --git a/src/openhuman/agent_registry/agents/image_agent/prompt.rs b/src/openhuman/agent_registry/agents/image_agent/prompt.rs new file mode 100644 index 000000000..6a0ea97ad --- /dev/null +++ b/src/openhuman/agent_registry/agents/image_agent/prompt.rs @@ -0,0 +1,71 @@ +//! System prompt builder for the `image_agent` built-in agent. +//! +//! Returns the final, fully-assembled system prompt — archetype body (from the +//! sibling `prompt.md`) plus the shared section helpers every agent uses. + +use crate::openhuman::context::prompt::{ + render_tools, render_user_files, render_workspace, PromptContext, +}; +use anyhow::Result; + +const ARCHETYPE: &str = include_str!("prompt.md"); + +pub fn build(ctx: &PromptContext<'_>) -> Result { + let mut out = String::with_capacity(4096); + out.push_str(ARCHETYPE.trim_end()); + out.push_str("\n\n"); + + let user_files = render_user_files(ctx)?; + if !user_files.trim().is_empty() { + out.push_str(user_files.trim_end()); + out.push_str("\n\n"); + } + + let tools = render_tools(ctx)?; + if !tools.trim().is_empty() { + out.push_str(tools.trim_end()); + out.push_str("\n\n"); + } + + let workspace = render_workspace(ctx)?; + if !workspace.trim().is_empty() { + out.push_str(workspace.trim_end()); + out.push('\n'); + } + + Ok(out) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::openhuman::context::prompt::{LearnedContextData, ToolCallFormat}; + use std::collections::HashSet; + + #[test] + fn build_returns_nonempty_body() { + let visible: HashSet = HashSet::new(); + let ctx = PromptContext { + workspace_dir: std::path::Path::new("."), + model_name: "vision-v1", + agent_id: "image_agent", + tools: &[], + workflows: &[], + dispatcher_instructions: "", + learned: LearnedContextData::default(), + visible_tool_names: &visible, + tool_call_format: ToolCallFormat::PFormat, + connected_integrations: &[], + connected_identities_md: String::new(), + include_profile: false, + include_memory_md: false, + curated_snapshot: None, + user_identity: None, + personality_soul_md: None, + personality_memory_md: None, + personality_roster: vec![], + }; + let body = build(&ctx).unwrap(); + assert!(body.contains("Image-generation specialist")); + } +} diff --git a/src/openhuman/agent_registry/agents/loader.rs b/src/openhuman/agent_registry/agents/loader.rs index 713c9e17e..e16ed6bf6 100644 --- a/src/openhuman/agent_registry/agents/loader.rs +++ b/src/openhuman/agent_registry/agents/loader.rs @@ -169,6 +169,16 @@ pub const BUILTINS: &[BuiltinAgent] = &[ toml: include_str!("vision_agent/agent.toml"), prompt_fn: super::vision_agent::prompt::build, }, + BuiltinAgent { + id: "image_agent", + toml: include_str!("image_agent/agent.toml"), + prompt_fn: super::image_agent::prompt::build, + }, + BuiltinAgent { + id: "video_agent", + toml: include_str!("video_agent/agent.toml"), + prompt_fn: super::video_agent::prompt::build, + }, BuiltinAgent { id: "archivist", toml: include_str!("archivist/agent.toml"), diff --git a/src/openhuman/agent_registry/agents/mod.rs b/src/openhuman/agent_registry/agents/mod.rs index 40b9be3ed..aa30640cb 100644 --- a/src/openhuman/agent_registry/agents/mod.rs +++ b/src/openhuman/agent_registry/agents/mod.rs @@ -13,6 +13,7 @@ pub mod crypto_agent; pub mod desktop_control_agent; pub mod goals_agent; pub mod help; +pub mod image_agent; pub mod integrations_agent; pub mod markets_agent; pub mod mcp_agent; @@ -33,6 +34,7 @@ pub mod tool_maker; pub mod tools_agent; pub mod trigger_reactor; pub mod trigger_triage; +pub mod video_agent; pub mod vision_agent; pub use loader::{load_builtins, validate_tier_hierarchy, BuiltinAgent, BUILTINS}; diff --git a/src/openhuman/agent_registry/agents/orchestrator/agent.toml b/src/openhuman/agent_registry/agents/orchestrator/agent.toml index 3e21c7379..df7abd01c 100644 --- a/src/openhuman/agent_registry/agents/orchestrator/agent.toml +++ b/src/openhuman/agent_registry/agents/orchestrator/agent.toml @@ -85,6 +85,14 @@ allowlist = [ # of an attached image / screenshot / on-disk image file here — it rides the # multimodal `vision-v1` tier, so it can actually see the image. "vision_agent", + # Image-generation specialist. Synthesised into a `delegate_create_image` + # tool. Route make/generate/edit an image requests here — it owns prompt + # craft, model selection, and saving the result to the workspace. + "image_agent", + # Video-generation specialist. Synthesised into a `delegate_create_video` + # tool. Route make/generate a video or animate this image requests here — + # generation is async (minutes) and the agent blocks until the clip is saved. + "video_agent", "skill_creator", "critic", "archivist", diff --git a/src/openhuman/agent_registry/agents/video_agent/agent.toml b/src/openhuman/agent_registry/agents/video_agent/agent.toml new file mode 100644 index 000000000..54c92d06f --- /dev/null +++ b/src/openhuman/agent_registry/agents/video_agent/agent.toml @@ -0,0 +1,29 @@ +id = "video_agent" +display_name = "Video Creator" +delegate_name = "create_video" +when_to_use = "Video-generation specialist — create short video clips from a text prompt (text-to-video) or animate a supplied image (image-to-video) using the hosted GMI models (Seedance / Veo). Route here for make/generate a video, animate this image, or short clip requests. Generation can take a few minutes; it blocks until the clip is ready, saves it to the workspace, and returns the local file path." +temperature = 0.6 +max_iterations = 8 +agent_tier = "worker" +omit_identity = true +omit_memory_context = true +omit_safety_preamble = false +omit_skills_catalog = true +omit_profile = true +omit_memory_md = true + +# Multimodal tier so the agent can inspect a reference/first-frame image or the +# returned thumbnail when shaping an image-to-video request. +[model] +hint = "vision" + +[tools] +# media_generate_video submits the generation and returns a saved local path; +# media_list_models surfaces the catalog; image_info / file_read let it inspect +# a reference image before animating it. +named = [ + "media_generate_video", + "media_list_models", + "image_info", + "file_read", +] diff --git a/src/openhuman/agent_registry/agents/video_agent/mod.rs b/src/openhuman/agent_registry/agents/video_agent/mod.rs new file mode 100644 index 000000000..8bf84783c --- /dev/null +++ b/src/openhuman/agent_registry/agents/video_agent/mod.rs @@ -0,0 +1 @@ +pub mod prompt; diff --git a/src/openhuman/agent_registry/agents/video_agent/prompt.md b/src/openhuman/agent_registry/agents/video_agent/prompt.md new file mode 100644 index 000000000..97d6a7c76 --- /dev/null +++ b/src/openhuman/agent_registry/agents/video_agent/prompt.md @@ -0,0 +1,38 @@ +# Video-generation specialist + +You are a focused **video-creation** sub-agent. You turn a delegating agent's +request into a finished video clip using the hosted GMI video models (Seedance +for fast clips, Veo for premium-tier output). You can do text-to-video or +animate a supplied first-frame/reference image (image-to-video). + +## Your job + +- **Create** a clip from a text prompt (`media_generate_video`). +- **Animate** a supplied image by passing its URL as `input_image`. +- **Pick the right model** when it matters — call `media_list_models` to see the + catalog (the fast Seedance default suits most requests; `include_upstream` + exposes the full GMI list, including premium tiers). + +## How to work + +- Write a concrete prompt describing the motion, subject, and scene — what + happens over the clip, not just a static description. Mention camera movement, + pacing, and style when relevant. +- Use `duration_seconds` and `aspect_ratio` (e.g. `16:9`, `9:16`, `1:1`) when the + task specifies them; otherwise let the model default. +- For image-to-video, pass the source image URL in `input_image` and describe + the motion you want applied to it. +- Generation is **asynchronous and can take minutes** — the tool blocks until the + clip is ready, saves it to the workspace, and returns a local file path. Report + that path back. Set expectations: tell the delegating agent it may take a + little while. +- Generation is billed and slow. Don't re-run on near-identical prompts — only + iterate if the result materially misses the brief. + +## Boundaries + +- Report results to the delegating agent — you are not talking to the end user. +- If a request is unsafe or disallowed, decline rather than attempting a + work-around. +- If generation fails or times out, say so plainly and surface the request id; + don't fabricate a path or claim success. diff --git a/src/openhuman/agent_registry/agents/video_agent/prompt.rs b/src/openhuman/agent_registry/agents/video_agent/prompt.rs new file mode 100644 index 000000000..38708ec7b --- /dev/null +++ b/src/openhuman/agent_registry/agents/video_agent/prompt.rs @@ -0,0 +1,71 @@ +//! System prompt builder for the `video_agent` built-in agent. +//! +//! Returns the final, fully-assembled system prompt — archetype body (from the +//! sibling `prompt.md`) plus the shared section helpers every agent uses. + +use crate::openhuman::context::prompt::{ + render_tools, render_user_files, render_workspace, PromptContext, +}; +use anyhow::Result; + +const ARCHETYPE: &str = include_str!("prompt.md"); + +pub fn build(ctx: &PromptContext<'_>) -> Result { + let mut out = String::with_capacity(4096); + out.push_str(ARCHETYPE.trim_end()); + out.push_str("\n\n"); + + let user_files = render_user_files(ctx)?; + if !user_files.trim().is_empty() { + out.push_str(user_files.trim_end()); + out.push_str("\n\n"); + } + + let tools = render_tools(ctx)?; + if !tools.trim().is_empty() { + out.push_str(tools.trim_end()); + out.push_str("\n\n"); + } + + let workspace = render_workspace(ctx)?; + if !workspace.trim().is_empty() { + out.push_str(workspace.trim_end()); + out.push('\n'); + } + + Ok(out) +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::openhuman::context::prompt::{LearnedContextData, ToolCallFormat}; + use std::collections::HashSet; + + #[test] + fn build_returns_nonempty_body() { + let visible: HashSet = HashSet::new(); + let ctx = PromptContext { + workspace_dir: std::path::Path::new("."), + model_name: "vision-v1", + agent_id: "video_agent", + tools: &[], + workflows: &[], + dispatcher_instructions: "", + learned: LearnedContextData::default(), + visible_tool_names: &visible, + tool_call_format: ToolCallFormat::PFormat, + connected_integrations: &[], + connected_identities_md: String::new(), + include_profile: false, + include_memory_md: false, + curated_snapshot: None, + user_identity: None, + personality_soul_md: None, + personality_memory_md: None, + personality_roster: vec![], + }; + let body = build(&ctx).unwrap(); + assert!(body.contains("Video-generation specialist")); + } +} diff --git a/src/openhuman/mcp_server/resources.rs b/src/openhuman/mcp_server/resources.rs index 49ef248b0..2a08f744e 100644 --- a/src/openhuman/mcp_server/resources.rs +++ b/src/openhuman/mcp_server/resources.rs @@ -133,6 +133,18 @@ const RESOURCE_CATALOG: &[PromptResource] = &[ description: "Multimodal worker that analyses attached images for the vision tier.", content: include_str!("../agent_registry/agents/vision_agent/prompt.md"), }, + PromptResource { + uri: "openhuman://prompts/agents/image_agent", + name: "image_agent", + description: "Worker that generates or edits images via GMI and saves them to the workspace.", + content: include_str!("../agent_registry/agents/image_agent/prompt.md"), + }, + PromptResource { + uri: "openhuman://prompts/agents/video_agent", + name: "video_agent", + description: "Worker that generates short videos via GMI and saves them to the workspace.", + content: include_str!("../agent_registry/agents/video_agent/prompt.md"), + }, PromptResource { uri: "openhuman://prompts/agents/archivist", name: "archivist", diff --git a/src/openhuman/media_generation/download.rs b/src/openhuman/media_generation/download.rs new file mode 100644 index 000000000..abd3d2a5c --- /dev/null +++ b/src/openhuman/media_generation/download.rs @@ -0,0 +1,176 @@ +//! Persist generated media to the agent's action directory. +//! +//! GMI returns expiring signed URLs; we download the bytes and write them under +//! a `generated-media/` root inside `action_dir` so final answers can reference +//! a stable local file path (per the `image_generation` contract). The action +//! directory is the agent's canonical read/write root. + +use std::path::{Path, PathBuf}; + +use anyhow::{Context, Result}; + +use super::types::MediaItem; + +/// Subdirectory (under `action_dir`) where generated artifacts are stored. +const GENERATED_MEDIA_DIR: &str = "generated-media"; + +/// A downloaded artifact and where it landed on disk. +#[derive(Debug, Clone)] +pub struct PersistedArtifact { + pub kind: String, + pub path: PathBuf, + pub source_url: String, + pub thumbnail_url: Option, +} + +/// Pick a file extension from the artifact kind + content type / URL. +fn extension_for(kind: &str, content_type: Option<&str>, url: &str) -> String { + if let Some(ct) = content_type { + let ct = ct.to_ascii_lowercase(); + if ct.contains("png") { + return "png".to_string(); + } + if ct.contains("webp") { + return "webp".to_string(); + } + if ct.contains("jpeg") || ct.contains("jpg") { + return "jpg".to_string(); + } + if ct.contains("mp4") { + return "mp4".to_string(); + } + if ct.contains("webm") { + return "webm".to_string(); + } + } + // Fall back to the URL path suffix, then a per-kind default. + let lower = url.split('?').next().unwrap_or(url).to_ascii_lowercase(); + for ext in ["png", "webp", "jpg", "jpeg", "mp4", "webm"] { + if lower.ends_with(&format!(".{ext}")) { + return if ext == "jpeg" { + "jpg".to_string() + } else { + ext.to_string() + }; + } + } + if kind.eq_ignore_ascii_case("video") { + "mp4".to_string() + } else { + "png".to_string() + } +} + +/// Download a single media URL into `dir`, returning the written path. +async fn download_one( + http: &reqwest::Client, + dir: &Path, + item: &MediaItem, + request_id: &str, + index: usize, +) -> Result { + tracing::info!( + "[media_generation] downloading {} artifact {} for request={}", + item.kind, + index, + request_id + ); + let resp = http + .get(&item.url) + .send() + .await + .with_context(|| format!("failed to fetch generated media from {}", item.url))? + .error_for_status() + .with_context(|| format!("generated media URL returned an error: {}", item.url))?; + + let content_type = resp + .headers() + .get(reqwest::header::CONTENT_TYPE) + .and_then(|v| v.to_str().ok()) + .map(|s| s.to_string()); + let ext = extension_for(&item.kind, content_type.as_deref(), &item.url); + + let bytes = resp + .bytes() + .await + .with_context(|| format!("failed to read generated media body from {}", item.url))?; + + // Sanitize the request id for use in a filename (it is a UUID from GMI, but + // be defensive against path separators). + let safe_id: String = request_id + .chars() + .map(|c| { + if c.is_ascii_alphanumeric() || c == '-' { + c + } else { + '_' + } + }) + .collect(); + let filename = format!("{safe_id}-{index}.{ext}"); + let path = dir.join(&filename); + tokio::fs::write(&path, &bytes) + .await + .with_context(|| format!("failed to write generated media to {}", path.display()))?; + + Ok(PersistedArtifact { + kind: item.kind.clone(), + path, + source_url: item.url.clone(), + thumbnail_url: item.thumbnail_url.clone(), + }) +} + +/// Download + persist all media items for a request under +/// `{action_dir}/generated-media/`. Returns the written artifacts. +pub async fn persist_media( + action_dir: &Path, + request_id: &str, + items: &[MediaItem], +) -> Result> { + if items.is_empty() { + return Ok(Vec::new()); + } + let dir = action_dir.join(GENERATED_MEDIA_DIR); + tokio::fs::create_dir_all(&dir) + .await + .with_context(|| format!("failed to create generated-media dir {}", dir.display()))?; + + let http = reqwest::Client::new(); + let mut out = Vec::with_capacity(items.len()); + for (i, item) in items.iter().enumerate() { + out.push(download_one(&http, &dir, item, request_id, i).await?); + } + Ok(out) +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn extension_prefers_content_type() { + assert_eq!( + extension_for("image", Some("image/png"), "https://x/y"), + "png" + ); + assert_eq!( + extension_for("image", Some("image/webp"), "https://x/y"), + "webp" + ); + assert_eq!( + extension_for("video", Some("video/mp4"), "https://x/y"), + "mp4" + ); + } + + #[test] + fn extension_falls_back_to_url_then_kind() { + assert_eq!( + extension_for("image", None, "https://x/y/a.webp?sig=1"), + "webp" + ); + assert_eq!(extension_for("video", None, "https://x/y/clip"), "mp4"); + assert_eq!(extension_for("image", None, "https://x/y/clip"), "png"); + } +} diff --git a/src/openhuman/media_generation/mod.rs b/src/openhuman/media_generation/mod.rs new file mode 100644 index 000000000..2b0d699c3 --- /dev/null +++ b/src/openhuman/media_generation/mod.rs @@ -0,0 +1,15 @@ +//! Media generation domain — agent tools for image/video generation backed by +//! GMI via the OpenHuman backend's `media_generation` provider. +//! +//! The backend (`/agent-integrations/media-generation/*`) owns provider keys, +//! billing, and the standardized contract; these tools submit a request, block +//! with progress until it completes, download the resulting media into the +//! agent's `generated-media/` root, and return local file paths. + +pub mod download; +pub mod tools; +pub mod types; + +pub use tools::{ + build_media_tools, MediaGenerateImageTool, MediaGenerateVideoTool, MediaListModelsTool, +}; diff --git a/src/openhuman/media_generation/tools.rs b/src/openhuman/media_generation/tools.rs new file mode 100644 index 000000000..4c0055bd8 --- /dev/null +++ b/src/openhuman/media_generation/tools.rs @@ -0,0 +1,417 @@ +//! Agent-facing media-generation tools (image + video) backed by GMI via the +//! OpenHuman backend's `media_generation` provider. +//! +//! **Endpoints** (see `backend/docs/media-generation.md`): +//! - `POST /agent-integrations/media-generation/images` +//! - `POST /agent-integrations/media-generation/videos` +//! - `GET /agent-integrations/media-generation/requests/{requestId}` +//! - `GET /agent-integrations/media-generation/models` +//! +//! Generation is asynchronous. These tools **block with progress**: they submit +//! (`wait:false`, so the backend charges + returns a request id immediately), +//! then poll the request until it reaches a terminal state, download each +//! resulting artifact into the agent's `generated-media/` root, and return the +//! local file paths. The backend owns GMI keys, billing, and rate limiting. + +use std::path::PathBuf; +use std::sync::Arc; +use std::time::{Duration, Instant}; + +use async_trait::async_trait; +use serde_json::{json, Value}; + +use crate::openhuman::config::Config; +use crate::openhuman::integrations::IntegrationClient; +use crate::openhuman::tools::traits::{PermissionLevel, Tool, ToolCategory, ToolResult}; + +use super::download::persist_media; +use super::types::MediaResponse; + +const IMAGES_PATH: &str = "/agent-integrations/media-generation/images"; +const VIDEOS_PATH: &str = "/agent-integrations/media-generation/videos"; +const MODELS_PATH: &str = "/agent-integrations/media-generation/models"; + +/// Poll cadence + caps. Images are fast; video can take minutes. +const POLL_INTERVAL: Duration = Duration::from_secs(4); +const IMAGE_MAX_WAIT_SECS: u64 = 180; +const VIDEO_MAX_WAIT_SECS: u64 = 420; + +/// Shared submit-then-poll-then-persist flow for both modalities. +async fn generate_and_persist( + client: &IntegrationClient, + action_dir: &std::path::Path, + submit_path: &str, + body: Value, + max_wait_secs: u64, +) -> ToolResult { + // Submit without server-side blocking; the backend charges on submit and + // returns a request id we poll ourselves (so the core owns the progress UX). + let submitted: MediaResponse = match client.post::(submit_path, &body).await { + Ok(resp) => resp, + Err(e) => return ToolResult::error(format!("Media generation submit failed: {e}")), + }; + + let request_id = submitted.request_id.clone(); + tracing::info!( + "[media_generation] submitted request={} status={} cost=${:.4}", + request_id, + submitted.status, + submitted.cost_usd + ); + + let status_path = format!( + "/agent-integrations/media-generation/requests/{}", + request_id + ); + + let mut latest = submitted; + let deadline = Instant::now() + Duration::from_secs(max_wait_secs); + while !latest.is_terminal() { + if Instant::now() >= deadline { + tracing::warn!( + "[media_generation] wait budget elapsed for request={} (status={})", + request_id, + latest.status + ); + return ToolResult::success(format!( + "Media generation is still {} after {}s. It was accepted (request_id: {}) and \ + billed; it may finish shortly — check again later.", + latest.status, max_wait_secs, request_id + )); + } + tokio::time::sleep(POLL_INTERVAL).await; + match client.get::(&status_path).await { + Ok(resp) => { + tracing::debug!( + "[media_generation] poll request={} status={}", + request_id, + resp.status + ); + latest = resp; + } + Err(e) => { + tracing::warn!( + "[media_generation] poll error for request={}: {e}", + request_id + ); + // Transient poll failures shouldn't abort a paid generation — + // keep polling until the deadline. + } + } + } + + if latest.is_failed() { + return ToolResult::error(format!( + "Media generation failed (request_id: {request_id})." + )); + } + + if latest.media.is_empty() { + return ToolResult::error(format!( + "Media generation reported success but returned no media (request_id: {request_id})." + )); + } + + match persist_media(action_dir, &request_id, &latest.media).await { + Ok(artifacts) => { + let mut lines = vec![format!( + "Generated {} artifact(s) (request_id: {}, model: {}):", + artifacts.len(), + request_id, + latest.model + )]; + for art in &artifacts { + lines.push(format!("- {} → {}", art.kind, art.path.display())); + if let Some(thumb) = &art.thumbnail_url { + lines.push(format!(" thumbnail: {thumb}")); + } + } + lines.push(format!("\nCost: ${:.4}", latest.cost_usd)); + let payload = json!({ + "request_id": request_id, + "model": latest.model, + "cost_usd": latest.cost_usd, + "artifacts": artifacts.iter().map(|a| json!({ + "type": a.kind, + "path": a.path.display().to_string(), + "source_url": a.source_url, + "thumbnail_url": a.thumbnail_url, + })).collect::>(), + }); + ToolResult::success_with_markdown(payload, lines.join("\n")) + } + Err(e) => ToolResult::error(format!( + "Generation succeeded but persisting media failed (request_id: {request_id}): {e}" + )), + } +} + +// ── MediaGenerateImageTool ────────────────────────────────────────── + +pub struct MediaGenerateImageTool { + client: Arc, + action_dir: PathBuf, +} + +impl MediaGenerateImageTool { + pub fn new(client: Arc, action_dir: PathBuf) -> Self { + Self { client, action_dir } + } +} + +#[async_trait] +impl Tool for MediaGenerateImageTool { + fn name(&self) -> &str { + "media_generate_image" + } + + fn description(&self) -> &str { + "Generate or edit an image from a text prompt using GMI (Seedream / SeedEdit). \ + Optionally pass reference image URLs to edit/condition (image-to-image). \ + Blocks until the image is ready and saves it under the workspace \ + generated-media folder, returning the local file path. Cost is billed by the backend." + } + + fn parameters_schema(&self) -> Value { + json!({ + "type": "object", + "properties": { + "prompt": { "type": "string", "description": "Detailed visual prompt or edit instruction" }, + "model": { "type": "string", "description": "Optional GMI model id (default: seedream-4-0-250828). Use media_list_models to discover." }, + "size": { "type": "string", "description": "Optional output size, e.g. 1024x1024 or 1536x1024" }, + "n": { "type": "integer", "minimum": 1, "maximum": 8, "description": "Number of images (default 1)" }, + "input_images": { + "type": "array", + "items": { "type": "string" }, + "description": "Optional reference image URLs for edit / image-to-image" + }, + "seed": { "type": "integer", "description": "Optional seed for reproducibility" } + }, + "required": ["prompt"] + }) + } + + fn permission_level(&self) -> PermissionLevel { + PermissionLevel::Execute + } + + fn category(&self) -> ToolCategory { + ToolCategory::Workflow + } + + async fn execute(&self, args: Value) -> anyhow::Result { + let prompt = match args.get("prompt").and_then(|v| v.as_str()) { + Some(p) if !p.trim().is_empty() => p, + _ => return Ok(ToolResult::error("prompt is required")), + }; + + let mut body = json!({ "prompt": prompt, "wait": false }); + if let Some(model) = args.get("model").and_then(|v| v.as_str()) { + body["model"] = json!(model); + } + if let Some(size) = args.get("size").and_then(|v| v.as_str()) { + body["size"] = json!(size); + } + if let Some(n) = args.get("n").and_then(|v| v.as_u64()) { + body["n"] = json!(n.clamp(1, 8)); + } + if let Some(imgs) = args.get("input_images").and_then(|v| v.as_array()) { + let urls: Vec<&str> = imgs.iter().filter_map(|v| v.as_str()).collect(); + if !urls.is_empty() { + body["inputImages"] = json!(urls); + } + } + if let Some(seed) = args.get("seed").and_then(|v| v.as_i64()) { + body["seed"] = json!(seed); + } + + tracing::info!("[media_generate_image] prompt_len={}", prompt.len()); + Ok(generate_and_persist( + &self.client, + &self.action_dir, + IMAGES_PATH, + body, + IMAGE_MAX_WAIT_SECS, + ) + .await) + } +} + +// ── MediaGenerateVideoTool ────────────────────────────────────────── + +pub struct MediaGenerateVideoTool { + client: Arc, + action_dir: PathBuf, +} + +impl MediaGenerateVideoTool { + pub fn new(client: Arc, action_dir: PathBuf) -> Self { + Self { client, action_dir } + } +} + +#[async_trait] +impl Tool for MediaGenerateVideoTool { + fn name(&self) -> &str { + "media_generate_video" + } + + fn description(&self) -> &str { + "Generate a short video from a text prompt using GMI (Seedance / Veo). \ + Optionally pass a first-frame/reference image URL for image-to-video. \ + Video can take a few minutes; this blocks until it is ready, saves the \ + clip under the workspace generated-media folder, and returns the local \ + file path. Cost is billed by the backend." + } + + fn parameters_schema(&self) -> Value { + json!({ + "type": "object", + "properties": { + "prompt": { "type": "string", "description": "Detailed description of the video to generate" }, + "model": { "type": "string", "description": "Optional GMI model id (default: seedance-1-0-pro-fast-251015). Use media_list_models to discover." }, + "input_image": { "type": "string", "description": "Optional first-frame / reference image URL for image-to-video" }, + "duration_seconds": { "type": "integer", "minimum": 1, "maximum": 60, "description": "Optional clip duration in seconds" }, + "aspect_ratio": { "type": "string", "description": "Optional aspect ratio, e.g. 16:9, 9:16, 1:1" }, + "negative_prompt": { "type": "string", "description": "Optional description of what to avoid" }, + "seed": { "type": "integer", "description": "Optional seed for reproducibility" } + }, + "required": ["prompt"] + }) + } + + fn permission_level(&self) -> PermissionLevel { + PermissionLevel::Execute + } + + fn category(&self) -> ToolCategory { + ToolCategory::Workflow + } + + async fn execute(&self, args: Value) -> anyhow::Result { + let prompt = match args.get("prompt").and_then(|v| v.as_str()) { + Some(p) if !p.trim().is_empty() => p, + _ => return Ok(ToolResult::error("prompt is required")), + }; + + let mut body = json!({ "prompt": prompt, "wait": false }); + if let Some(model) = args.get("model").and_then(|v| v.as_str()) { + body["model"] = json!(model); + } + if let Some(img) = args.get("input_image").and_then(|v| v.as_str()) { + body["inputImage"] = json!(img); + } + if let Some(d) = args.get("duration_seconds").and_then(|v| v.as_u64()) { + body["durationSeconds"] = json!(d.clamp(1, 60)); + } + if let Some(ar) = args.get("aspect_ratio").and_then(|v| v.as_str()) { + body["aspectRatio"] = json!(ar); + } + if let Some(np) = args.get("negative_prompt").and_then(|v| v.as_str()) { + body["negativePrompt"] = json!(np); + } + if let Some(seed) = args.get("seed").and_then(|v| v.as_i64()) { + body["seed"] = json!(seed); + } + + tracing::info!("[media_generate_video] prompt_len={}", prompt.len()); + Ok(generate_and_persist( + &self.client, + &self.action_dir, + VIDEOS_PATH, + body, + VIDEO_MAX_WAIT_SECS, + ) + .await) + } +} + +// ── MediaListModelsTool ───────────────────────────────────────────── + +pub struct MediaListModelsTool { + client: Arc, +} + +impl MediaListModelsTool { + pub fn new(client: Arc) -> Self { + Self { client } + } +} + +#[async_trait] +impl Tool for MediaListModelsTool { + fn name(&self) -> &str { + "media_list_models" + } + + fn description(&self) -> &str { + "List available image/video generation models — a curated catalog with \ + pricing, plus (with include_upstream) GMI's full live model list. Use to \ + pick a `model` id for media_generate_image / media_generate_video." + } + + fn parameters_schema(&self) -> Value { + json!({ + "type": "object", + "properties": { + "include_upstream": { + "type": "boolean", + "description": "Also fetch GMI's full live model list (default false)" + } + } + }) + } + + fn category(&self) -> ToolCategory { + ToolCategory::Workflow + } + + async fn execute(&self, args: Value) -> anyhow::Result { + let include_upstream = args + .get("include_upstream") + .and_then(|v| v.as_bool()) + .unwrap_or(false); + let path = if include_upstream { + format!("{MODELS_PATH}?includeUpstream=true") + } else { + MODELS_PATH.to_string() + }; + match self.client.get::(&path).await { + Ok(resp) => Ok(ToolResult::success_with_markdown( + resp.clone(), + serde_json::to_string_pretty(&resp).unwrap_or_else(|_| resp.to_string()), + )), + Err(e) => Ok(ToolResult::error(format!( + "Failed to list media models: {e}" + ))), + } + } +} + +// ── Builder ───────────────────────────────────────────────────────── + +/// Build the media-generation tool surface. Returns empty when no integration +/// client is configured (no backend URL / not signed in), mirroring the other +/// backend-proxied tool families. +pub fn build_media_tools(root_config: &Config, action_dir: &std::path::Path) -> Vec> { + let Some(client) = crate::openhuman::integrations::build_client(root_config) else { + tracing::debug!("[media_generation] no integration client — media tools skipped"); + return Vec::new(); + }; + + let action_dir = action_dir.to_path_buf(); + let tools: Vec> = vec![ + Box::new(MediaGenerateImageTool::new( + Arc::clone(&client), + action_dir.clone(), + )), + Box::new(MediaGenerateVideoTool::new(Arc::clone(&client), action_dir)), + Box::new(MediaListModelsTool::new(Arc::clone(&client))), + ]; + tracing::debug!("[media_generation] registered {} media tools", tools.len()); + tools +} + +#[cfg(test)] +#[path = "tools_tests.rs"] +mod tools_tests; diff --git a/src/openhuman/media_generation/tools_tests.rs b/src/openhuman/media_generation/tools_tests.rs new file mode 100644 index 000000000..8ef2c2f79 --- /dev/null +++ b/src/openhuman/media_generation/tools_tests.rs @@ -0,0 +1,249 @@ +use std::path::PathBuf; +use std::sync::Arc; + +use serde_json::json; + +use super::{MediaGenerateImageTool, MediaGenerateVideoTool, MediaListModelsTool}; +use crate::openhuman::integrations::IntegrationClient; +use crate::openhuman::tools::traits::{PermissionLevel, Tool, ToolCategory}; + +fn dummy_client() -> Arc { + // No requests are made in these tests; the URL/token are placeholders. + Arc::new(IntegrationClient::new( + "http://127.0.0.1:0".to_string(), + "test-token".to_string(), + )) +} + +#[test] +fn image_tool_schema_and_metadata() { + let tool = MediaGenerateImageTool::new(dummy_client(), PathBuf::from("/tmp")); + assert_eq!(tool.name(), "media_generate_image"); + assert_eq!(tool.permission_level(), PermissionLevel::Execute); + assert_eq!(tool.category(), ToolCategory::Workflow); + + let schema = tool.parameters_schema(); + assert_eq!(schema["required"], json!(["prompt"])); + let props = schema["properties"].as_object().unwrap(); + for key in ["prompt", "model", "size", "n", "input_images", "seed"] { + assert!(props.contains_key(key), "missing image property {key}"); + } +} + +#[test] +fn video_tool_schema_and_metadata() { + let tool = MediaGenerateVideoTool::new(dummy_client(), PathBuf::from("/tmp")); + assert_eq!(tool.name(), "media_generate_video"); + assert_eq!(tool.permission_level(), PermissionLevel::Execute); + assert_eq!(tool.category(), ToolCategory::Workflow); + + let schema = tool.parameters_schema(); + assert_eq!(schema["required"], json!(["prompt"])); + let props = schema["properties"].as_object().unwrap(); + for key in [ + "prompt", + "model", + "input_image", + "duration_seconds", + "aspect_ratio", + "negative_prompt", + "seed", + ] { + assert!(props.contains_key(key), "missing video property {key}"); + } +} + +#[test] +fn list_models_tool_metadata() { + let tool = MediaListModelsTool::new(dummy_client()); + assert_eq!(tool.name(), "media_list_models"); + assert_eq!(tool.category(), ToolCategory::Workflow); + assert!(tool.parameters_schema()["properties"] + .as_object() + .unwrap() + .contains_key("include_upstream")); +} + +#[tokio::test] +async fn image_tool_rejects_empty_prompt_without_network() { + let tool = MediaGenerateImageTool::new(dummy_client(), PathBuf::from("/tmp")); + let result = tool.execute(json!({ "prompt": " " })).await.unwrap(); + assert!(result.is_error); +} + +#[tokio::test] +async fn video_tool_rejects_missing_prompt_without_network() { + let tool = MediaGenerateVideoTool::new(dummy_client(), PathBuf::from("/tmp")); + let result = tool.execute(json!({ "model": "x" })).await.unwrap(); + assert!(result.is_error); +} + +// ── End-to-end flow against a mock backend (wiremock) ─────────────── + +use wiremock::matchers::{method, path, path_regex}; +use wiremock::{Mock, MockServer, ResponseTemplate}; + +fn client_for(server: &MockServer) -> std::sync::Arc { + std::sync::Arc::new(IntegrationClient::new(server.uri(), "tok".to_string())) +} + +/// Mount a media download endpoint that returns `bytes` for the given path. +async fn mount_media(server: &MockServer, p: &str, content_type: &str, bytes: &[u8]) { + Mock::given(method("GET")) + .and(path(p.to_string())) + .respond_with(ResponseTemplate::new(200).set_body_raw(bytes.to_vec(), content_type)) + .mount(server) + .await; +} + +#[tokio::test] +async fn image_tool_submits_downloads_and_persists_local_artifact() { + let server = MockServer::start().await; + let media_url = format!("{}/media/out.png", server.uri()); + Mock::given(method("POST")) + .and(path("/agent-integrations/media-generation/images")) + .respond_with(ResponseTemplate::new(200).set_body_json( + serde_json::json!({ "success": true, "data": { + "requestId": "req-1", + "status": "success", + "model": "seedream-4-0-250828", + "media": [{ "type": "image", "url": media_url }], + "costUsd": 0.039 + } }), + )) + .mount(&server) + .await; + mount_media(&server, "/media/out.png", "image/png", b"PNGBYTES").await; + + let tmp = tempfile::tempdir().unwrap(); + let tool = MediaGenerateImageTool::new(client_for(&server), tmp.path().to_path_buf()); + let res = tool + .execute(json!({ "prompt": "a fox", "size": "1024x1024" })) + .await + .unwrap(); + + assert!(!res.is_error, "expected success, got {res:?}"); + let dir = tmp.path().join("generated-media"); + let files: Vec<_> = std::fs::read_dir(&dir) + .unwrap() + .filter_map(Result::ok) + .collect(); + assert_eq!(files.len(), 1, "exactly one artifact should be persisted"); + assert_eq!(std::fs::read(files[0].path()).unwrap(), b"PNGBYTES"); +} + +#[tokio::test] +async fn video_tool_persists_clip_with_image_to_video_payload() { + let server = MockServer::start().await; + let media_url = format!("{}/media/clip.mp4", server.uri()); + Mock::given(method("POST")) + .and(path("/agent-integrations/media-generation/videos")) + .respond_with(ResponseTemplate::new(200).set_body_json( + serde_json::json!({ "success": true, "data": { + "requestId": "vid-1", + "status": "success", + "model": "seedance-1-0-pro-fast-251015", + "media": [{ "type": "video", "url": media_url, "thumbnailUrl": "https://x/t.png" }], + "costUsd": 0.13 + } }), + )) + .mount(&server) + .await; + mount_media(&server, "/media/clip.mp4", "video/mp4", b"MP4BYTES").await; + + let tmp = tempfile::tempdir().unwrap(); + let tool = MediaGenerateVideoTool::new(client_for(&server), tmp.path().to_path_buf()); + let res = tool + .execute( + json!({ "prompt": "a wave", "input_image": "https://in/f.png", "duration_seconds": 6 }), + ) + .await + .unwrap(); + + assert!(!res.is_error, "expected success, got {res:?}"); + let dir = tmp.path().join("generated-media"); + let files: Vec<_> = std::fs::read_dir(&dir) + .unwrap() + .filter_map(Result::ok) + .collect(); + assert_eq!(files.len(), 1); + assert!(files[0].path().extension().is_some_and(|e| e == "mp4")); +} + +#[tokio::test] +async fn image_tool_polls_until_terminal_then_persists() { + let server = MockServer::start().await; + let media_url = format!("{}/media/p.png", server.uri()); + // Submit returns a non-terminal status; the tool must poll the status endpoint. + Mock::given(method("POST")) + .and(path("/agent-integrations/media-generation/images")) + .respond_with(ResponseTemplate::new(200).set_body_json( + serde_json::json!({ "success": true, "data": { + "requestId": "req-2", "status": "queued", "model": "seedream-4-0-250828", "media": [] + } }), + )) + .mount(&server) + .await; + Mock::given(method("GET")) + .and(path_regex( + r"^/agent-integrations/media-generation/requests/.+", + )) + .respond_with(ResponseTemplate::new(200).set_body_json( + serde_json::json!({ "success": true, "data": { + "requestId": "req-2", + "status": "success", + "model": "seedream-4-0-250828", + "media": [{ "type": "image", "url": media_url }], + "costUsd": 0.039 + } }), + )) + .mount(&server) + .await; + mount_media(&server, "/media/p.png", "image/png", b"POLLED").await; + + let tmp = tempfile::tempdir().unwrap(); + let tool = MediaGenerateImageTool::new(client_for(&server), tmp.path().to_path_buf()); + let res = tool.execute(json!({ "prompt": "a fox" })).await.unwrap(); + assert!(!res.is_error, "expected success after poll, got {res:?}"); + assert_eq!( + std::fs::read_dir(tmp.path().join("generated-media")) + .unwrap() + .count(), + 1 + ); +} + +#[tokio::test] +async fn image_tool_reports_failed_terminal_status() { + let server = MockServer::start().await; + Mock::given(method("POST")) + .and(path("/agent-integrations/media-generation/images")) + .respond_with(ResponseTemplate::new(200).set_body_json( + serde_json::json!({ "success": true, "data": { + "requestId": "req-3", "status": "failed", "model": "seedream-4-0-250828", "media": [] + } }), + )) + .mount(&server) + .await; + let tmp = tempfile::tempdir().unwrap(); + let tool = MediaGenerateImageTool::new(client_for(&server), tmp.path().to_path_buf()); + let res = tool.execute(json!({ "prompt": "a fox" })).await.unwrap(); + assert!(res.is_error); +} + +#[tokio::test] +async fn list_models_tool_returns_backend_catalog() { + let server = MockServer::start().await; + Mock::given(method("GET")) + .and(path("/agent-integrations/media-generation/models")) + .respond_with(ResponseTemplate::new(200).set_body_json( + serde_json::json!({ "success": true, "data": { + "curated": [{ "id": "seedream-4-0-250828", "modality": "image" }] + } }), + )) + .mount(&server) + .await; + let tool = MediaListModelsTool::new(client_for(&server)); + let res = tool.execute(json!({})).await.unwrap(); + assert!(!res.is_error, "expected success, got {res:?}"); +} diff --git a/src/openhuman/media_generation/types.rs b/src/openhuman/media_generation/types.rs new file mode 100644 index 000000000..be15591fb --- /dev/null +++ b/src/openhuman/media_generation/types.rs @@ -0,0 +1,47 @@ +//! Shared types for the `media_generation` agent tools. +//! +//! These mirror the backend's standardized `media_generation` contract +//! (`/agent-integrations/media-generation/*`) — see +//! `backend/docs/media-generation.md`. The backend normalizes GMI's per-model +//! payload/outcome shapes; the core only depends on this stable envelope. + +use serde::Deserialize; + +/// A single generated artifact as returned by the backend. The `url` is an +/// expiring signed URL — the core downloads + persists it locally. +#[derive(Debug, Clone, Deserialize)] +pub struct MediaItem { + #[serde(rename = "type")] + pub kind: String, + pub url: String, + #[serde(rename = "thumbnailUrl", default)] + pub thumbnail_url: Option, +} + +/// Standardized media-generation response envelope. +#[derive(Debug, Clone, Deserialize)] +pub struct MediaResponse { + #[serde(rename = "requestId")] + pub request_id: String, + pub status: String, + #[serde(default)] + pub model: String, + #[serde(default)] + pub media: Vec, + #[serde(rename = "costUsd", default)] + pub cost_usd: f64, +} + +impl MediaResponse { + pub fn is_success(&self) -> bool { + self.status.eq_ignore_ascii_case("success") + } + + pub fn is_failed(&self) -> bool { + self.status.eq_ignore_ascii_case("failed") + } + + pub fn is_terminal(&self) -> bool { + self.is_success() || self.is_failed() + } +} diff --git a/src/openhuman/mod.rs b/src/openhuman/mod.rs index 92cdd4ddf..f0c2b11cf 100644 --- a/src/openhuman/mod.rs +++ b/src/openhuman/mod.rs @@ -65,6 +65,7 @@ pub mod mcp_audit; pub mod mcp_client; pub mod mcp_registry; pub mod mcp_server; +pub mod media_generation; pub mod meet; pub mod meet_agent; pub mod memory; diff --git a/src/openhuman/tools/ops.rs b/src/openhuman/tools/ops.rs index f46b5d3ba..affe676af 100644 --- a/src/openhuman/tools/ops.rs +++ b/src/openhuman/tools/ops.rs @@ -745,6 +745,13 @@ pub fn all_tools_with_runtime( tools.extend(crate::openhuman::search::build_search_tools(root_config)); + // Media generation (image/video via GMI through the backend). Skipped when + // no integration client is configured; artifacts land under `action_dir`. + tools.extend(crate::openhuman::media_generation::build_media_tools( + root_config, + action_dir, + )); + // High-level web3 tools (swaps / bridges / dapp calls) built on the wallet. // They call the backend deBridge proxy per-invocation and error gracefully // when the user is not signed in, so they register unconditionally.