Chuyển đến nội dung chính

第10課:AI編輯助手-用自然語言編輯設計

建立一個人工智慧編輯器,接收英語/越南語命令:「使霓虹燈更亮」、「將設計移得更高」、「將顏色更改為紫色」。 InstructPix2Pix、Instruct-NeRF2NeRF 用於影像編輯。 LLM路由意圖。

🧠 人工智慧與機器學習 — 第 9 課 第10課:AI編輯助理-編輯 自然語言設計

人工智慧在行動:建構時尚和按需印刷的人工智慧平台

第 3 部分:AI 設計優化與編輯

亞洲開發網

簡介

建立設計後,使用者想要編輯:「讓霓虹燈更亮」,「向上移動設計」,「將顏色改為紫色」。該平台並沒有強迫使用者使用 Photoshop,而是允許 用自然語言進行設計編輯。本文搭建AI編輯助理。


1. 編輯類別

User instructions phân loại thành 4 nhóm:

1. LAYOUT EDITING
   "move design higher"
   "make it smaller"
   "rotate 15 degrees"
   "center the design"

2. STYLE EDITING
   "make neon brighter"
   "add shadow effect"
   "change to grayscale"
   "add texture"

3. COLOR EDITING
   "change color to purple"
   "make it more vibrant"
   "darken the background"
   "invert colors"

4. CONTENT EDITING
   "remove the text"
   "add a skull"
   "replace cat with dog"
   "add lightning effect"

2. 意圖路由器(LLM 支援)

class EditIntentRouter:
    """Phân tích lệnh edit và route đến handler phù hợp"""

    ROUTING_PROMPT = """
You are an edit intent classifier for a t-shirt design editor.

Classify the user's edit instruction into exactly ONE category:
- LAYOUT: move, resize, rotate, center, align, position
- STYLE: brightness, contrast, shadow, glow, texture, filter, effect
- COLOR: change color, hue, saturation, vibrance, invert, grayscale
- CONTENT: add element, remove element, replace, modify content
- REGENERATE: completely redo, start over, try again

Also extract parameters:
- For LAYOUT: direction, amount, anchor
- For COLOR: target_color, source_color
- For STYLE: effect_name, intensity
- For CONTENT: action, subject

Respond in JSON format.
User instruction: {instruction}
"""

    async def route(self, instruction: str) -> EditIntent:
        response = await self.llm.chat.completions.create(
            model="gpt-4o-mini",
            messages=[{
                "role": "system",
                "content": self.ROUTING_PROMPT.format(
                    instruction=instruction
                )
            }],
            response_format={"type": "json_object"},
            temperature=0,
        )

        intent_data = json.loads(
            response.choices[0].message.content
        )

        return EditIntent(
            category=intent_data["category"],
            params=intent_data.get("params", {}),
            original_instruction=instruction,
        )

3.佈局編輯器(程式設計)

class LayoutEditor:
    """Xử lý layout editing — không cần AI model"""

    def apply(
        self, design: Image.Image, intent: EditIntent
    ) -> Image.Image:
        params = intent.params

        action = params.get("action", "move")

        if action == "move":
            return self._move(
                design,
                direction=params.get("direction", "up"),
                amount=params.get("amount", 10),  # percent
            )
        elif action == "resize":
            return self._resize(
                design,
                scale=params.get("scale", 1.1),
            )
        elif action == "rotate":
            return self._rotate(
                design,
                angle=params.get("angle", 15),
            )
        elif action == "center":
            return self._center(design)

        return design

    def _move(
        self, img: Image.Image, direction: str, amount: int
    ) -> Image.Image:
        """Move design trong canvas"""
        canvas = Image.new("RGBA", img.size, (0, 0, 0, 0))
        offset_px = int(img.height * amount / 100)

        offsets = {
            "up": (0, -offset_px),
            "down": (0, offset_px),
            "left": (-offset_px, 0),
            "right": (offset_px, 0),
        }
        dx, dy = offsets.get(direction, (0, 0))
        canvas.paste(img, (dx, dy), img)
        return canvas

4. 風格編輯器(AI 驅動)

class StyleEditor:
    """AI-based style editing với InstructPix2Pix"""

    def __init__(self):
        from diffusers import (
            StableDiffusionInstructPix2PixPipeline
        )
        self.pipe = (
            StableDiffusionInstructPix2PixPipeline.from_pretrained(
                "timbrooks/instruct-pix2pix",
                torch_dtype=torch.float16,
            )
        )
        self.pipe.to("cuda")

    def apply(
        self,
        design: Image.Image,
        instruction: str,
        strength: float = 0.5,
    ) -> Image.Image:
        """
        Apply style edit bằng natural language

        strength: 0.3 (subtle) → 0.8 (dramatic)
        """
        result = self.pipe(
            prompt=instruction,
            image=design,
            num_inference_steps=20,
            image_guidance_scale=1.5,
            guidance_scale=7.5,
        ).images[0]

        return result

5. 顏色編輯器

class ColorEditor:
    """Chỉnh màu design"""

    def change_color(
        self,
        design: Image.Image,
        source_color: str | None,
        target_color: str,
    ) -> Image.Image:
        """Đổi màu design"""
        import numpy as np
        from colorsys import rgb_to_hsv, hsv_to_rgb

        img_array = np.array(design.convert("RGBA"))
        rgb = img_array[:, :, :3].astype(float) / 255

        target_rgb = self._hex_to_rgb(target_color)
        target_hsv = rgb_to_hsv(*target_rgb)

        # Convert to HSV
        h, s, v = np.vectorize(rgb_to_hsv)(
            rgb[:, :, 0], rgb[:, :, 1], rgb[:, :, 2]
        )

        # Shift hue to target, keep saturation and value
        h_new = np.full_like(h, target_hsv[0])
        s_new = s * (target_hsv[1] / max(np.mean(s), 0.01))
        s_new = np.clip(s_new, 0, 1)

        # Convert back
        r, g, b = np.vectorize(hsv_to_rgb)(h_new, s_new, v)
        result = np.stack([r, g, b], axis=-1) * 255

        img_array[:, :, :3] = result.astype(np.uint8)
        return Image.fromarray(img_array)

    def adjust_brightness(
        self, design: Image.Image, factor: float
    ) -> Image.Image:
        from PIL import ImageEnhance
        enhancer = ImageEnhance.Brightness(design)
        return enhancer.enhance(factor)

    def adjust_vibrance(
        self, design: Image.Image, factor: float
    ) -> Image.Image:
        from PIL import ImageEnhance
        enhancer = ImageEnhance.Color(design)
        return enhancer.enhance(factor)

6. 統一編輯管道

class EditingAssistant:
    """Unified pipeline cho tất cả editing operations"""

    def __init__(self):
        self.router = EditIntentRouter()
        self.editors = {
            "LAYOUT": LayoutEditor(),
            "STYLE": StyleEditor(),
            "COLOR": ColorEditor(),
            "CONTENT": ContentEditor(),
        }

    async def edit(
        self,
        design: Image.Image,
        instruction: str,
    ) -> EditResult:
        # 1. Route intent
        intent = await self.router.route(instruction)

        # 2. Get appropriate editor
        editor = self.editors.get(intent.category)
        if not editor:
            return EditResult(
                success=False,
                message=f"Unsupported edit type: {intent.category}"
            )

        # 3. Apply edit
        edited = editor.apply(design, intent)

        # 4. Validate result
        quality = PrintQualityGate().check_all(edited)

        return EditResult(
            success=True,
            image=edited,
            intent=intent,
            quality_report=quality,
        )

總結

人工智慧編輯助理:

  1. 意圖路由-LLM將編輯指令分為4類
  2. 佈局編輯器 — 編程移動/調整大小/旋轉/居中
  3. 樣式編輯器 — InstructPix2Pix 用於基於 AI 的樣式更改
  4. 顏色編輯器 — 色調偏移、亮度、鮮豔度調整
  5. 品質驗證 — 每次編輯後自動檢查

下一篇文章:AI Typography — 產生 T 卹文字、字體樣式和自動放置。