[
    {
        "id": "osp-24015",
        "type": "article-journal",
        "title": "Form and Void: Entangled Composition through an Autonomous AI Agent",
        "author": [
            {
                "family": "Wang",
                "given": "Shiwen"
            },
            {
                "family": "Yang",
                "given": "Jian"
            },
            {
                "family": "Wang",
                "given": "Xu"
            },
            {
                "family": "Wang",
                "given": "Xincan"
            },
            {
                "family": "Dong",
                "given": "Weiming"
            }
        ],
        "URL": "https://omanscience.com/en/articles/form-and-void-entangled-composition-through-an-autonomous-ai-agent",
        "language": "en",
        "issued": {
            "date-parts": [
                [
                    2026
                ]
            ]
        },
        "abstract": "Positive and negative space is a fundamental principle in visual composition, supporting visually coherent forms and layered semantic relationships. Generating such compositions is challenging because it requires coordinated control over two semantic concepts that share a common boundary. Although recent text-to-image models and multimodal large language models (MLLMs) have achieved strong performance in image generation and visual understanding, positive-negative space generation remains difficult, particularly under direct single-pass prompting. In this work, we present the \\textbf{F}orm \\textbf{a}nd \\textbf{V}oid \\textbf{A}gent (\\textbf{FaV-A}), a multimodal agent designed for staged positive-negative space generation. FaV-A follows a progressive workflow: it first generates a base object, then analyzes its shape and spatial structure to identify candidate negative-space semantics, and finally produces compositional instructions for the final image generation stage. Experimental results and ablation analyses suggest that FaV-A provides a more effective framework than direct zero-shot MLLM baselines for producing visually coherent and semantically aligned positive-negative space compositions."
    }
]