{
  "schema_version": "0.1.0",
  "record_type": "research-topic",
  "topic_id": "multimodal-ai",
  "label": "Multimodal AI",
  "title": "Multimodal AI Research",
  "description": "Multimodal AI research papers in the Naser Ezzati-Jivan publication catalog.",
  "introduction": "This topic page groups Naser Ezzati-Jivan research papers related to multimodal ai. Each linked record provides the paper's problem, method, findings, limitations, keywords, and authoritative source links.",
  "aliases": [
    "multimodal ai"
  ],
  "search_terms": [
    "Multimodal AI"
  ],
  "related_topics": [],
  "canonical_url": "https://threadslab.org/research-publications/topics/multimodal-ai.html",
  "paper_count": 2,
  "papers": [
    {
      "paper_id": "ai-video-retrieval-semantic-search-timestamp-alignment",
      "title": "AI Video Retrieval: A Semantic Search & Timestamp Alignment System",
      "year": 2025,
      "authors": [
        "Hridoy Rahman",
        "Naser Ezzati-Jivan",
        "Blessing Ogbuokiri"
      ],
      "page_url": "https://threadslab.org/research-publications/papers/ai-video-retrieval-semantic-search-timestamp-alignment/",
      "canonical_source_url": "https://doi.org/10.1109/ACDSA65407.2025.11166430",
      "core_contribution": "The paper implements a timestamp-aware multimodal video-retrieval pipeline that joins speech transcription, sampled-frame captioning, text embeddings, and approximate-nearest-neighbor search.",
      "tags": [
        "multimodal-ai",
        "machine-learning",
        "benchmark-datasets"
      ],
      "keywords": [
        "video retrieval",
        "semantic search",
        "timestamp alignment",
        "AI video search",
        "ACDSA 2025"
      ]
    },
    {
      "paper_id": "picturing-ambiguity-winograd-schema",
      "title": "Picturing Ambiguity: A Visual Twist on the Winograd Schema Challenge",
      "year": 2024,
      "authors": [
        "Brendan Park",
        "Madeline Janecek",
        "Naser Ezzati-Jivan",
        "Yifeng Li",
        "Ali Emami"
      ],
      "page_url": "https://threadslab.org/research-publications/papers/picturing-ambiguity-winograd-schema/",
      "canonical_source_url": "https://aclanthology.org/2024.acl-long.22/",
      "core_contribution": "The paper introduces WinoVis, a multimodal benchmark and analysis framework for testing pronoun disambiguation in text-to-image models.",
      "tags": [
        "multimodal-ai",
        "benchmark-datasets",
        "common-sense-reasoning",
        "machine-learning"
      ],
      "keywords": [
        "Winograd Schema Challenge",
        "WinoVis",
        "text-to-image models",
        "pronoun disambiguation",
        "DAAM",
        "Stable Diffusion"
      ]
    }
  ]
}
