\documentclass[11pt]{beamer}

\geometry{paperwidth=215.9mm,paperheight=279.4mm}
\usepackage{amsmath,amssymb}
\usepackage[T1]{fontenc}
\usepackage{lmodern}
\setbeamersize{text margin left=18mm,text margin right=18mm}
\setbeamertemplate{navigation symbols}{}
\setbeamertemplate{enumerate items}[default]
\setbeamercovered{invisible}
\setbeamercolor{structure}{fg=blue!70!black}
\setbeamerfont{frametitle}{size=\Large,series=\bfseries}

\title{Adapter-Style Vision-Language Models: Oral Knowledge Questions}
\author{}
\date{}

\newcommand{\questionone}{%
    \item A Vision Transformer already produces vector embeddings. Why can
    these embeddings not always be inserted unchanged into a decoder-only
    language model? Discuss embedding width, sequence length, representation
    compatibility, and preservation of spatial information.}

\newcommand{\questiontwo}{%
    \item Describe the path followed by an image from the input image to the
    LLaVA language model. Explain how the vision encoder creates patch
    features, how the projector transforms them, where the projected visual
    tokens are inserted, how the visual prefix differs from ordinary text
    context, and why later assistant tokens can use the visual information.}

\newcommand{\questionthree}{%
    \item What changes between LLaVA Stage 1 feature alignment and Stage 2
    visual instruction tuning? For each stage, explain the training examples,
    conditioning inputs, supervised target tokens, trainable parameters, and
    behavior the model is expected to learn.}

\newcommand{\questionfour}{%
    \item Suppose full visual instruction tuning produces a much higher
    evaluation score than feature alignment alone. What does this ablation
    demonstrate? Explain what each stage teaches, why caption generation does
    not guarantee diverse visual instruction following, what conclusion the
    performance gap supports, and what the ablation does not prove by itself.}

\newcommand{\questionfive}{%
    \item LLaVA places projected visual tokens before the user prompt and
    assistant answer in a decoder-only sequence. Explain why assistant tokens
    can attend to the earlier visual tokens, how the causal attention mask
    operates, why visual-token and user-prompt positions are normally excluded
    from the supervised loss, and how the answer-token loss can be written
    mathematically.}

\begin{document}
\begin{frame}[t]{Adapter-Style Vision-Language Models:\\Oral Knowledge Questions}
\raggedright
\small Explain each answer in your own words and justify the underlying mechanism.
\vspace{1.5em}
\normalsize
\begin{enumerate}
\setlength{\itemsep}{1.4em}
\questionone
\pause
\questiontwo
\pause
\questionthree
\pause
\questionfour
\pause
\questionfive
\end{enumerate}
\end{frame}

\end{document}
