% ai-coding-agents-at-work.tex — using AI coding agents professionally as a senior iOS
% engineer: tool landscape (only confirmed Xcode features), workflow (spec -> plan ->
% tests first -> small diffs -> verify -> review), context management, verification
% gates, failure modes, how to talk about it in an interview, when not to use it.
% Sources (checked 2026-09-25): Xcode 26 release notes (ChatGPT + Claude accounts, any
% Chat Completions provider by API key, local models on Apple silicon, Coding Tools,
% local predictive completion), Xcode 26.3 release notes + Apple Newsroom 2026-02-03
% (agentic coding: Claude Agent + Codex, MCP, permissions), Xcode 27 release notes
% (plan mode, Gemini, simulator control, MCP debugger/build-settings tools, plug-ins
% with skills/MCP/ACP, filesystem security layer, lldb-mcp, Organizer AI recommendations,
% agent String Catalog translation). `xcrun mcpbridge` = secondary sources (tech talk
% 111428 summaries, Anthropic/9to5Mac) — printed as the documented bridge command.
% Worked example of a post-cutoff API trap: WWDC25 286 code iterates PartiallyGenerated
% directly; the shipped FoundationModels API yields ResponseStream.Snapshot (.content).
% No invented metrics or anecdotes on paper: the interview part is a template.
% Build ONLY with: tools/print/print-sheet.py <this>.tex --dry-run
% @source: hiot monorepo, docs/school/sheets/ai/ai-coding-agents-at-work.tex — the SOURCE OF TRUTH; a copy anywhere else (e.g. artur.gurgul.pro) is regenerated from it, never edited
% @labels: area=ai kind=process level=senior platform=general new=yes round=market-2026-09-25 topic=ai,tooling,career
% @tags: coding-agents, agentic-coding, claude-code, codex, xcode-26-3, mcp, rules-file, agents-md, prompt-injection, hallucinated-api, verification-gates, plan-mode
\documentclass[8pt]{extarticle}
\usepackage{printup-sheet}

\lstdefinelanguage{RulesSheet}{
  morekeywords={Build,Test,Stack,Rules,Done,Never,Task,Scope,First},
  sensitive=true, morecomment=[l]{\#}, morestring=[b]',
  literate={->}{{\hbox{-}\hbox{>}}}2 {==}{{\hbox{=}\hbox{=}}}2
           {--}{{\hbox{-}\hbox{-}}}2 {!=}{{\hbox{!}\hbox{=}}}2}

\tikzset{
  lbl/.style={font=\scriptsize, text=black!75, inner sep=1pt},
  you/.style={box, font=\scriptsize, minimum height=10mm, minimum width=19mm, inner sep=2pt, align=center},
  agt/.style={you, draw=sheetOrange, fill=sheetOrange!10},
  gate/.style={draw=sheetGreen, thick, fill=sheetGreen!8, rounded corners=2pt, font=\scriptsize,
               inner sep=2pt, align=left},
}

\begin{document}

\sheettitle{AI coding agents at work — the senior iOS way}{ai · memo}

\oneliner{An agent (Claude Code, Codex, Cursor, Copilot, Xcode's coding assistant) \emph{drafts
and executes}; \textbf{you stay accountable} for every line that ships. The senior skill is not
prompting — it is \textbf{specifying} the task, \textbf{constraining} the context, and
\textbf{verifying} each small change with the compiler, the tests, the running app and
Instruments. Speed comes from short verified loops, not big generated diffs.}

\noindent\begin{tikzpicture}[sheet]
  \node[you] (sp) at (0,0) {\textbf{1 Spec}\\\tiny goal, constraints,\\[-2pt]\tiny ``done'' criteria};
  \node[agt] (pl) at (2.35,0) {\textbf{2 Plan}\\\tiny agent drafts,\\[-2pt]\tiny \textbf{you approve}};
  \node[you] (te) at (4.7,0) {\textbf{3 Failing test}\\\tiny makes ``done''\\[-2pt]\tiny checkable};
  \node[agt] (df) at (7.05,0) {\textbf{4 Small diff}\\\tiny one concern,\\[-2pt]\tiny readable in one go};
  \node[gate] (g) at (9.85,0) {\textbf{5 Gates} (machine)\\$\square$ builds, 0 warnings\\$\square$ tests green\\$\square$ app runs — \emph{look}\\$\square$ Instruments if perf};
  \node[you] (rv) at (12.65,0) {\textbf{6 Review}\\\tiny every line, as if\\[-2pt]\tiny a junior wrote it};
  \node[you, draw=sheetGreen, fill=sheetGreen!10] (cm) at (15.0,0) {\textbf{7 Commit}\\\tiny small; message says\\[-2pt]\tiny what was verified};
  \foreach \a/\b in {sp/pl,pl/te,te/df,df/g,g/rv,rv/cm} \draw[flow] (\a) -- (\b);
  \draw[hot, sheetRed] (g.south) |- (8.4,-1.0) node[lbl, below=1pt, text=sheetRed, anchor=north]{gate red: paste the \emph{exact} error back} -| (df.south);
  \draw[hot, sheetRed] (rv.south) |- (4.0,-1.5) node[lbl, below=1pt, text=sheetRed, anchor=north]{wrong approach: re-plan, don't patch the patch} -| (pl.south);
  \draw[flow, sheetBlue, dashed] (cm.north) -- ++(0,0.3) -| node[lbl, above, pos=0.25, text=sheetBlue]{next small task — fresh context} (sp.north);
  \node[lbl, anchor=west, text=sheetRed, align=left] at (9.6,-1.75) {\textbf{stop rule}: same fix fails twice, or the diff\\outgrows your review $\to$ take over / split the task};
  \node[lbl, anchor=west] at (-1.0,-1.1) {\textcolor{sheetBlue}{\rule{2mm}{2mm}} you \quad \textcolor{sheetOrange}{\rule{2mm}{2mm}} agent \quad \textcolor{sheetGreen}{\rule{2mm}{2mm}} tools};
\end{tikzpicture}

\begin{multicols}{2}

\section{The tools (confirmed features only)}
\begin{itemize}
  \item \textbf{Xcode 26}: coding assistant with \textbf{ChatGPT and Claude} accounts, any
        Chat-Completions provider by API key, or a \textbf{local model} on Apple silicon;
        Coding Tools (explain, document, generate previews/playgrounds); predictive
        completion runs locally.
  \item \textbf{Xcode 26.3} (Feb 2026): \textbf{agentic coding} — Claude Agent and OpenAI Codex
        built in; agents search docs, edit settings, build, test, capture Previews. Xcode's
        tools are exposed over \textbf{MCP} (\texttt{xcrun mcpbridge}\unverified) to any agent, e.g. Claude
        Code in a terminal; a \textbf{permissions} system gates what they may do.
  \item \textbf{Xcode 27}: \textbf{plan mode} (plan = editable Markdown, approve before it
        builds), Gemini, agents drive the \textbf{Simulator} (tap, screenshot), debugger + build
        settings via MCP, plug-ins (skills, MCP servers), a filesystem \textbf{security layer},
        \texttt{lldb-mcp}, AI recommendations in Organizer, agent translation of String Catalogs.
  \item \textbf{Outside Xcode}: Claude Code, Codex CLI, Cursor, GitHub Copilot agent mode.
        Project rules files: \texttt{CLAUDE.md}, \texttt{AGENTS.md}, \texttt{.cursor/rules},
        \texttt{.github/copilot-instructions.md}.
\end{itemize}

\section{Context management}
\begin{itemize}
  \item A \textbf{rules file} in the repo: build/test commands, stack + minimum iOS, conventions,
        forbidden moves. Reviewed like code; it is the team's shared prompt.
  \item \textbf{Small tasks, fresh session each} (long chats drift); name the \emph{exact}
        files/types; paste the \emph{exact} compiler error or crash log.
  \item \textbf{Post-cutoff APIs} (iOS 26/27, Swift 6.2+): the model's training may predate them —
        give it the docs page, or let it search Apple docs via Xcode's MCP.
\end{itemize}

\section{Example — a rules file (excerpt)}
\begin{lstlisting}[language=RulesSheet]
# AGENTS.md / CLAUDE.md
Build: xcodebuild -scheme App build   # warnings = errors
Test:  xcodebuild test -scheme App -only-testing:AppTests
Stack: Swift 6 strict concurrency; iOS 18+; SwiftUI
Rules: @MainActor view models; no force unwraps; no new deps
       API newer than iOS 18: availability check, say so
Done:  clean build, tests green, screenshot of the screen
Never: edit .pbxproj by hand, read Secrets.xcconfig, push
\end{lstlisting}

\columnbreak

\section{Failure modes $\to$ the check}
\begin{itemize}
  \item \textbf{Hallucinated / stale API}, worst after the cutoff. Real case: WWDC25 sample code
        streams \texttt{PartiallyGenerated} values; the shipped Foundation Models API yields a
        \texttt{Snapshot} (\texttt{.content}). \emph{Check}: compile + read the docs page.
  \item \textbf{Confident wrong fix} that silences the symptom: \texttt{@unchecked Sendable},
        \texttt{nonisolated(unsafe)}, \texttt{try!}, a sprinkled \texttt{DispatchQueue.main.async},
        a weakened or deleted test. \emph{Check}: review the diff of the \emph{tests}.
  \item \textbf{Over-large diff}: drive-by refactors hide the one real change. Reject, split.
  \item \textbf{Security}: secrets pasted into prompts or read from the repo; agents running shell
        commands; \textbf{prompt injection} via fetched pages, issues, dependencies. Least
        permissions, sandbox, never auto-approve destructive commands.
  \item \textbf{Licence / policy}: company rules on which tool may see which code; generated
        code resembling copyleft snippets.
  \item \textbf{Understanding debt}: you ship code you can't explain on call at 3 a.m.
\end{itemize}

\section{In the interview}
\begin{itemize}
  \item Tell \textbf{one concrete story}: task · how you constrained it (rules, plan, tests) ·
        how you verified · \textbf{what it got wrong and how you caught it} · outcome.
  \item Metrics that matter: lead time, review rounds, escaped defects —
        \emph{not} ``lines generated''.
  \item Guardrails: human review mandatory, CI gates, no secrets, least-privilege agents.
  \item \textbf{When NOT to}: auth/crypto/payments without expert review; concurrency or
        performance you can't test or measure; code you can't verify; when explaining takes
        longer than doing.
\end{itemize}

\section{Interview traps}
\begin{itemize}
  \trap{``It writes 80\% of my code'' — sounds unreviewed; talk verification.}
  \trap{``Never use it'' — reads as out of date; postings now require it.}
  \trap{Claiming Xcode features you haven't actually run.}
\end{itemize}

\section{Remember}
\textbf{``Spec small, test first, trust nothing unverified.''} The agent types; \textbf{you sign}.

\section{Likely questions}
\begin{enumerate}
  \item Hallucinated APIs? — compile, docs in context, \texttt{\#available}.
  \item Reviewing AI code? — tests first, small diffs, run it.
  \item Team guardrails? — rules file, permissions, CI, review.
\end{enumerate}

\end{multicols}

\noindent{\footnotesize\color{sheetGrey}\textit{Related:} foundation-models (a post-cutoff API) ·
instruments-performance (verify perf claims) · app-hardening-privacy (secrets) · Swift Testing ·
code review · CI}

\end{document}
