% coreml-vision-on-device.tex — Core ML (formats, compute units, configuration, async
% prediction, conversion + compression with coremltools, MLTensor / MLState, on-device
% update), Vision (Swift-native API iOS 18 vs VN*), Natural Language, Speech
% (SpeechAnalyzer iOS 26), Create ML, Core AI (iOS 27); choosing FM vs Core ML vs server;
% performance, memory, privacy.
% Sources (checked 2026-09-25): developer.apple.com/documentation/coreml (MLModel,
% MLComputeUnits, MLTensor, makeState, overview "see Core AI"), /vision (Swift-only API
% from iOS 18: RecognizeTextRequest, CoreMLRequest, CoreMLModelContainer, RecognizedText,
% RecognizeDocumentsRequest 26), /speech/speechanalyzer (26), /naturallanguage, /coreai
% (27: AIModel, InferenceFunction, NDArray, AIModelCache; "non-neural-network models: see
% Core ML"), WWDC26 324 "Meet Core AI", WWDC23 10049 (async prediction, iOS 17),
% coremltools optimize docs (palettize 1-8 bit, linear int8/int4, int8-int8 on ANE from
% iPhone 15 Pro).
% NOT on paper (could not confirm): whether Core ML is formally deprecated (docs only
% point to Core AI for newest architectures); exact first-load cache rules.
% Build ONLY with: tools/print/print-sheet.py <this>.tex --dry-run
% @source: hiot monorepo, docs/school/sheets/ai/coreml-vision-on-device.tex — the SOURCE OF TRUTH; a copy anywhere else (e.g. artur.gurgul.pro) is regenerated from it, never edited
% @labels: area=ai kind=api level=senior platform=apple new=no round=market-2026-09-25 topic=ai,platform-apis,performance
% @tags: coreml, mlpackage, mlmodelc, neural-engine, computeunits, coremltools, quantization, palettization, vision, recognizetextrequest, speechanalyzer, core-ai
\documentclass[8pt]{extarticle}
\usepackage{printup-sheet}

\lstdefinelanguage{SwiftSheet}{
  morekeywords={protocol,class,final,struct,enum,func,var,let,static,some,init,
    if,else,return,guard,self,nil,private,true,false,in,try,await,async,throws,
    for,case,String,Int},
  sensitive=true, morecomment=[l]{//}, morestring=[b]",
  literate={->}{{\hbox{-}\hbox{>}}}2 {==}{{\hbox{=}\hbox{=}}}2
           {??}{{\hbox{?}\hbox{?}}}2 {!=}{{\hbox{!}\hbox{=}}}2}

\tikzset{
  lbl/.style={font=\scriptsize, text=black!75, inner sep=1pt},
  stg/.style={box, font=\scriptsize, minimum height=9mm, inner sep=2pt, align=center},
  lay/.style={draw, rounded corners=2pt, font=\scriptsize, minimum height=6mm, inner sep=1.5pt, align=center},
  hw/.style={lay, draw=sheetGrey, fill=black!6, minimum width=21mm},
}

\begin{document}

\sheettitle{Core ML · Vision · on-device ML (+ Core AI, iOS 27)}{ai · memo}

\oneliner{\textbf{Core ML} runs a \emph{trained model file} on the device and splits its graph
across \textbf{CPU, GPU and Neural Engine}; \textbf{Vision}, \textbf{Natural Language} and
\textbf{Speech} are task frameworks on top (Apple's models, or yours); \textbf{Create ML} and
\textbf{coremltools} make the file. iOS 27 adds \textbf{Core AI} for modern neural nets. The
data never leaves the phone — the \emph{model} ships in your bundle.}

\noindent\begin{tikzpicture}[sheet]
  \node[stg, draw=sheetGrey, fill=black!4] (tr) at (0,0) {train\\\tiny PyTorch · Create ML};
  \node[stg, draw=sheetBrown, fill=sheetBrown!8] (cv) at (3.0,0) {\texttt{ct.convert}\\\tiny + palettize /\\[-2pt]\tiny quantize / prune};
  \node[stg] (pk) at (6.0,0) {\texttt{.mlpackage}\\\tiny ML Program, source};
  \node[stg, draw=sheetOrange, fill=sheetOrange!8] (xc) at (9.1,0) {Xcode build\\\tiny\texttt{.mlmodelc} + typed\\[-2pt]\tiny Swift class};
  \node[stg, draw=sheetGreen, fill=sheetGreen!10] (ld) at (12.2,0) {first \texttt{load}\\\tiny specialise for this\\[-2pt]\tiny device, \textbf{cache}};
  \node[stg, draw=sheetRed, fill=sheetRed!6] (pr) at (15.1,0) {\texttt{prediction}\\\tiny async (iOS 17)};
  \foreach \a/\b in {tr/cv,cv/pk,pk/xc,xc/ld,ld/pr} \draw[flow] (\a) -- (\b);
  \node[lbl, text=sheetBrown, anchor=north] at (3.0,-0.55) {Python, on your Mac};
  \node[lbl, text=sheetOrange, anchor=north, align=center] at (9.1,-0.55) {or download \texttt{.mlpackage}\\$\to$ \texttt{MLModel.compileModel(at:)}};
  \node[lbl, text=sheetGreen!60!black, anchor=north, align=center] at (12.2,-0.55) {slow once (seconds):\\load early, off main, reuse};
  \node[lbl, anchor=north, align=center] at (15.1,-0.55) {\texttt{computeUnits}:\\\texttt{.all} = CPU+GPU+ANE};
\end{tikzpicture}

\begin{multicols}{2}

\section{How it works — Core ML}
\begin{itemize}
  \item \textbf{Formats}: \texttt{.mlmodel} (one file, older \emph{neuralnetwork} type) ·
        \texttt{.mlpackage} (folder, \textbf{ML Program}, weights apart; coremltools' default) ·
        \texttt{.mlmodelc} (compiled, what runs). Xcode compiles at build and generates a class.
  \item \texttt{MLModelConfiguration().computeUnits}: \texttt{.all} (default) ·
        \texttt{.cpuOnly} · \texttt{.cpuAndGPU} · \texttt{.cpuAndNeuralEngine}. Core ML
        \emph{partitions} the graph; ops the ANE can't run fall back. \textbf{Background}: no
        GPU — use CPU/ANE. \texttt{MLComputePlan} shows where each op runs.
  \item \textbf{Load}: \texttt{await MLModel.load(contentsOf:}\allowbreak\texttt{configuration:)};
        the first load on a device specialises (ANE compile) and is cached — later loads are fast.
  \item \textbf{Predict}: \texttt{prediction(from:)} (\texttt{MLFeatureProvider} in/out);
        iOS 17 \textbf{async} overload: thread-safe, cancellable, runs concurrently — cap
        in-flight requests (memory). Batch: \texttt{predictions(fromBatch:)}.
  \item \textbf{iOS 18}: \texttt{MLTensor} (tensor maths), stateful models
        (\texttt{makeState()} $\to$ \texttt{MLState}, e.g. an LLM KV-cache), multifunction models.
  \item \textbf{Compress} (\texttt{ct.optimize.coreml}): \texttt{palettize\_weights}
        (k-means LUT, 1--8 bit) · \texttt{linear\_quantize\_weights} (int8/int4) ·
        \texttt{prune\_weights}. Smaller app, less memory traffic; re-check accuracy.
  \item \textbf{On-device training}: updatable models + \texttt{MLUpdateTask}
        (personalise on user data, never uploaded).
  \item \textbf{Core AI (iOS 27)}: \texttt{.aimodel}, \texttt{AIModel}, \texttt{InferenceFunction}
        \texttt{.run(inputs:)}, \texttt{NDArray}; PyTorch $\to$ \texttt{coreai\_torch}; ahead-of-time
        compile, \texttt{AIModelCache}. Docs: non-neural models (trees, tabular) $\to$ Core ML.
\end{itemize}

\section{The task frameworks}
\begin{itemize}
  \item \textbf{Vision}: iOS 18 \textbf{Swift-only API} — struct requests
        (\texttt{RecognizeTextRequest}, \texttt{ClassifyImageRequest},
        \texttt{DetectFaceRectanglesRequest}, \texttt{DetectBarcodesRequest},
        \texttt{CoreMLRequest}\ldots), \texttt{try await req.perform(on:)} $\to$ typed observations.
        Legacy: \texttt{VNImageRequestHandler} + \texttt{VNRequest}s, cast results. iOS 26:
        \texttt{RecognizeDocumentsRequest} (paragraphs, tables, lists). Coordinates are
        \textbf{normalised} 0--1 — convert before drawing.
  \item \textbf{Natural Language}: \texttt{NLLanguageRecognizer}, \texttt{NLTokenizer},
        \texttt{NLTagger} (lemma, names, sentiment), \texttt{NLEmbedding},
        \texttt{NLContextualEmbedding}, custom \texttt{NLModel}.
  \item \textbf{Speech}: iOS 26 \texttt{SpeechAnalyzer} (actor) + \texttt{SpeechTranscriber}
        module; \texttt{AssetInventory} downloads the language model; on-device, long-form.
        Older: \texttt{SFSpeechRecognizer}.
  \item \textbf{Create ML}: no-code training (image/text/sound/tabular), transfer learning on
        Apple's feature extractors $\to$ small \texttt{.mlmodel}.
\end{itemize}

\columnbreak

\section{Picture — where each framework sits}
\begin{tikzpicture}[sheet]
  \node[lay, draw=sheetGrey, fill=white, minimum width=76mm] at (3.8,2.1) {\textbf{your app}};
  \node[lay, draw=sheetOrange, fill=sheetOrange!10, minimum width=17mm] at (0.85,1.3) {Foundation\\[-2pt]Models};
  \node[lay, draw=sheetBlue, fill=sheetBlue!10, minimum width=12mm] at (2.55,1.3) {Vision};
  \node[lay, draw=sheetBlue, fill=sheetBlue!10, minimum width=13mm] at (4.0,1.3) {Natural\\[-2pt]Language};
  \node[lay, draw=sheetBlue, fill=sheetBlue!10, minimum width=12mm] at (5.45,1.3) {Speech};
  \node[lay, draw=sheetGreen, fill=sheetGreen!10, minimum width=13mm] at (6.9,1.3) {your\\[-2pt]model};
  \node[lay, draw=sheetBrown, fill=sheetBrown!10, minimum width=62mm] at (4.6,0.5) {\textbf{Core ML} (\texttt{.mlmodelc}) \quad·\quad \textbf{Core AI} (\texttt{.aimodel}, 27)};
  \node[lay, draw=sheetOrange, fill=sheetOrange!4, minimum width=13mm] at (0.85,0.5) {system LLM};
  \node[hw] at (1.2,-0.3) {CPU\\[-2pt]\tiny precise, any op};
  \node[hw] at (3.8,-0.3) {GPU\\[-2pt]\tiny parallel, shares UI};
  \node[hw, draw=sheetGreen, fill=sheetGreen!8] at (6.4,-0.3) {Neural Engine\\[-2pt]\tiny best perf/watt};
  \node[lbl, text=sheetBlue, anchor=west] at (-0.15,-0.95) {Vision's \texttt{CoreMLRequest} = your model + Vision's crop/scale/orientation};
\end{tikzpicture}

\section{Example — OCR, then your own model}
\begin{lstlisting}[language=SwiftSheet]
var ocr = RecognizeTextRequest()          // Vision, iOS 18
ocr.recognitionLevel = .accurate
let lines = try await ocr.perform(on: cgImage)
  .compactMap { $0.topCandidates(1).first?.string }

let cfg = MLModelConfiguration()
cfg.computeUnits = .cpuAndNeuralEngine    // keep GPU for UI
let m = try await MLModel.load(contentsOf: url, configuration: cfg)
let req = CoreMLRequest(model:
  try CoreMLModelContainer(model: m, featureProvider: nil))
let result = try await req.perform(on: cgImage) // classifier
\end{lstlisting}

\section{Which one?}
{\small
\begin{tabular}{@{}p{19mm}p{59mm}@{}}
\toprule
\textbf{Foundation M.} & language: summarise, extract, tag; no model to ship \\
\textbf{Vision/NL/Speech} & common tasks (OCR, faces, barcodes, dictation) — try first \\
\textbf{Core ML / AI} & your domain (defects, own classes), offline, per frame \\
\textbf{Server} & huge models, world knowledge; cost, latency, privacy \\
\bottomrule
\end{tabular}}

\section{Interview traps}
\begin{itemize}
  \trap{Loading per request / on main — load once, early, reuse.}
  \trap{\texttt{.all} in a background task — GPU denied; pick CPU/ANE.}
  \trap{``On-device = free'': battery, heat, app size — compress, measure (Core ML
        instrument, Xcode performance report).}
  \trap{Vision boxes are normalised (VN: bottom-left origin).}
  \trap{The \texttt{.mlmodelc} in the IPA is extractable — encrypt it if it is IP.}
\end{itemize}

\section{Remember}
\textbf{``Task framework first, own model second, server last''} ·
\textbf{load once, compress, pick the unit}.

\section{Likely questions}
\begin{enumerate}
  \item \texttt{.mlpackage} vs \texttt{.mlmodelc}? — source vs compiled.
  \item Why the ANE? — perf/watt; unsupported ops fall back.
  \item Shrink a model? — palettise / quantise / prune in coremltools.
  \item Vision vs raw Core ML? — Vision scales/crops; typed results.
\end{enumerate}

\end{multicols}

\noindent{\footnotesize\color{sheetGrey}\textit{Related:} foundation-models · instruments-performance
(memory, energy) · app-hardening-privacy (IP in the bundle) · Swift concurrency (cancellation) ·
Background Tasks}

\end{document}
