% c-volatile-atomics-isr.tex — what volatile does and does not do, C11 atomics
% + memory orders, ISR-shared state, critical sections, ESP32 dual core.
% Sources: docs/memos/c-systems.md (Q1, Q2, tips), docs/memos/embedded-interrupts-dma.md,
% docs/memos/c-undefined-behavior.md (Q9), docs/school/qaa/c-systems.md Q16, Q17, Q20,
% docs/school/session/2026-06-24.md (c-systems Q1 PARTIAL, Q2 INCORRECT).
% Build ONLY with: tools/print/print-sheet.py <this>.tex --dry-run
% @source: hiot monorepo, docs/school/sheets/c/c-volatile-atomics-isr.tex — the SOURCE OF TRUTH; a copy anywhere else (e.g. artur.gurgul.pro) is regenerated from it, never edited
% @labels: area=c kind=concept level=deep platform=embedded new=no round=c-esp32-2026-09-24 topic=concurrency,hardware,language
% @tags: volatile, atomic, stdatomic, memory-order, acquire-release, isr, lost-update, critical-section, portmux, dual-core, memory-barrier
\documentclass[8pt]{extarticle}
\usepackage{printup-sheet}

\lstdefinestyle{CSheet}{style=printup, language=C,
  morekeywords={uint8_t,uint16_t,uint32_t,uint64_t,size_t,bool,_Atomic,restrict,
    inline,TaskHandle_t,BaseType_t,portMUX_TYPE},
  morekeywords=[2]{IRAM_ATTR}, keywordstyle=[2]\color{tangoOperator}}

\tikzset{
  lane/.style={font=\bfseries\scriptsize, anchor=east},
  op/.style={draw=sheetGrey, fill=white, rounded corners=1pt, font=\ttfamily\scriptsize,
             inner sep=1.5pt, minimum height=4mm},
  lbl/.style={font=\scriptsize, text=black!75, inner sep=1pt},
}

\begin{document}

\sheettitle{volatile · \_Atomic · sharing state with an ISR or another core}{c · memo}

\oneliner{\texttt{volatile} is a promise about the \textbf{compiler}: every read
and write in the source becomes a real load/store, never cached in a register or
removed. It gives \textbf{no atomicity}, \textbf{no ordering} of ordinary memory
and \textbf{no CPU barrier}. State shared by a task and an ISR or another core
needs \texttt{\_Atomic} or a \textbf{critical section}.}

\begin{multicols}{2}

\section{What volatile does}
\begin{itemize}
  \item Each access is \emph{observable behaviour}: emitted, never merged or
        dropped, not reordered \emph{against other volatile accesses}.
  \item The ISR-flag bug: without it \texttt{while(!flag)\{\}} is read
        \emph{once} (nothing in the loop writes it) $\to$
        \texttt{if(!flag) for(;;);}.
  \item For: \textbf{memory-mapped registers} (a read may pop a FIFO or clear
        a status bit), \textbf{vars written by an ISR}, \texttt{volatile
        sig\_atomic\_t} signal flags, locals live across \texttt{longjmp}.
  \item \texttt{volatile uint32\_t *r} = pointer \emph{to} volatile (registers);
        \texttt{uint32\_t *volatile p} = volatile pointer (rarely meant).
\end{itemize}

\section{What volatile does NOT do}
\begin{itemize}
  \item \textbf{Atomicity:} \texttt{v++} = LOAD, ADD, STORE; an ISR fits
        between any two. 64-bit on a 32-bit CPU = 2 loads $\to$ \textbf{torn read}.
  \item \textbf{Order plain memory:} a normal \texttt{buf[i]=x} may move
        across a volatile store to \texttt{ready}.
  \item \textbf{CPU barrier:} none emitted — another core may see the
        writes in a different order.
  \item \textbf{Cache bypass:} it stops \emph{register} caching, not the data
        cache (DMA $\to$ invalidate / non-cacheable RAM).
  \item \textbf{Thread safety:} a data race (unsynchronised, one side writes)
        is UB in C11, volatile or not.
\end{itemize}

\section{C11 atomics — \texttt{<stdatomic.h>}}
\begin{itemize}
  \item \texttt{\_Atomic uint32\_t n;}: plain \texttt{n++}, \texttt{n+=2},
        \texttt{n=0} are atomic, \texttt{seq\_cst}. API:
        \texttt{atomic\_load/store}, \texttt{atomic\_fetch\_add},
        \texttt{atomic\_exchange}, \texttt{atomic\_compare\_exchange\_strong/weak},
        each with an \texttt{\_explicit(\dots, order)} form.
  \item Hardware: ARMv7-M \texttt{LDREX/STREX} loop · x86 \texttt{LOCK} ·
        ESP32/S3 Xtensa \texttt{S32C1I} (CAS). 64-bit on a 32-bit MCU may
        not be lock-free (\texttt{atomic\_is\_lock\_free}).
  \item \texttt{asm volatile("":::"memory")} = \emph{compiler-only} barrier;
        \texttt{\_\_DMB()} / \texttt{atomic\_thread\_fence} = CPU barrier.
\end{itemize}
\begin{tabular}{@{}p{11mm}p{65mm}@{}}
\toprule
\textbf{order} & \textbf{guarantees}\\ \midrule
\texttt{relaxed} & atomic only, no ordering — counters, stats\\
\texttt{acquire} & load: later accesses can't move \emph{above} it\\
\texttt{release} & store: earlier accesses can't move \emph{below} it; an
  acquire that reads it sees them all\\
\texttt{acq\_rel} & both — for RMW (\texttt{fetch\_add}, CAS)\\
\texttt{seq\_cst} & default: acq/rel + one global order; costliest\\
\bottomrule
\end{tabular}

\section{Example — the ISR counter, right}
\begin{lstlisting}[style=CSheet]
static _Atomic uint32_t events;   // ISR <-> task
static TaskHandle_t worker;
void IRAM_ATTR gpio_isr(void *arg) {
  atomic_fetch_add_explicit(&events, 1,
                            memory_order_relaxed);
  BaseType_t woken = pdFALSE;
  vTaskNotifyGiveFromISR(worker, &woken); // wake
  portYIELD_FROM_ISR(woken);
}
void worker_task(void *arg) {
  for (;;) {
    ulTaskNotifyTake(pdTRUE, portMAX_DELAY); // sleep
    handle(atomic_exchange(&events, 0)); // read+reset=1 op
  }
}
\end{lstlisting}

\columnbreak

\section{Picture — the lost update}
\begin{tikzpicture}[sheet]
  \node[lane] at (0,0.6) {task};
  \node[lane] at (0,-0.2) {ISR};
  \node[lane] at (0,-0.8) {\texttt{count}};
  \draw[sheetGrey!50] (0,0.6) -- (7.3,0.6);
  \draw[sheetGrey!50] (0,-0.2) -- (7.3,-0.2);
  \node[op] (a) at (0.6,0.6) {LOAD 5};
  \node[op, draw=sheetOrange] (b) at (2.0,-0.2) {LOAD 5};
  \node[op, draw=sheetOrange] (c) at (3.15,-0.2) {ADD};
  \node[op, draw=sheetOrange] (d) at (4.35,-0.2) {STORE 6};
  \node[op] (e) at (5.45,0.6) {ADD};
  \node[op, draw=sheetRed, very thick] (f) at (6.6,0.6) {STORE 6};
  \draw[hot] (a.south east) -- node[lbl, left, pos=0.6]{IRQ} (b.north west);
  \draw[flow] (d.north east) -- node[lbl, right, pos=0.4]{return} (e.south west);
  \foreach \x/\v in {0.6/5, 2.0/5, 4.35/6}
    \node[font=\ttfamily\scriptsize] at (\x,-0.8) {\v};
  \node[font=\ttfamily\scriptsize\bfseries, text=sheetRed] at (6.6,-0.8) {6, not 7};
  \node[note, anchor=west] at (-0.9,-1.2) {\texttt{volatile} changes nothing here. Two cores: same, no IRQ needed.};
\end{tikzpicture}

\section{Picture — publish with release / acquire}
\begin{tikzpicture}[sheet]
  \node[font=\bfseries\scriptsize] at (1.1,0.5) {core 0 — producer};
  \node[font=\bfseries\scriptsize] at (5.2,0.5) {core 1 — consumer};
  \node[op, anchor=west] (w1) at (0,0) {buf = data;};
  \node[op, anchor=west, draw=sheetBlue, very thick] (w2) at (0,-0.65) {store(\&ready,1,release)};
  \node[op, anchor=west, draw=sheetBlue, very thick] (r1) at (3.6,0) {while(!load(\&ready,acquire));};
  \node[op, anchor=west] (r2) at (3.6,-0.65) {use(buf); // sees data};
  \draw[flow] (w1) -- (w2);
  \draw[flow] (r1) -- (r2);
  \draw[hot] (w2.east) -- node[lbl, below, sloped, pos=0.55]{syncs-with} (r1.south west);
  \node[note, anchor=west] at (0,-1.1) {\texttt{volatile ready} instead: \texttt{buf} may land \emph{after} \texttt{ready}.};
\end{tikzpicture}

\section{Critical sections, FreeRTOS, ESP32}
\begin{itemize}
  \item Use one for \textbf{several fields}, a 64-bit value, or
        check-then-act. Single core: \texttt{taskENTER\_CRITICAL()} masks
        interrupts. A few instructions only; never block inside.
  \item \textbf{ESP32 is dual-core.} Masking interrupts stops only
        \emph{this} core, so ESP-IDF adds a spinlock:
        \texttt{portENTER\_CRITICAL(\&mux)} (\texttt{\_ISR} variant in ISRs),
        \texttt{portMUX\_TYPE mux = portMUX\_INITIALIZER\_UNLOCKED}.
        Unpinned tasks (\texttt{tskNO\_AFFINITY}) run truly in parallel.
  \item A mutex \textbf{cannot be taken in an ISR} (may block). A semaphore /
        notification \emph{wakes} a task — it does not make access atomic.
\end{itemize}

\section{Interview traps}
\begin{itemize}
  \trap{\textbf{Your Q1:} name the missing \texttt{volatile} \emph{first}
        (correctness), then the busy-wait (efficiency).}
  \trap{\textbf{Your Q2:} \texttt{volatile counter++} ``looks fine'' — 3
        instructions; the task's check-then-reset is a 2nd race.}
  \trap{\textbf{Your Q16:} ``volatile skips the cache'' — no, the
        \emph{register} copy; CPU caches are untouched.}
  \trap{\textbf{Your Q20:} semaphore $\neq$ atomicity; \texttt{handle(c); c=0;}
        drops ISR increments $\to$ \texttt{atomic\_exchange}.}
  \trap{No ``Mars rover + volatile'' story: Pathfinder 1997 was priority inversion.}
\end{itemize}

\section{Remember}
\textbf{volatile: the compiler won't cache it. \_Atomic: nobody can split it.
Release publishes, acquire subscribes. Register $\to$ volatile · one word $\to$
atomic · several $\to$ critical section · wait $\to$ \texttt{FromISR} signal.}

\section{Likely questions}
\begin{enumerate}
  \item Is \texttt{volatile} thread-safe? — No: no atomicity, no ordering.
  \item \texttt{\_Atomic} \emph{plus} \texttt{volatile}? — atomic suffices for sharing; volatile is for MMIO.
  \item Aligned 32-bit read on ESP32 atomic? — in hardware yes; in C still a race $\to$ \texttt{atomic\_load\_explicit(relaxed)}, same instruction.
  \item Why \texttt{portMUX}? — interrupts-off does not stop the other core.
\end{enumerate}

\end{multicols}

\noindent{\footnotesize\color{sheetGrey}\textit{Related:} c-undefined-behavior-and-build (data races) · DMA cache coherency · \texttt{IRAM\_ATTR} · FreeRTOS queues / notifications · Swift actors}

\end{document}
