1import{j as e}from"./index-ab38db77.js";import{B as n}from"./BlogPost-920de3a8.js";import{E as o}from"./EnglishGlossaryLinker-e3ba6683.js";import{A as s}from"./AdPlaceholder-9ddf71ab.js";import{C as t}from"./CopyableCodeBlock-51c83e81.js";import{C as a}from"./cpu-1ebd5e89.js";import"./bookmarks-5e6598df.js";import"./TopicFollowButton-3468db39.js";import"./Layout-ae221617.js";import"./button-aceb6222.js";import"./index-38ea8514.js";import"./input-88a8d561.js";import"./bot-f88ad613.js";import"./createLucideIcon-350a23d4.js";import"./streakUtils-4796b5f1.js";import"./mail-d6f89251.js";import"./x-dbcb06f2.js";import"./book-open-d6b4bdaf.js";import"./studyVault-4c6970a9.js";import"./shield-check-89bf9027.js";import"./target-1c7ddc66.js";import"./authRedirect-9431f409.js";import"./search-b7443866.js";import"./users-e8160425.js";import"./chevron-left-3f1f1bc1.js";import"./trophy-93bb35a2.js";import"./code-2-e970a78e.js";import"./loader-2-1b0f84cc.js";import"./check-826c63d6.js";import"./plus-0d9cf947.js";import"./BlogNavbar-43011ed4.js";import"./sheet-bd5a9df9.js";import"./index-b1c1ae7a.js";import"./index-1622c2bb.js";import"./index-74502314.js";import"./index-5f1a4c86.js";import"./blogCategories-557cbee4.js";import"./arrow-left-78cc89c4.js";import"./user-plus-1a615a3a.js";import"./terminal-d75759e3.js";import"./BlogFooter-0dec2c82.js";import"./badge-a8d68d49.js";import"./AnnotationPanel-a8595e5f.js";import"./trash-2-455aecb7.js";import"./save-a3420a34.js";import"./chevron-down-7df805dd.js";import"./chevron-up-ae8aca77.js";import"./customFlashcards-a28c57f8.js";import"./blogImages-9118753f.js";import"./urdfCatalog-dc9e6e1e.js";import"./glossaryData-ad15d072.js";import"./glossaryLinking-0d3c2fc9.js";import"./glossaryRegex-d95d90d2.js";import"./GlossaryLinker-5310e348.js";import"./glossaryRender-03caa63e.js";import"./react-katex-c8b55abe.js";import"./index-08c7c3df.js";import"./katex-a5db0322.js";import"./continueLearning-b7dee3d3.js";import"./lastReadPosition-3d456d91.js";import"./achievementToast-3307f533.js";import"./proxy-4188af44.js";import"./star-3fdb0fdf.js";import"./eye-2c99b8de.js";import"./robobotClient-a6b235fb.js";import"./quizOptionBalancing-b8b82275.js";import"./EmbeddedQuizPractice-e895c045.js";import"./ExplanationRenderer-fd148189.js";/* empty css */import"./copy-61a61755.js";import"./prism-51c3acd6.js";import"./extends-4c19d496.js";import"./prism-9b2d42de.js";import"./index-3835e30c.js";import"./index.dom-42aa2b8a.js";import"./one-dark-2d2e2c01.js";import"./mistakeJournal-f029efa8.js";import"./message-circle-da57a6de.js";import"./x-circle-fdf008a7.js";import"./rotate-ccw-c11f09b7.js";import"./arrow-up-7e55e13f.js";import"./calendar-65001705.js";import"./maximize-2-92b2233e.js";import"./share-2-7aa2c950.js";import"./link-2-54f53ab8.js";import"./layers-3ca4a474.js";import"./index-d6888bdf.js";import"./index-128356f7.js";import"./katex-991ac7df.js";import"./englishGlossaryRegex-3e048418.js";import"./englishGlossaryData-b5c4c758.js";import"./heart-16f11fbb.js";const He=()=>{const r=e.jsxs("div",{className:"space-y-6 text-gray-700",children:[e.jsxs("p",{className:"text-xl text-gray-600 leading-relaxed",children:["A race condition is one of the nastiest bugs in systems programming: it only appears under load, disappears when you add a print statement, and can corrupt data silently for hours before the crash. Both C++ and Python are vulnerable, but for different reasons, and the fixes differ considerably. It's also the core argument behind teams reaching for ",e.jsx("a",{href:"/learn/blog/rust-for-robotics-2026",className:"text-primary underline",children:"Rust in mission-critical robot software"}),": memory and data-race safety enforced at compile time instead of caught at 2am."]}),e.jsx(s,{variant:"inline",className:"my-8"}),e.jsx("h2",{className:"text-2xl font-bold text-gray-900 mt-8 mb-4",children:"What Is a Race Condition?"}),e.jsx("p",{className:"text-gray-700",children:"A race condition occurs when two or more threads access shared mutable state concurrently and at least one of them writes. The final value depends on the interleaving of thread sc
1heduling. Which the OS controls non-deterministically."}),e.jsx("p",{className:"text-gray-700 mt-4",children:"The canonical example is incrementing a counter. It looks atomic at the source level but compiles down to three separate machine operations:"}),e.jsxs("ol",{className:"list-decimal pl-6 space-y-2 text-gray-700 mt-3",children:[e.jsxs("li",{children:[e.jsx("strong",{children:"Read"})," the current value from memory into a register"]}),e.jsxs("li",{children:[e.jsx("strong",{children:"Increment"})," the register"]}),e.jsxs("li",{children:[e.jsx("strong",{children:"Write"})," the new value back to memory"]})]}),e.jsx("p",{className:"text-gray-700 mt-4",children:"If Thread A reads the value (say, 42), then Thread B also reads 42, both increment to 43, and both write back 43, you have lost a count. Do this a million times and you can be off by hundreds of thousands."}),e.jsx("h2",{className:"text-2xl font-bold text-gray-900 mt-8 mb-4",children:"The Race in C++: Undefined Behavior"}),e.jsxs("p",{className:"text-gray-700",children:["C++ offers no guard rails. An unsynchronized data race is ",e.jsx("strong",{children:"undefined behavior"}),": the C++ standard lets the compiler assume it never happens. This means the optimizer can eliminate your check, reorder reads and writes across thread boundaries, or produce code that crashes non-deterministically."]}),e.jsx(t,{code:`#include <thread> 2#include <iostream> 3 4int counter = 0; // shared, unprotected 5 6void increment() { 7 for (int i = 0; i < 1'000'000; i++) 8 counter++; // UB: data race: three non-atomic ops 9} 10 11int main() { 12 std::thread t1(increment), t2(increment); 13 t1.join(); t2.join(); 14 std::cout << counter << "\\n"; // Almost never 2000000 15}`,language:"cpp"}),e.jsxs("p",{className:"text-gray-700 mt-4",children:["Run this with ",e.jsx("code",{children:"-fsanitize=thread"})," (ThreadSanitizer) and it will flag every access. The sanitizer is your best friend for catching these before they reach production."]}),e.jsx(s,{variant:"inline",className:"my-8"}),e.jsx("h2",{className:"text-2xl font-bold text-gray-900 mt-8 mb-4",children:"The Race in Python: GIL False Security"}),e.jsxs("p",{className:"text-gray-700",children:["Python's Global Interpreter Lock (GIL) is often cited as making threading safe. It does not. The GIL prevents two threads from executing Python bytecode simultaneously, but it releases between bytecode instructions, not between Python source lines. ",e.jsx("code",{children:"counter += 1"})," compiles to multiple bytecodes (LOAD, BINARY_ADD, STORE), and the GIL can switch threads between any of them."]}),e.jsx(t,{code:`import threading 16import dis 17 18counter = 0 19 20def increment(): 21 global counter 22 for _ in range(1_000_000): 23 counter += 1 # LOAD_GLOBAL, BINARY_OP, STORE_GLOBAL: GIL can switch between each 24 25# Disassemble to see the bytecode seams where GIL can switch 26dis.dis(increment)`,language:"python"}),e.jsx(t,{code:`import threading 27 28counter = 0 29 30def increment(): 31 global counter 32 for _ in range(1_000_000): 33 counter += 1 34 35t1 = threading.Thread(target=increment) 36t2 = threading.Thread(target=increment) 37t1.start(); t2.start() 38t1.join(); t2.join() 39 40print(counter) # Frequently != 2000000, even with the GIL`,language:"python"}),e.jsx("p",{className:"text-gray-700 mt-4",children:'The GIL reduces races compared to a fully unlocked interpreter, but it never eliminates them. The Python 3.13 "free-threaded" mode removes the GIL entirely. Which makes true parallelism possible but also makes every existing race condition instantly worse.'}),e.jsx(s,{variant:"inline",className:"my-8"}),e.jsx("h2",{className:"text-2xl font-bold text-gray-900 mt-8 mb-4",children:"Fixing Races in C++"}),e.jsxs("h3",{className:"text-xl font-semibold text-gray-900 mt-6 mb-3",children:["Option 1: ",e.jsx("code",{children:"std::atomic"})," (preferred for scalars)"]}),e.jsxs("p",{className:"text-gray-700",children:["For simple numeric types, ",e.jsx("code",{children:"std::atomic"})," is the right tool. It compiles down to a single CPU instruction (e.g., ",e.jsx("code",{children:"LOCK XADD"})," on x86) and avoids all lock overhead for the common case."]}),e.jsx(t,{code:`#include <atomic> 41#include <thread> 42#include <iostream> 43 44std::atomic<int> counter = 0; 45 46void increment() { 47 for (int i = 0; i < 1'000'000; i++) 48 counter.fetch_add(1, std::memory_order_relaxed); 49 // relaxed: no cross-thread ordering needed here, 50 // just atomicity of the add itself 51} 52 53int main() { 54 std::thread t1(increment), t2(increment); 55 t1.join(); t2.join(); 56 std::cout << counter << "\\n"; // Always 2000000 57}`,language:"cpp"}),e.jsxs("div",{className:"bg-blue-50 rounded-xl p-5 border border-blue-200 my-6 not-prose",children:[e.jsx("h4",{className:"text-blue-900 font-bold mb-2",children:"Memory Order Quick Reference"}),e.jsxs("ul",{className:"text-sm text-blue-800 space-y-1",children:[e.jsxs("li",{children:[e.jsx("code",{children:"memory_order_relaxed"}),", just atomicity, no ordering. Use for independent counters and statistics."]}),e.jsxs("li",{children:[e.jsx("code",{children:"memory_order_acquire"})," / ",e.jsx("code",{children:"memory_order_release"}),", producer-consumer handoff. Ensures the write is visible before the flag."]}),e.jsxs("li",{children:[e.jsx("code",{children:"memory_order_seq_cst"}),", full sequential consistency. Default, safest, slowest."]})]})]}),e.jsx(s,{variant:"sidebar",className:"my-8"}),e.jsxs("h3",{className:"text-xl font-semibold text-gray-900 mt-6 mb-3",children:["Option 2: ",e.jsx("code",{children:"std::mutex"})," with ",e.jsx("code",{children:"std::lock_guard"})]}),e.jsxs("p",{className:"text-gray-700",children:["For protecting a critical section that involves multiple variables or non-trivial logic, use a mutex. Always use RAII wrappers, never call ",e.jsx("code",{children:"unlock()"})," manually; an early return or exception will leave the mutex locked forever."]}),e.jsx(t,{code:`#include <mutex> 58#include <thread> 59#include <iostream> 60 61std::mutex mtx; 62int counter = 0; 63 64void increment() { 65 for (int i = 0; i < 1'000'000; i++) { 66 std::lock_guard<std::mutex> lock(mtx); // locks on construction, unlocks on destruction 67 counter++; 68 } 69} 70 71int main() { 72 std::thread t1(increment), t2(increment); 73 t1.join(); t2.join(); 74 std::cout << counter << "\\n"; // Always 2000000 75}`,language:"cpp"}),e.jsxs("h3",{className:"text-xl font-semibold text-gray-900 mt-6 mb-3",children:["Option 3: ",e.jsx("code",{children:"std::shared_mutex"})," for Read-Heavy Data"]}),e.jsx("p",{className:"text-gray-700",children:"When reads vastly outnumber writes (e.g., a configuration store or sensor cache), a shared mutex allows multiple concurrent readers but exclusive writers, giving much higher throughput than an exclusive mutex."}),e.jsx(t,{code:`#include <shared_mutex> 76#include <string> 77 78std::shared_mutex rw_mutex; 79std::string config_value; 80 81// Many threads can call this simultaneously 82std::string read_config() { 83 std::shared_lock<std::shared_mutex> lock(rw_mutex); 84 return config_value; 85} 86 87// Only one thread at a time can call this 88void update_config(const std::string& new_val) { 89 std::unique_lock<std::shared_mutex> lock(rw_mutex); 90 config_value = new_val; 91}`,language:"cpp"}),e.jsx(s,{variant:"banner",className:"my-10"}),e.jsx("h2",{className:"text-2xl font-bold text-gray-900 mt-8 mb-4",children:"Fixing Races in Python"}),e.jsxs("h3",{className:"text-xl font-semibold text-gray-900 mt-6 mb-3",children:["Option 1: ",e.jsx("code",{children:"threading.Lock"})]}),e.jsxs("p",{className:"text-gray-700",children:["The direct equivalent of ",e.jsx("code",{children:"std::mutex"}),". Always use it as a context manager (",e.jsx("code",{children:"with lock:"}),"), never call ",e.jsx("code",{children:"acquire()"}),"/",e.jsx("code",{children:"release()"})," manually for the same reason as C++."]}),e.jsx(t,{code:`import threading 92 93counter = 0 94lock = threading.Lock() 95 96def increment(): 97 global counter 98 for _ in range(1_000_000): 99 with lock: 100 counter += 1 101 102t1 = threading.Thread(target=increment) 103t2 = threading.Thread(target=increment) 104t1.start(); t2.start() 105t1.join(); t2.join() 106 107print(counter) # Always 2000000`,language:"python"}),e.jsxs("h3",{className:"text-xl font-semibold text-gray-900 mt-6 mb-3",children:["Option 2: ",e.jsx("code",{children:"threading.RLock"})," (Reentrant Lock)"]}),e.jsxs("p",{className:"text-gray-700",children:["Use ",e.jsx("code",{children:"RLock"})," when the same thread may acquire the lock more than once (e.g., a recursive function or a method that calls another method on the same object). A regular ",e.jsx("code",{children:"Lock"})," would deadlock; an ",e.jsx("code",{children:"RLock"})," counts acquisitions and only releases on the matching number of releases."]}),e.jsx(t,{code:`import threading 108 109class SafeCounter: 110 def __init__(self): 111 self._lock = threading.RLock() 112 self._value = 0 113 114 def increment(self): 115 with self._lock: 116 self._do_increment() # _do_increment also acquires the lock, safe with RLock 117 118 def _do_increment(self): 119 with self._lock: # second acquisition by same thread, ok with RLock 120 self._value += 1`,language:"python"}),e.jsx(s,{variant:"sidebar",className:"my-8"}),e.jsxs("h3",{className:"text-xl font-semibold text-gray-900 mt-6 mb-3",children:["Option 3: ",e.jsx("code",{children:"queue.Queue"})," (preferred for thread communication)"]}),e.jsxs("p",{className:"text-gray-700",children:["For producer-consumer patterns, avoid shared state entirely. ",e.jsx("code",{children:"queue.Queue"})," is thread-safe by design, internally it uses a lock, but it also provides blocking ",e.jsx("code",{children:"put()"}),"/",e.jsx("code",{children:"get()"})," semantics that are far easier to reason about than raw locks."]}),e.jsx(t,{code:`import threading 121import queue 122 123q = queue.Queue(maxsize=100) 124 125def producer(): 126 for i in range(50): 127 q.put(i) # blocks if queue is full 128 q.put(None) # sentinel 129 130def consumer(): 131 while True: 132 item = q.get() # blocks until item is available 133 if item is None: 134 break 135 print(f"Processing {item}") 136 q.task_done() 137 138t1 = threading.Thread(target=producer) 139t2 = threading.Thread(target=consumer) 140t1.start(); t2.start() 141t1.join(); t2.join()`,language:"python"}),e.jsxs("h3",{className:"text-xl font-semibold text-gray-900 mt-6 mb-3",children:["Option 4: For CPU-bound work, use ",e.jsx("code",{children:"multiprocessing"})]}),e.jsxs("p",{className:"text-gray-700",children:["The GIL means Python threads cannot parallelize CPU-bound work, only one thread runs Python bytecode at a time. For parallelism on a multi-core machine, use ",e.jsx("code",{children:"multiprocessing"}),". Each process has its own GIL and memory space; shared state requires explicit inter-process communication."]}),e.jsx(t,{code:`from multiprocessing import Process, Value, Lock 142 143def increment(counter, lock, n): 144 for _ in range(n): 145 with lock: 146 counter.value += 1 147 148if __name__ == "__main__": 149 counter = Value("i", 0) # shared memory integer 150 lock = Lock() # cross-process lock 151 152 p1 = Process(target=increment, args=(counter, lock, 1_000_000)) 153 p2 = Process(target=increment, args=(counter, lock, 1_000_000)) 154 p1.start(); p2.start() 155 p1.join(); p2.join() 156 157 print(counter.value) # Always 2000000`,language:"python"}),e.jsx(s,{variant:"inline",className:"my-8"}),e.jsx("h2",{className:"text-2xl font-bold text-gray-900 mt-8 mb-4",children:"Deadlock: The Other Thread Bug"}),e.jsx("p",{className:"text-gray-700",children:"Fixing race conditions with locks introduces the risk of deadlock, two threads each holding a lock the other needs, waiting forever. The classic pattern:"}),e.jsxs("div",{className:"grid md:grid-cols-2 gap-4 mt-4 not-prose",children:[e.jsxs("div",{className:"bg-gray-900 rounded-lg p-4 border border-gray-700",children:[e.jsx("h4",{className:"text-rose-400 font-bold mb-2",children:"C++ Deadlock"}),e.jsx(t,{code:`std::mutex A, B; 158 159// Thread 1: locks A then B 160std::lock_guard lkA(A); 161std::lock_guard lkB(B); // waits for B 162 163// Thread 2: locks B then A 164std::lock_guard lkB(B); 165std::lock_guard lkA(A); // waits for A: deadlock`,language:"cpp"})]}),e.jsxs("div",{className:"bg-gray-900 rounded-lg p-4 border border-gray-700",children:[e.jsx("h4",{className:"text-yellow-400 font-bold mb-2",children:"C++ Fix: std::scoped_lock"}),e.jsx(t,{code:`std::mutex A, B; 166 167// Acquires both in a deadlock-free order 168// regardless of which thread gets here first 169std::scoped_lock lk(A, B); 170 171// Now safe to use both A and B`,language:"cpp"})]})]}),e.jsxs("p",{className:"text-gray-700 mt-4",children:["In Python, the equivalent is to always acquire locks in the same order across all threads, or use ",e.jsx("code",{children:"threading.RLock"})," for reentrant cases. Python's ",e.jsx("code",{children:"threading"})," module does not have a ",e.jsx("code",{children:"scoped_lock"})," equivalent, so consistent ordering is the primary defense."]}),e.jsx("h2",{className:"text-2xl font-bold text-gray-900 mt-8 mb-4",children:"Side-by-Side Comparison"}),e.jsx("div",{className:"overflow-x-auto my-8 not-prose",children:e.jsxs("table",{className:"w-full border-collapse text-sm text-left",children:[e.jsx("thead",{children:e.jsxs("tr",{className:"bg-slate-900 text-white",children:[e.jsx("th",{className:"p-4 font-bold rounded-tl-lg",children:"Aspect"}),e.jsx("th",{className:"p-4 font-bold",children:"C++"}),e.jsx("th",{className:"p-4 font-bold rounded-tr-lg",children:"Python"})]})}),e.jsxs("tbody",{className:"divide-y divide-slate-200",children:[e.jsxs("tr",{className:"bg-white",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"Race detection"}),e.jsxs("td",{className:"p-4 text-slate-700",children:[e.jsx("code",{children:"-fsanitize=thread"})," (ThreadSanitizer)"]}),e.jsxs("td",{className:"p-4 text-slate-700",children:["No built-in; use careful testing + ",e.jsx("code",{children:"dis"})]})]}),e.jsxs("tr",{className:"bg-slate-50",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"UB on data race"}),e.jsx("td",{className:"p-4 text-rose-700 font-medium",children:"Yes, optimizer can do anything"}),e.jsx("td",{className:"p-4 text-slate-700",children:"No UB, result is just wrong, not undefined"})]}),e.jsxs("tr",{className:"bg-white",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"Atomic scalars"}),e.jsx("td",{className:"p-4 text-slate-700",children:e.jsx("code",{children:"std::atomic<T>"})}),e.jsxs("td",{className:"p-4 text-slate-700",children:["No direct equivalent; use ",e.jsx("code",{children:"Lock"})]})]}),e.jsxs("tr",{className:"bg-slate-50",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"Mutex"}),e.jsxs("td",{className:"p-4 text-slate-700",children:[e.jsx("code",{children:"std::mutex"})," + ",e.jsx("code",{children:"lock_guard"})]}),e.jsxs("td",{className:"p-4 text-slate-700",children:[e.jsx("code",{children:"threading.Lock"})," + ",e.jsx("code",{children:"with"})]})]}),e.jsxs("tr",{className:"bg-white",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"Readers-writers lock"}),e.jsx("td",{className:"p-4 text-slate-700",children:e.jsx("code",{children:"std::shared_mutex"})}),e.jsxs("td",{className:"p-4 text-slate-700",children:[e.jsx("code",{children:"threading.RLock"})," (not truly shared-read)"]})]}),e.jsxs("tr",{className:"bg-slate-50",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"CPU parallelism"}),e.jsx("td",{className:"p-4 text-emerald-700 font-medium",children:"True parallelism with threads"}),e.jsxs("td",{className:"p-4 text-slate-700",children:["Need ",e.jsx("code",{children:"multiprocessing"})," (GIL blocks threads)"]})]}),e.jsxs("tr",{className:"bg-white",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"Safe queue"}),e.jsxs("td",{className:"p-4 text-slate-700",children:[e.jsx("code",{children:"std::queue"})," + mutex, or concurrent libs"]}),e.jsxs("td",{className:"p-4 text-emerald-700 font-medium",children:[e.jsx("code",{children:"queue.Queue"})," (built-in thread safety)"]})]}),e.jsxs("tr",{className:"bg-slate-50",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"Deadlock prevention"}),e.jsxs("td",{className:"p-4 text-slate-700",children:[e.jsx("code",{children:"std::scoped_lock"})," for multi-lock acquire"]}),e.jsx("td",{className:"p-4 text-slate-700",children:"Consistent lock ordering; no stdlib helper"})]})]})]})}
171),e.jsx(s,{variant:"banner",className:"my-10"}),e.jsx("h2",{className:"text-2xl font-bold text-gray-900 mt-8 mb-4",children:"Race Conditions in ROS 2"}),e.jsx("p",{className:"text-gray-700",children:"In robotics specifically, race conditions show up most in ROS 2 nodes that use callbacks. When a node subscribes to two topics and the callback for each modifies shared state, those callbacks can run on different executor threads concurrently."}),e.jsxs("div",{className:"grid md:grid-cols-2 gap-4 mt-4 not-prose",children:[e.jsxs("div",{className:"bg-gray-900 rounded-lg p-4 border border-gray-700",children:[e.jsx("h4",{className:"text-rose-400 font-bold mb-2",children:"C++ rclcpp: Racy"}),e.jsx(t,{code:`class MyNode : public rclcpp::Node { 172 double latest_odom_x_; // written from odom CB 173 double latest_scan_x_; // written from scan CB 174 175 void odom_cb(const nav_msgs::msg::Odometry::SharedPtr msg) { 176 latest_odom_x_ = msg->pose.pose.position.x; // race 177 } 178 void scan_cb(const sensor_msgs::msg::LaserScan::SharedPtr msg) { 179 use(latest_odom_x_); // reads while odom_cb may write 180 } 181};`,language:"cpp"})]}),e.jsxs("div",{className:"bg-gray-900 rounded-lg p-4 border border-gray-700",children:[e.jsx("h4",{className:"text-emerald-400 font-bold mb-2",children:"C++ rclcpp: Safe"}),e.jsx(t,{code:`class MyNode : public rclcpp::Node { 182 std::mutex state_mutex_; 183 double latest_odom_x_; 184 185 void odom_cb(const nav_msgs::msg::Odometry::SharedPtr msg) { 186 std::lock_guard<std::mutex> lock(state_mutex_); 187 latest_odom_x_ = msg->pose.pose.position.x; 188 } 189 void scan_cb(const sensor_msgs::msg::LaserScan::SharedPtr msg) { 190 std::lock_guard<std::mutex> lock(state_mutex_); 191 use(latest_odom_x_); 192 } 193};`,language:"cpp"})]})]}),e.jsxs("p",{className:"text-gray-700 mt-4",children:["Alternatively, use a ",e.jsx("strong",{children:"Single-Threaded Executor"})," in ROS 2, all callbacks run serially and there are no races. Switch to a ",e.jsx("strong",{children:"Multi-Threaded Executor"})," only when you need true concurrency and are willing to add locking."]}),e.jsx("h2",{className:"text-2xl font-bold text-gray-900 mt-8 mb-4",children:"The 10-Second Diagnostic Checklist"}),e.jsx("div",{className:"bg-slate-50 p-6 rounded-xl border border-slate-200 my-6 not-prose",children:e.jsxs("ul",{className:"space-y-3 text-sm text-slate-700",children:[e.jsxs("li",{className:"flex items-start gap-2",children:[e.jsx("span",{className:"text-emerald-600 font-bold mt-0.5",children:e.jsx(a,{className:"w-4 h-4 inline text-green-600"})}),e.jsx("span",{children:"Is the bug intermittent and load-dependent? Likely a race."})]}),e.jsxs("li",{className:"flex items-start gap-2",children:[e.jsx("span",{className:"text-emerald-600 font-bold mt-0.5",children:e.jsx(a,{className:"w-4 h-4 inline text-green-600"})}),e.jsxs("span",{children:["Does adding a ",e.jsx("code",{children:"print()"})," or ",e.jsx("code",{children:"sleep()"})," make the bug disappear? Race condition (Heisenbug)."]})]}),e.jsxs("li",{className:"flex items-start gap-2",children:[e.jsx("span",{className:"text-emerald-600 font-bold mt-0.5",children:e.jsx(a,{className:"w-4 h-4 inline text-green-600"})}),e.jsx("span",{children:"Is there any shared mutable variable accessed from more than one thread? Audit every access."})]}),e.jsxs("li",{className:"flex items-start gap-2",children:[e.jsx("span",{className:"text-emerald-600 font-bold mt-0.5",children:e.jsx(a,{className:"w-4 h-4 inline text-green-600"})}),e.jsxs("span",{children:["In C++: run with ",e.jsx("code",{children:"-fsanitize=thread"})," immediately."]})]}),e.jsxs("li",{className:"flex items-start gap-2",children:[e.jsx("span",{className:"text-emerald-600 font-bold mt-0.5",children:e.jsx(a,{className:"w-4 h-4 inline text-green-600"})}),e.jsxs("span",{children:["In Python: check if you're mutating a global or object attribute from a thread without a ",e.jsx("code",{children:"Lock"}),"."]})]}),e.jsxs("li",{className:"flex items-start gap-2",children:[e.jsx("span",{className:"text-emerald-600 font-bold mt-0.5",children:e.jsx(a,{className:"w-4 h-4 inline text-green-600"})}),e.jsx("span",{children:"In ROS 2: check executor type and whether callbacks share state."})]})]})}),e.jsx(s,{variant:"inline",className:"my-8"}),e.jsx("h2",{className:"text-2xl font-bold text-gray-900 mt-8 mb-4",children:"GPU Parallelism: A Completely Different Model"}),e.jsxs("p",{className:"text-gray-700",children:["When people talk about threading, they mean CPU threads, typically 8 to 128 on a modern workstation. A GPU flips this: an Nvidia RTX 4090 has ",e.jsx("strong",{children:"16,384 CUDA
193cores"}),". The programming model, the race conditions, and the decision of when to use it are fundamentally different from anything above."]}),e.jsx("h3",{className:"text-xl font-semibold text-gray-900 mt-6 mb-3",children:"CPU Threads vs GPU Threads: The Mental Model"}),e.jsx("div",{className:"overflow-x-auto my-6 not-prose",children:e.jsxs("table",{className:"w-full border-collapse text-sm text-left",children:[e.jsx("thead",{children:e.jsxs("tr",{className:"bg-slate-900 text-white",children:[e.jsx("th",{className:"p-4 font-bold rounded-tl-lg",children:"Property"}),e.jsx("th",{className:"p-4 font-bold",children:"CPU Thread"}),e.jsx("th",{className:"p-4 font-bold rounded-tr-lg",children:"GPU Thread (CUDA)"})]})}),e.jsxs("tbody",{className:"divide-y divide-slate-200",children:[e.jsxs("tr",{className:"bg-white",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"Count"}),e.jsx("td",{className:"p-4 text-slate-700",children:"4â128 threads typical"}),e.jsx("td",{className:"p-4 text-slate-700",children:"Thousands to tens of thousands"})]}),e.jsxs("tr",{className:"bg-slate-50",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"Execution model"}),e.jsx("td",{className:"p-4 text-slate-700",children:"MIMD, each thread runs independent instructions"}),e.jsx("td",{className:"p-4 text-slate-700",children:"SIMT, warps of 32 threads execute the same instruction"})]}),e.jsxs("tr",{className:"bg-white",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"Thread cost"}),e.jsx("td",{className:"p-4 text-slate-700",children:"Heavy: MB of stack, OS scheduling"}),e.jsx("td",{className:"p-4 text-slate-700",children:"Ultralight, context switch is free (hardware register file)"})]}),e.jsxs("tr",{className:"bg-slate-50",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"Memory"}),e.jsx("td",{className:"p-4 text-slate-700",children:"Shared system RAM, cache coherent"}),e.jsx("td",{className:"p-4 text-slate-700",children:"Separate VRAM, data must be copied to/from GPU explicitly"})]}),e.jsxs("tr",{className:"bg-white",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"Best for"}),e.jsx("td",{className:"p-4 text-slate-700",children:"Task parallelism, I/O, latency-sensitive work"}),e.jsx("td",{className:"p-4 text-slate-700",children:"Data parallelism, same op on millions of elements"})]}),e.jsxs("tr",{className:"bg-slate-50",children:[e.jsx("td",{className:"p-4 font-semibold text-slate-900",children:"Branching"}),e.jsx("td",{className:"p-4 text-slate-700",children:"Each thread branches independently"}),e.jsx("td",{className:"p-4 text-slate-700",children:"Warp divergence, branches inside a warp serialize (slow)"})]})]})]})}),e.jsx("h3",{className:"text-xl font-semibold text-gray-900 mt-6 mb-3",children:"When GPU Parallelism Makes Sense"}),e.jsxs("p",{className:"text-gray-700",children:["The GPU wins when you have ",e.jsx("strong",{children:"embarrassingly parallel"})," work: the same operation applied independently to a huge dataset with minimal branching. The GPU loses, badly. When work is sequential, branchy, or when the data transfer overhead swamps the compute gain."]}),e.jsxs("div",{className:"bg-emerald-50 rounded-xl p-5 border border-emerald-200 my-6 not-prose",children:[e.jsx("h4",{className:"text-emerald-900 font-bold mb-3",children:"Use the GPU when:"}),e.jsxs("ul",{className:"text-sm text-emerald-800 space-y-2",children:[e.jsxs("li",{children:[e.jsx("strong",{children:"Neural network inference"}),", matrix multiplications over millions of weights. PyTorch/TensorFlow are GPU-first for this reason."]}),e.jsxs("li",{children:[e.jsx("strong",{children:"Point cloud processing"}),", transforming, filtering, or voxelizing millions of LiDAR points simultaneously. CUDA-based PCL operations, Open3D GPU backend."]}),e.jsxs("li",{children:[e.jsx("strong",{children:"Image/video batch processing"}),", convolutions, resizing, color space conversion on full camera frames at 30â120 fps."]}),e.jsxs("li",{children:[e.jsx("strong",{children:"Physics simulation"}),": IsaacSim, MuJoCo MJX, and Genesis all run contact physics across thousands of parallel environments on the GPU."]}),e.jsxs("li",{children:[e.jsx("strong",{children:"Large matrix math"}),": Jacobians, Hessians, covariance matrices in optimization where N is in the thousands."]})]})]}),e.jsxs("div",{className:"bg-rose-50 rounded-xl p-5 border border-rose-200 my-6 not-prose",children:[e.jsx("h4",{className:"text-rose-900 font-bold mb-3",children:"Stay on CPU threads when:"}),e.jsxs("ul",{className:"text-sm text-rose-800 space-y-2",children:[e.jsxs("li",{children:[e.jsx("strong",{children:"Data is small"}
193),", copying 1 KB to VRAM and back costs more than just doing the work on CPU."]}),e.jsxs("li",{children:[e.jsx("strong",{children:"Work is sequential"}),": a state machine, a behavior tree, a control loop with data-dependent branches cannot be parallelized across GPU cores."]}),e.jsxs("li",{children:[e.jsx("strong",{children:"Latency is critical"}),": GPU kernel launch overhead is microseconds; a 1 kHz control loop cannot afford that."]}),e.jsxs("li",{children:[e.jsx("strong",{children:"I/O bound"}),", waiting on a serial port, a network packet, or a ROS topic. CPU threads + async I/O is the right model."]}),e.jsxs("li",{children:[e.jsx("strong",{children:"Logic is diverse"}),", different threads doing entirely different tasks (planner + controller + logger) is task parallelism, not data parallelism."]})]})]}),e.jsx(s,{variant:"sidebar",className:"my-8"}),e.jsx("h3",{className:"text-xl font-semibold text-gray-900 mt-6 mb-3",children:"Race Conditions on the GPU"}),e.jsx("p",{className:"text-gray-700",children:"GPUs have race conditions too, they just look different. When multiple CUDA threads write to the same memory address, you need atomic operations just like on the CPU."}),e.jsx(t,{code:`// CUDA C++: race condition on global counter 194__global__ void bad_increment(int* counter) { 195 (*counter)++; // read-modify-write: race between all threads 196} 197 198// Fixed with atomicAdd 199__global__ void safe_increment(int* counter) { 200 atomicAdd(counter, 1); // GPU-native atomic, works across all threads 201} 202 203// Launch 1 million threads: both produce "1000000" only with atomicAdd 204safe_increment<<<1000, 1000>>>(d_counter);`,language:"cpp"}),e.jsxs("p",{className:"text-gray-700 mt-4",children:["Within a warp (32 threads), you can use ",e.jsx("strong",{children:"warp-level primitives"})," like ",e.jsx("code",{children:"__reduce_add_sync"})," to sum values across the warp without atomics, faster because all 32 threads are synchronized by definition. Across thread blocks, you need ",e.jsx("code",{children:"atomicAdd"})," or a reduction pattern."]}),e.jsx("p",{className:"text-gray-700 mt-4",children:"In Python, GPU race conditions are largely hidden behind library abstractions. PyTorch operations on tensors are internally thread-safe at the kernel level. You can introduce races by mixing CPU tensors and CUDA tensors without synchronizing streams:"}),e.jsx(t,{code:`import torch 205 206# Two CUDA streams, operations may overlap 207stream1 = torch.cuda.Stream() 208stream2 = torch.cuda.Stream() 209 210tensor = torch.zeros(1000, device="cuda") 211 212with torch.cuda.stream(stream1): 213 tensor.add_(1.0) # write on stream1 214 215with torch.cuda.stream(stream2): 216 result = tensor.sum() # read on stream2, may see partial write 217 218# Fix: synchronize before reading across streams 219stream1.synchronize() 220with torch.cuda.stream(stream2): 221 result = tensor.sum() # safe`,language:"python"}),e.jsx("h3",{className:"text-xl font-semibold text-gray-900 mt-6 mb-3",children:"The CPU-GPU Pipeline in Robotics"}),e.jsx("p",{className:"text-gray-700",children:"Real robotics systems use both. The pattern is consistent across most production stacks:"}),e.jsx("div",{className:"bg-slate-900 rounded-xl p-6 my-6 not-prose font-mono text-sm",children:e.jsxs("div",{className:"space-y-2",children:[e.jsx("div",{className:"text-slate-400",children:"// CPU threads handle I/O, control, and orchestration"}),e.jsx("div",{className:"text-blue-300",children:"Thread 1: ROS 2 subscriber â receives camera frame"}),e.jsx("div",{className:"text-blue-300",children:"Thread 2: ROS 2 subscriber â receives LiDAR scan"}),e.jsx("div",{className:"text-yellow-300",children:"Thread 3: Control loop @ 500 Hz â motor commands (CPU, deterministic)"}),e.jsx("div",{className:"text-slate-500 mt-3",children:"// GPU handles the heavy compute, asynchronously"}),e.jsx("div",{className:"text-emerald-300",children:"CUDA stream A: object detection (YOLO) on camera frame"}),e.jsx("div",{className:"text-emerald-300",children:"CUDA stream B: point cloud segmentation on LiDAR scan"}),e.jsx("div",{className:"text-slate-500 mt-3",children:"// CPU thread collects GPU results and feeds planner"}),e.jsx("div",{className:"text-blue-300",children:"Thread 4: waits on GPU results â updates costmap â triggers re-plan"})]})}),e.jsxs("p",{className:"text-gray-700",children:["The key insight: the GPU is a ",e.jsx("strong",{children:"co-processor"}),", not a replacement for CPU threads. You still need CPU threads for anything latency-sensitive, sequential, or I/O-bound. The GPU handles the bulk compute work in parallel, and CPU threads coordinate the results."]}),e.jsx("h3",{className:"text-xl font-semibold text-gray-900 mt-6 mb-3",children:"The GIL and GPU: Python's Accidental Win"}),e.jsxs("p",{className:"text-gray-700",children:["Python's GIL blocks CPU parallelism, but it has ",e.jsx("strong",{children:"no effect on GPU compute"}),". A PyTorch CUDA kernel runs entirely on the GPU hardware: the Python interpreter is not involved during the kernel execution. This means Python actually achieves true parallelism for GPU workloads despite the GIL, which is a large reason why Python dominates ML even though it's single-threaded on the CPU."]}),e.jsx(t,{code:`import torch 222import threading 223 224# Both threads launch GPU kernels concurrently 225# GIL is released during CUDA kernel execution 226# True GPU parallelism happens even in Python 227def gpu_work(data): 228 result = torch.nn.functional.conv2d(data, weight) # runs on GPU, GIL released 229 return result 230 231t1 = threading.Thread(target=gpu_work, args=(batch1,)) 232t2 = threading.Thread(target=gpu_work, args=(batch2,)) 233t1.start(); t2.start() # both CUDA kernels run concurrently on the GPU 234t1.join(); t2.join()`,language:"python"}),e.jsx(s,{variant:"inline",className:"my-8"}),e.jsx("h2",{className:"text-2xl font-bold text-gray-900 mt-8 mb-4",children:"Summary"}),e.jsxs("p",{className:"text-gray-700",children:["C++ and Python race conditions arise from the same root cause, shared mutable state across threads, but differ in severity. C++ races are undefined behavior; Python races produce wrong values but no UB, and the GIL softens but doesn't eliminate them. Both require explicit synchronization: ",e.jsx("code",{children:"std::atomic"})," or ",e.jsx("code",{children:"std::mutex"})," in C++, ",e.jsx("code",{children:"threading.Lock"})," or ",e.jsx("code",{children:"queue.Queue"})," in Python. GPU parallelism is a separate axis entirely: use it when work is embarrassingly parallel and data is large, stay on CPU threads for sequential logic, latency-critical control loops, and I/O. In robotics, the strongest systems layer all three: CPU threads for orchestration and control, GPU for bulk inference and point cloud processing, and explicit synchronization at every shared boundary."]})]});return e.jsx(n,{title:"C++ vs Python Threading: Race Conditions, Mutexes, and the GIL Explained",seoTitle:"C++ vs Python Threading",seoDescription:"Deep dive into race conditions in C++ and Python. Covers the GIL fal
234se-security problem, std::atomic, std::mutex, threading.Lock, queue.Queue, deadlocks,.",description:"Deep dive into race conditions in C++ and Python. Covers the GIL false-security problem, std::atomic, std::mutex, threading.Lock, queue.Queue, deadlocks, and ROS 2 callback races.",date:"Jun 13, 2026",author:"Darsh Menon",readTime:"13 min",image:"/assets/images/generated/robot_arm_python.webp",category:"Software & Systems",keywords:["race condition C++","race condition Python","Python GIL threading","std::atomic","std::mutex","threading.Lock","deadlock C++","ROS 2 multithreading","concurrent programming robotics"],canonicalUrl:"https://robocloud-dashboard.vercel.app/learn/blog/cpp-python-threading-race-conditions",lastUpdated:"Jun 13, 2026",difficulty:"Advanced",wordCount:2800,faqs:[{question:"Does Python's GIL prevent race conditions?",answer:"No. The GIL prevents two threads from running Python bytecode simultaneously, but it releases between bytecode instructions. A single Python statement like counter += 1 compiles to multiple bytecodes, so threads can still interleave and produce incorrect results. The GIL reduces (but does not eliminate) race conditions."},{question:"When should I use std::atomic vs std::mutex in C++?",answer:"Use std::atomic for simple scalar operations like counters, flags, and pointers where you need one operation to be atomic. Use std::mutex when protecting a critical section involving multiple variables or non-trivial logic. Std::atomic is typically lock-free and faster; std::mutex is more flexible."},{question:"How do race conditions show up in ROS 2 nodes?",answer:"In ROS 2, the Multi-Threaded Executor runs callbacks on a thread pool. If two callbacks (e.g., an odometry callback and a laser scan callback) both access shared member variables, a race condition exists. Fix it with std::mutex in C++ or threading.Lock in Python, or switch to a Single-Threaded Executor to serialize all callbacks."}],relatedLinks:[{title:"Embedded Systems (Interactive Lessons)",link:"/robotics-concepts/embedded-systems",category:"Interactive",description:"Real-time threading, RTOS scheduling, and shared-memory concurrency in embedded robot systems."},{title:"ROS2 Advanced (Interactive Lessons)",link:"/robotics-concepts/ros2-advanced",category:"Interactive",description:"ROS 2 multi-threaded executors, callback groups, and thread-safe node patterns."},{title:"C++ for Robotics",link:"/learn/blog/cpp-robotics",category:"Blog",description:"Modern C++17/20 patterns for high-performance robot control and perception pipelines."},{title:"Python vs C++ for Robotics",link:"/learn/blog/python-vs-cpp-robotics",category:"Blog",description:"When to use Python for flexibility vs C++ for real-time determinism in robot nodes."},{title:"Production ROS 2 Architecture",link:"/learn/blog/production-ros2-architecture",category:"Blog",description:"Thread-safe ROS 2 node design patterns for industrial production robot deployments."},{title:"SSH Security for Robotics: Hardening, Tunneling & WireGuard VPN",link:"/learn/blog/ssh-robotics-security",category:"Software & Systems",description:"Harden SSH on Jetson, Raspberry Pi, and robot computers: disable password auth, configure sshd_config, tunnel ROS 2 traffic with SSH port forwardingâ¦"}],content:e.jsx(o,{children:r})})};export{He as default};
Line numbers count LF bytes from the start of the resource, as the search results do. Vendor segments are library code the classifier recognised; they are stored but not indexed. Bytes are shown as Latin1 characters, one per byte.