[{"data":1,"prerenderedAt":95},["ShallowReactive",2],{"notebook-gpu-synchronization-changed-the-result":3},{"id":4,"title":5,"algorithms":6,"author":7,"body":8,"canonical_url":62,"description":54,"experiment_id":63,"extension":64,"kind":65,"languages":66,"maturity":68,"meta":69,"modified":70,"navigation":71,"path":72,"projects":73,"published":75,"relations":76,"repository_commit":80,"seo":81,"slug":82,"stem":83,"summary":84,"tags":85,"technologies":88,"topics":91,"__hash__":94},"notebook\u002Fnotebook\u002Fgpu-synchronization-changed-the-result.md","GPU synchronization changed the ranking",[],"Lucas Aruodore Adomi",{"type":9,"value":10,"toc":53},"minimark",[11,16,25,29,32,36,46,50],[12,13,15],"h2",{"id":14},"context","Context",[17,18,19,20,24],"p",{},"Two inference paths were timed with ",[21,22,23],"code",{},"perf_counter_ns",". The supposedly faster path launched more asynchronous work.",[12,26,28],{"id":27},"observation","Observation",[17,30,31],{},"Without device synchronization, the benchmark mostly measured host dispatch. Synchronizing before and after each measured region reversed the ranking.",[12,33,35],{"id":34},"questions","Questions",[37,38,39,43],"ul",{},[40,41,42],"li",{},"Should end-to-end latency and device-only latency appear as separate metrics?",[40,44,45],{},"Which synchronization overhead belongs to the real request path?",[12,47,49],{"id":48},"next-steps","Next steps",[17,51,52],{},"Repeat with device events, preserve all samples, and document the measured boundary in the result manifest.",{"title":54,"searchDepth":55,"depth":55,"links":56},"",3,[57,59,60,61],{"id":14,"depth":58,"text":15},2,{"id":27,"depth":58,"text":28},{"id":34,"depth":58,"text":35},{"id":48,"depth":58,"text":49},"https:\u002F\u002Fjournal.aruodore.com\u002Fnotebook\u002Fgpu-synchronization-changed-the-result","exp-2026-07-18-gpu-03","md","benchmark-observation",[67],"python","verified",{},"2026-07-19",true,"\u002Fnotebook\u002Fgpu-synchronization-changed-the-result",[74],"benchmark-protocol","2026-07-18",[77],{"type":78,"target":79},"develops","blog:2026:reproducible-gpu-benchmarks",null,{"title":5,"description":54},"gpu-synchronization-changed-the-result","notebook\u002Fgpu-synchronization-changed-the-result","Adding synchronization reversed an apparent latency advantage between two inference paths.",[86,87],"gpu","benchmarking",[89,90],"pytorch","cuda",[92,93],"scientific-computing","ai-engineering","R4a44xKfaWSpy6uDVV6utN2f2ApuET3LQ6GTDvFj2Jo",1785815463963]