From 35d035ce9e1920b69c734388eca81114b936a142 Mon Sep 17 00:00:00 2001 From: Chao Wang <26245345+ChaoWao@users.noreply.github.com> Date: Sun, 13 Sep 2026 07:21:43 -0700 Subject: [PATCH] Refactor: name the per-run H2D copy-in instead of overloading "staged" "stage" carries four unrelated meanings in this repo and is defined nowhere. Two of them meet inside one file: host_build_graph's runtime_maker.cpp reports `staged=%d` for caller tensors it copied to the device, and a hundred lines away holds a `staging` block that is the host scratch the Graph Definitions are assembled in. The scheduler adds a third (`staged_core_mask`, cores held ready before release) and `drain_stage` a fourth (a step in a sequence). A reader cannot tell which is meant without following the code, and error text shown to users inherits the ambiguity. This renames one sense: the per-run copy of a caller tensor into device memory. The other senses keep the word -- a staging buffer, a pipeline stage, a staged IPC frame and the args-dump capture stage are all ordinary uses of it, and the sense renamed here had the weakest claim, since the device buffer it produces is the live one the kernel reads rather than somewhere bytes pass through. The replacement reuses names the repo already has rather than coining any. A tensor on this path is `child_memory=False`, i.e. `AddressSpace::HOST`, so it is a host-memory tensor; the operation is an H2D copy-in, and `h2d` is already how the sibling bind phase spells it (`BindArenaH2d`, `arena_h2d`). So `bind.args` now reports `h2d=%d bytes=%llu`, and tensormap_and_ringbuffer's `stage_device_args` becomes `copy_in_device_args`. Both runtimes carry this sense, so both move, and each arch sibling moves with its pair. Nothing parses the attribute string programmatically -- `h2d=` replaces `staged=` in logs and in docs/dfx/hbg-bind-phases.md only. The remaining occurrences were classified by reading them rather than by pattern: a regex over the obvious spellings missed tmr's "Failed to stage tensor" diagnostic, SCALAR_DATA_ACCESS.md's "used to stage", and the prose in task-flow.md and buffer-abi.md. `docs/investigations/` is left as written, since those are dated records of past measurements. docs/testing.md also said the default copies every tensor in on every round, which overstates it: pure OUT buffers skip the H2D. No behaviour change. --- docs/buffer-abi.md | 8 +++--- docs/dfx/hbg-bind-phases.md | 20 +++++++------- docs/dfx/l2-timing.md | 4 +-- docs/task-flow.md | 10 +++---- docs/testing.md | 10 +++---- .../benchmark_bgemm/test_benchmark_bgemm.py | 2 +- .../benchmark_bgemm/test_benchmark_bgemm.py | 2 +- .../kernels/aiv/comb_sinkhorn.cpp | 2 +- .../kernels/aiv/comb_sinkhorn_0.cpp | 2 +- .../kernels/aiv/comb_sinkhorn_1.cpp | 2 +- .../kernels/aiv/comb_sinkhorn_2.cpp | 2 +- .../kernels/aiv/comb_sinkhorn_3.cpp | 2 +- .../kernels/aiv/comb_sinkhorn_4.cpp | 2 +- .../kernels/aiv/comb_sinkhorn_5.cpp | 2 +- .../kernels/aiv/comb_sinkhorn_6.cpp | 2 +- .../kernels/aiv/comb_sinkhorn_7.cpp | 2 +- .../kernels/aiv/comb_sinkhorn_8.cpp | 2 +- .../kernels/aiv/split_pre_post.cpp | 2 +- .../kernels/aiv/split_pre_post_0.cpp | 2 +- .../kernels/aiv/split_pre_post_1.cpp | 2 +- .../kernels/aiv/split_pre_post_2.cpp | 2 +- .../kernels/aiv/split_pre_post_3.cpp | 2 +- .../kernels/aiv/split_pre_post_4.cpp | 2 +- .../kernels/aiv/split_pre_post_5.cpp | 2 +- .../kernels/aiv/split_pre_post_6.cpp | 2 +- .../kernels/aiv/split_pre_post_7.cpp | 2 +- .../kernels/aiv/split_pre_post_8.cpp | 2 +- .../orchestration/prefetch_async_orch.cpp | 2 +- .../bgemm/test_bgemm.py | 2 +- simpler_setup/scene_test.py | 16 ++++++------ .../docs/SCALAR_DATA_ACCESS.md | 12 ++++----- .../host_build_graph/host/runtime_maker.cpp | 26 +++++++++---------- .../host/runtime_maker.cpp | 18 ++++++------- .../docs/SCALAR_DATA_ACCESS.md | 12 ++++----- .../host_build_graph/host/runtime_maker.cpp | 26 +++++++++---------- .../host/runtime_maker.cpp | 18 ++++++------- .../host/host_tensor_access.cpp | 4 +-- .../host_build_graph/host/runtime_core.cpp | 12 ++++----- .../host_build_graph/host_tensor_access.h | 14 +++++----- .../a2a3/host_build_graph/bgemm/test_bgemm.py | 4 +-- tests/ut/cpp/a2a3/test_hbg_tensor_access.cpp | 7 +++-- tests/ut/py/test_scene_test_child_memory.py | 4 +-- 42 files changed, 134 insertions(+), 139 deletions(-) diff --git a/docs/buffer-abi.md b/docs/buffer-abi.md index 78d66de544..d36c535942 100644 --- a/docs/buffer-abi.md +++ b/docs/buffer-abi.md @@ -68,7 +68,7 @@ collapse that into a single type were considered and dropped. ### Rejected: merge `Tensor` into `ChipTensor` -Drop `buffer.addr`, add the buffer descriptor, and have the H2D staging step +Drop `buffer.addr`, add the buffer descriptor, and have the H2D copy-in step rewrite the backend tag and body (and mint a fresh identity for the device copy). That is self-consistent, but it charges the device for host-side fields: @@ -132,8 +132,8 @@ task. > OverlapMap by it rather than by `buffer.addr` would make two views of one > backing bucket together by construction. That needs 32 B — which fits the > existing `_pad_cl2[36]` at `sizeof == 128`, i.e. **without** merging anything -> else. If it is ever done, the H2D staging step must mint a *new* identity for -> each staged copy, because the device buffer is a distinct backing from the host +> else. If it is ever done, the H2D copy-in step must mint a *new* identity for +> each copy, because the device buffer is a distinct backing from the host > one it was copied from. ### Rejected: keep the wire type transport-only, use `ChipTensor` in the L3 orch @@ -337,7 +337,7 @@ values are final, at submit: named as an output and then silently losing every write in the child. - **No overlapping writes within one task.** Two arguments of one task that name intersecting bytes of the same backing are rejected: they belong to one node, - so there is no order between them to express, and a device-staged copy of a + so there is no order between them to express, and a device-side copy of a host backing does not even alias on the device for the L2 overlap map to notice. Disjoint slices of one buffer stay legal — that is what `byte_offset` is for, and this check runs the same two-stage comparison dependency diff --git a/docs/dfx/hbg-bind-phases.md b/docs/dfx/hbg-bind-phases.md index 02becdb74c..48504e8c5c 100644 --- a/docs/dfx/hbg-bind-phases.md +++ b/docs/dfx/hbg-bind-phases.md @@ -1,7 +1,7 @@ # The `host_build_graph` bind phases `host_build_graph` builds the whole task graph on the host before the device -executes anything, so the host-side **`bind` stage** — argument staging, +executes anything, so the host-side **`bind` stage** — argument copy-in, orchestration, the Graph Definition, and every H2D copy — is a first-class cost. `bind` is the `chip.run.bind` `[STRACE]` span both runtimes emit; only this one subdivides it into **segments**, one `chip.run.bind.` span each. This @@ -26,7 +26,7 @@ the `chip.run.bind` span: | Segment | What it covers | | ------- | -------------- | -| `args` | staging readable caller tensors H2D and exposing their existing host buffers to orchestration; pure outputs skip both | +| `args` | copying readable caller tensors in H2D and exposing their existing host buffers to orchestration; pure outputs skip both | | `arena_build`, `static_arena`, `gm_heap`, `shared_mem`, `runtime_init` | arena layout, GM heap and shared-memory bring-up | | `host_orch` | **all** orchestration: every task submitted, every in-graph task recorded, the Definition built | | `graph_upload` | one H2D of the block holding every Definition object, and binding each Graph task to the one with its key. The recorders built the objects in that block's host staging during `host_orch`, so this segment writes their headers and copies in only what did not fit | @@ -462,7 +462,7 @@ meant to outlive it. | `arena_h2d` † | 0.035–0.039 ms / 632 B | 0.03–0.10 ms / 632 B | | `heap_used` | 127,673,344 | 2,038,508,544 | | device wall | 39.3 ms | does not complete yet (`sched_error_code=5 INVALID_ARGS`) | -| `args` (excluded) | 1.37 s / 40.9 GB, 19 of 20 staged | 1.48 s / 45.8 GB, 77 of 92 staged | +| `args` (excluded) | 1.37 s / 40.9 GB, 19 of 20 copied in | 1.48 s / 45.8 GB, 77 of 92 copied in | | `host_view_close` (excluded, legacy mapping path) | 0.25 s / 40.9 GB | 0.28 s / 45.8 GB | † The three upload rows are the markers as they read at that commit, before the @@ -474,16 +474,16 @@ remaining regions in `arena_h2d` — so the same case reports different figures the same work. **dsv4's `args` and `host_view_close` rows no longer describe that case at this -scale.** Both are per-byte costs over what a bind stages, and dsv4's parameters +scale.** Both are per-byte costs over what a bind copies in, and dsv4's parameters now live in child memory: allocated once before the first round, and passed through without malloc, H2D or a host view. What still crosses is `num_tokens_per_owner`, the one caller tensor the host orchestrator has to read — -so a bind stages **1 of its 92 tensors, 8 bytes**. On `dcf7559e8`, 12 binds +so a bind copies in **1 of its 92 tensors, 8 bytes**. On `dcf7559e8`, 12 binds (`--rounds 6`, both ranks) measure `args` at 0.036–0.075 ms and `host_view_close` at 0.0012–0.0030 ms with `count=0 bytes=0`, against 1.48 s and 0.28 s over 45.8 GB above. The same run peaks at 1.31 GiB of host RSS across the whole process tree under `--skip-golden`, and at 23.4 GiB when the fixture is -streamed in, where the row above cost ~45.5 GB per rank. qwen still stages its +streamed in, where the row above cost ~45.5 GB per rank. qwen still copies in its fixture. The rows also describe the legacy mapping behavior at the pinned commit. A @@ -491,7 +491,7 @@ current bind uses the caller's existing host buffers as its orchestration views, so it performs no `halHostRegister` calls and reports `host_view_close count=0 bytes=0`. On Qwen3-14B this makes the close marker 20.12–24.73 us instead of the 0.25 s shown above. The old `args` figure included -20 registrations in addition to staging 19 tensors H2D; current `args` retains +20 registrations in addition to copying 19 tensors in H2D; current `args` retains the H2D work but removes that registration side. Three of these deserve reading together. `host_orch` is the whole story on dsv4 — @@ -499,10 +499,10 @@ Three of these deserve reading together. `host_orch` is the whole story on dsv4 5, 277 and 2 — and its 2.3 ms of scatter is why a claim about it needs a sub-counter rather than a stopwatch. At the pinned commit, `args` plus `host_view_close` are two orders of magnitude above everything else while being -excluded from the control plane: they are staging and legacy mapping costs over +excluded from the control plane: they are copy-in and legacy mapping costs over the ~41–46 GB of weights, not graph dispatch. Current qwen runs retain the -staging cost in `args` but close no mappings; moving dsv4's parameters to child -memory left its bind staging one 8-byte tensor, whose caller-buffer view also +copy-in cost in `args` but close no mappings; moving dsv4's parameters to child +memory left its bind copying in one 8-byte tensor, whose caller-buffer view also needs no mapping. And dsv4's device wall is absent because the case did not complete on device at the pinned commit — it is a completion case with no golden whose host path is what these numbers describe, which is also why diff --git a/docs/dfx/l2-timing.md b/docs/dfx/l2-timing.md index 716906f959..906e1b9399 100644 --- a/docs/dfx/l2-timing.md +++ b/docs/dfx/l2-timing.md @@ -163,8 +163,8 @@ at case setup, outside `Worker.run` and its round markers. Include setup and final validation readback when reporting total case time; do not label the round table alone as end-to-end case latency. -A child-memory argument skips the per-round staging path entirely, so `bind.args` -reports fewer staged tensors and fewer staged bytes for it. Numbers taken +A child-memory argument skips the per-round copy-in path entirely, so `bind.args` +reports a smaller `h2d=` count and fewer bytes for it. Numbers taken before and after a case declares child memory are therefore not comparable on the host/bind component; re-measure both arms with identical fixtures, hardware, round counts and validation settings. diff --git a/docs/task-flow.md b/docs/task-flow.md index 066068be42..d55cb633ac 100644 --- a/docs/task-flow.md +++ b/docs/task-flow.md @@ -192,7 +192,7 @@ snapshot for that purpose. ④ ChipStorageTaskArgs (ChipTensor records + scalars) │ native run or prepare/launch/poll/finalize lifecycle ▼ - chip runtime stages host-backed data or uses owned device memory + chip runtime copies host-backed data in or uses owned device memory ``` A public L2 `Worker.run()` performs the same materialization in its own @@ -248,7 +248,7 @@ run token for the staged prepare/launch/poll/finalize path. This boundary is a descriptor resolution and materialization step, not a memcpy from the mailbox tensor array into `ChipTensor[]`. Host-backed arguments -may need device staging and output copy-back; device-backed arguments must +may need a device copy-in and output copy-back; device-backed arguments must resolve to allocations owned by the target chip. The native `ChipWorker` consumes the resulting POD and invokes the runtime's execution lifecycle. @@ -433,7 +433,7 @@ uncertain. #### TRB temporary buffer -`tensormap_and_ringbuffer` stages ordinary non-child tensor arguments through a +`tensormap_and_ringbuffer` copies ordinary non-child tensor arguments in through a retained temporary buffer owned per pipeline slot, instead of a per-run `device_malloc()` / `device_free()` pair. This is always on for TRB — an internal allocation optimization with no user-facing switch. It is not serialized in task mailboxes @@ -445,7 +445,7 @@ non-child tensors, growing it (free old + malloc new) only when a run needs more than is currently retained, and bump-slices each tensor from it. The buffer lives on the `DeviceRunner` across runs (freed once at finalize); the platform only stores its `{addr, size}` slot. If a grow allocation fails the -run fails before device argument staging. See the runtime's `RUNTIME_LOGIC.md` +run fails before the device arguments are copied in. See the runtime's `RUNTIME_LOGIC.md` §2.4 for the grow/reuse mechanics. ### SUB-type child loop (Python callable leaf) @@ -753,7 +753,7 @@ Step-by-step (one chip worker): | 5 | WT_chip_0 parent side | encode one leased task frame: write `config`, digest prefix, and the args blob; publish `TASK_READY` for the active lane or `PREPARE_READY` for a staged successor | | 6 | chip_0 child process | validate the frame and resolve its digest; ordinary HBG with an active predecessor also prepares the leased inactive arena bank before publishing `FRAME_STAGED`, while a frame with no active predecessor, diagnostic HBG, and TMR publish after validation and defer native prepare | | 7 | chip_0 native-run path | after activation and the predecessor's finalization fence, launch an already-prepared HBG run or finish deferred native preparation and then launch; poll it to completion and finalize it before another staged frame may launch. Compatibility endpoints perform the equivalent operation through blocking `ChipWorker::run` | -| 8 | runtime.so | stage resolved host-backed tensors on the device; dispatch AICPU / AICore; copy output back to `c` during finalization | +| 8 | runtime.so | copy resolved host-backed tensors in to the device; dispatch AICPU / AICore; copy output back to `c` during finalization | | 9 | chip_0 child | native finalization returns; write `TASK_DONE` | | 10 | WT_chip_0 parent | observe `TASK_DONE`; push success completion | | 11 | Scheduler | mark slot COMPLETED; fanout release (none in this DAG); scope_end will release scope ref | diff --git a/docs/testing.md b/docs/testing.md index fea4c3e065..e014123160 100644 --- a/docs/testing.md +++ b/docs/testing.md @@ -907,17 +907,17 @@ When kernels themselves differ (e.g., templated tile sizes tuned for device), se `TensorArg(name, value, child_memory=True)` keeps a case-owned device buffer across all rounds, including `--rounds 1`. `TaskArgsBuilder.add_tensor` accepts -the same keyword. The default remains host staging on every round. +the same keyword. The default remains host memory; IN and INOUT tensors are copied in on every round, while pure OUT buffers skip the copy. | Declaration / direction | Setup | Between rounds | Validation | | ----------------------- | ----- | -------------- | ---------- | -| Host-staged (default) | Existing path | Restore OUT/INOUT host fixtures | Existing per-round copy-back | +| Host memory (default) | Existing path | Restore OUT/INOUT host fixtures | Existing per-round copy-back | | Child-memory IN | Allocate and upload once | Keep device address and input contents | No output readback | | Child-memory OUT | Allocate without upload | Keep device contents; the case must define all compared elements | Final readback | | Child-memory INOUT | Allocate and upload once | Keep device state | Final readback | Golden evaluation follows the same state evolution: child-memory outputs retain -state and host-staged outputs reset. Cases with child-memory outputs compare after +state and host-memory outputs reset. Cases with child-memory outputs compare after the final round; other cases continue comparing every round. A tensor whose contents the HBG host orchestration reads (`get_tensor_data`) or @@ -933,7 +933,7 @@ touches costs nothing. On the second row the cost is per access, not per tensor, so a tensor the orchestration reads thousands of times — `paged_attention`'s `block_table` is -read once per (batch, block) pair — is better left host-staged there. The +read once per (batch, block) pair — is better left in host memory there. The declaration is per argument, so a data-dependent case can mix freely. The bind's `BindHostViewClose` phase attributes report `devcopy=N` when this path was taken. @@ -955,4 +955,4 @@ The HBG `paged_attention_unroll_manual_scope` examples include matched manual `HostStaged` and `ChildMemory` cases, the latter declaring every tensor — including the two the orchestration reads. The HBG `paged_attention` scene tests carry the same pairing as non-manual cases, so CI covers an orchestration -reading child memory on both arches. Existing default cases retain host staging. +reading child memory on both arches. Existing default cases retain host memory. diff --git a/examples/a2a3/host_build_graph/benchmark_bgemm/test_benchmark_bgemm.py b/examples/a2a3/host_build_graph/benchmark_bgemm/test_benchmark_bgemm.py index 37f1bd0f55..db3c8b4f91 100644 --- a/examples/a2a3/host_build_graph/benchmark_bgemm/test_benchmark_bgemm.py +++ b/examples/a2a3/host_build_graph/benchmark_bgemm/test_benchmark_bgemm.py @@ -26,7 +26,7 @@ class TestBenchmarkBgemmHostBuildGraph(SceneTestCase): "function_name": "aicpu_orchestration_entry", # C is a zero-initialized accumulator: the AIV add kernel reads C # from GM, adds the matmul result, and stores it back across grid_k - # iterations. Its host-provided zeros must be staged H2D, so C is + # iterations. Its host-provided zeros must be copied in H2D, so C is # INOUT (read-before-write), not a pure OUT. "signature": [D.IN, D.IN, D.INOUT, D.IN], }, diff --git a/examples/a2a3/tensormap_and_ringbuffer/benchmark_bgemm/test_benchmark_bgemm.py b/examples/a2a3/tensormap_and_ringbuffer/benchmark_bgemm/test_benchmark_bgemm.py index e072150486..19da432923 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/benchmark_bgemm/test_benchmark_bgemm.py +++ b/examples/a2a3/tensormap_and_ringbuffer/benchmark_bgemm/test_benchmark_bgemm.py @@ -26,7 +26,7 @@ class TestBenchmarkBgemm(SceneTestCase): "function_name": "aicpu_orchestration_entry", # C is a zero-initialized accumulator: the AIV add kernel reads C # from GM, adds the matmul result, and stores it back across grid_k - # iterations. Its host-provided zeros must be staged H2D, so C is + # iterations. Its host-provided zeros must be copied in H2D, so C is # INOUT (read-before-write), not a pure OUT. "signature": [D.IN, D.IN, D.INOUT, D.IN], }, diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn.cpp index 4b6c978494..3f9b8c1d22 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn.cpp @@ -1999,7 +1999,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(comb_t_inline9086__ssa_v0_tensor->buffer.addr) + comb_t_inline9086__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale2 read from GM instead of a host-staged scalar) + // Unpack tensor: hc_scale (scale2 read from GM instead of a host-passed scalar) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[4]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_0.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_0.cpp index b7be3573fd..7df3bc883a 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_0.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_0.cpp @@ -1999,7 +1999,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(comb_ffn_inline9269__ssa_v0_tensor->buffer.addr) + comb_ffn_inline9269__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale2 read from GM instead of a host-staged scalar) + // Unpack tensor: hc_scale (scale2 read from GM instead of a host-passed scalar) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[4]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_1.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_1.cpp index 76c371cfc4..4359202bb4 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_1.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_1.cpp @@ -1999,7 +1999,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(comb_t_inline9848__ssa_v0_tensor->buffer.addr) + comb_t_inline9848__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale2 read from GM instead of a host-staged scalar) + // Unpack tensor: hc_scale (scale2 read from GM instead of a host-passed scalar) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[4]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_2.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_2.cpp index 4771ac0336..01f8702553 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_2.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_2.cpp @@ -1999,7 +1999,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(comb_ffn_inline10031__ssa_v0_tensor->buffer.addr) + comb_ffn_inline10031__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale2 read from GM instead of a host-staged scalar) + // Unpack tensor: hc_scale (scale2 read from GM instead of a host-passed scalar) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[4]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_3.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_3.cpp index b532092f7e..9a806515cd 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_3.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_3.cpp @@ -1999,7 +1999,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(comb_t_inline10315__ssa_v0_tensor->buffer.addr) + comb_t_inline10315__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale2 read from GM instead of a host-staged scalar) + // Unpack tensor: hc_scale (scale2 read from GM instead of a host-passed scalar) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[4]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_4.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_4.cpp index 3bf9e1a2a3..b9ebf60465 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_4.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_4.cpp @@ -1999,7 +1999,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(comb_ffn_inline11076__ssa_v0_tensor->buffer.addr) + comb_ffn_inline11076__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale2 read from GM instead of a host-staged scalar) + // Unpack tensor: hc_scale (scale2 read from GM instead of a host-passed scalar) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[4]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_5.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_5.cpp index e82b2d0c1f..0267989f6c 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_5.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_5.cpp @@ -1999,7 +1999,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(comb_t_inline11617__ssa_v0_tensor->buffer.addr) + comb_t_inline11617__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale2 read from GM instead of a host-staged scalar) + // Unpack tensor: hc_scale (scale2 read from GM instead of a host-passed scalar) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[4]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_6.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_6.cpp index a493f77549..b2d2d364e4 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_6.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_6.cpp @@ -1999,7 +1999,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(comb_ffn_inline11955__ssa_v0_tensor->buffer.addr) + comb_ffn_inline11955__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale2 read from GM instead of a host-staged scalar) + // Unpack tensor: hc_scale (scale2 read from GM instead of a host-passed scalar) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[4]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_7.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_7.cpp index da8e63d6d7..3f833dd238 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_7.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_7.cpp @@ -1999,7 +1999,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(comb_t_inline12239__ssa_v0_tensor->buffer.addr) + comb_t_inline12239__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale2 read from GM instead of a host-staged scalar) + // Unpack tensor: hc_scale (scale2 read from GM instead of a host-passed scalar) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[4]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_8.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_8.cpp index 19e5641865..567b9343ec 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_8.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/comb_sinkhorn_8.cpp @@ -1999,7 +1999,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(comb_ffn_inline13000__ssa_v0_tensor->buffer.addr) + comb_ffn_inline13000__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale2 read from GM instead of a host-staged scalar) + // Unpack tensor: hc_scale (scale2 read from GM instead of a host-passed scalar) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[4]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post.cpp index 3bc22c4bed..0d8a6dcca3 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post.cpp @@ -535,7 +535,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(post_t_inline8947__ssa_v0_tensor->buffer.addr) + post_t_inline8947__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-staged scalars) + // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-passed scalars) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[5]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_0.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_0.cpp index b376e68640..e83804a3b5 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_0.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_0.cpp @@ -535,7 +535,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(post_ffn_inline9241__ssa_v0_tensor->buffer.addr) + post_ffn_inline9241__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-staged scalars) + // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-passed scalars) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[5]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_1.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_1.cpp index d4d39f31e6..bad3f4f0f9 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_1.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_1.cpp @@ -535,7 +535,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(post_t_inline9709__ssa_v0_tensor->buffer.addr) + post_t_inline9709__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-staged scalars) + // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-passed scalars) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[5]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_2.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_2.cpp index 2ec8cfacea..ac04bc7e66 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_2.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_2.cpp @@ -535,7 +535,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(post_ffn_inline10003__ssa_v0_tensor->buffer.addr) + post_ffn_inline10003__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-staged scalars) + // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-passed scalars) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[5]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_3.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_3.cpp index ee796f47bf..64ae481237 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_3.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_3.cpp @@ -535,7 +535,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(post_t_inline10558__ssa_v0_tensor->buffer.addr) + post_t_inline10558__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-staged scalars) + // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-passed scalars) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[5]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_4.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_4.cpp index fa6892c339..18947cd719 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_4.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_4.cpp @@ -535,7 +535,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(post_ffn_inline11048__ssa_v0_tensor->buffer.addr) + post_ffn_inline11048__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-staged scalars) + // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-passed scalars) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[5]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_5.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_5.cpp index 55276f446f..777ef8f965 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_5.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_5.cpp @@ -535,7 +535,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(post_t_inline11780__ssa_v0_tensor->buffer.addr) + post_t_inline11780__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-staged scalars) + // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-passed scalars) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[5]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_6.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_6.cpp index a01c8032b5..2236e9bc34 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_6.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_6.cpp @@ -535,7 +535,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(post_ffn_inline11927__ssa_v0_tensor->buffer.addr) + post_ffn_inline11927__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-staged scalars) + // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-passed scalars) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[5]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_7.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_7.cpp index 6374a3b9a6..48a99fe877 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_7.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_7.cpp @@ -535,7 +535,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(post_t_inline12482__ssa_v0_tensor->buffer.addr) + post_t_inline12482__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-staged scalars) + // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-passed scalars) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[5]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_8.cpp b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_8.cpp index f261ca8b84..e293f5b30d 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_8.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/deepseek_v4_flash_decode/kernels/aiv/split_pre_post_8.cpp @@ -535,7 +535,7 @@ extern "C" __aicore__ __attribute__((always_inline)) void kernel_entry(__gm__ in reinterpret_cast<__gm__ float *>(post_ffn_inline12972__ssa_v0_tensor->buffer.addr) + post_ffn_inline12972__ssa_v0_tensor->start_offset; - // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-staged scalars) + // Unpack tensor: hc_scale (scale0/scale1 read from GM instead of host-passed scalars) __gm__ Tensor *hc_scale_tensor = reinterpret_cast<__gm__ Tensor *>(args[5]); __gm__ float *hc_scale = reinterpret_cast<__gm__ float *>(hc_scale_tensor->buffer.addr) + hc_scale_tensor->start_offset; diff --git a/examples/a2a3/tensormap_and_ringbuffer/prefetch_async_demo/kernels/orchestration/prefetch_async_orch.cpp b/examples/a2a3/tensormap_and_ringbuffer/prefetch_async_demo/kernels/orchestration/prefetch_async_orch.cpp index 1787bd6106..343c1aa090 100644 --- a/examples/a2a3/tensormap_and_ringbuffer/prefetch_async_demo/kernels/orchestration/prefetch_async_orch.cpp +++ b/examples/a2a3/tensormap_and_ringbuffer/prefetch_async_demo/kernels/orchestration/prefetch_async_orch.cpp @@ -13,7 +13,7 @@ * * Only user data (in, out) is threaded: the SDMA workspace is a runtime-owned * device resource the kernel reads via get_dma_workspace(args, DMA_WORKSPACE_SDMA), - * so it is neither an orchestration arg nor staged H2D. + * so it is neither an orchestration arg nor copied in H2D. */ #include diff --git a/examples/a5/tensormap_and_ringbuffer/bgemm/test_bgemm.py b/examples/a5/tensormap_and_ringbuffer/bgemm/test_bgemm.py index 72af23eaf0..a8ece9ec9d 100644 --- a/examples/a5/tensormap_and_ringbuffer/bgemm/test_bgemm.py +++ b/examples/a5/tensormap_and_ringbuffer/bgemm/test_bgemm.py @@ -34,7 +34,7 @@ class TestBgemm(SceneTestCase): "function_name": "aicpu_orchestration_entry", # C is a zero-initialized accumulator: the AIV tile kernel reads C # from GM (TLOAD), adds the matmul result, and stores it back across - # GRID_K iterations. Its host-provided zeros must be staged H2D, so + # GRID_K iterations. Its host-provided zeros must be copied in H2D, so # C is INOUT (read-before-write), not a pure OUT. "signature": [D.IN, D.IN, D.INOUT], }, diff --git a/simpler_setup/scene_test.py b/simpler_setup/scene_test.py index a2aefd395f..767ee497c5 100644 --- a/simpler_setup/scene_test.py +++ b/simpler_setup/scene_test.py @@ -533,7 +533,7 @@ class ChildMemoryTaskArgs: The device-side counterpart of :class:`_RehostedTaskArgs`: that one relocates a builder's host tensors into born-shared child buffers so a forked child can reach them, this one relocates them onto the device so every round reuses one - address instead of re-staging. + address instead of re-copying it in every round. ``add`` consumes one CPU contiguous fixture at a time, so a streaming driver can discard each large weight before materializing its successor. Callers @@ -560,7 +560,7 @@ def add(self, name, host, direction): size = host.numel() * host.element_size() if not size: # An empty tensor names no device bytes. Leaving it unrecorded keeps - # it on the ordinary host-staging path, so `build_args` callers must + # it on the ordinary host-memory path, so `build_args` callers must # reconcile their own argument list -- see the count check there. return buf = self.worker.malloc(size) @@ -592,7 +592,7 @@ def build_args(self, expected_count=None): if expected_count is not None and expected_count != len(self.tensors): raise ValueError( f"build_args expected {expected_count} child-memory tensors but holds {len(self.tensors)}; " - "an empty tensor cannot be child memory -- keep it on the host-staging path instead." + "an empty tensor cannot be child memory -- keep it on the host-memory path instead." ) tags = {D.IN: TensorArgType.INPUT, D.OUT: TensorArgType.OUTPUT_EXISTING, D.INOUT: TensorArgType.INOUT} args = TaskArgs() @@ -676,7 +676,7 @@ def _child_memory_args(worker, test_args, signature): """Own the device buffers for every `child_memory` TensorArg, for the whole case. Returns an owner whose `tensors` is empty when nothing is declared, so the - caller's arg build falls through to ordinary host staging. Child-memory storage + caller's arg build falls through to ordinary host memory. Child-memory storage may not alias any other argument's storage: independent device buffers cannot preserve an overlap the orchestrator would otherwise see. """ @@ -718,7 +718,7 @@ def _build_l2_ref_args(test_args: TaskArgsBuilder, orch_signature: list, worker, but set for parity with the L3 path. Explicit `child_memory` arguments use case-owned device addresses; the rest - keep the per-round host-staging path. + keep the per-round host-memory path. Returns: args: TaskArgs (TensorArg) @@ -2022,14 +2022,14 @@ def _run_and_validate_l2( # noqa: PLR0913 -- threads CLI diagnostic flags + cas test_args = self.generate_args(params) with _child_memory_args(worker, test_args, orch_sig) as child_args: chip_args, output_names = _build_l2_ref_args(test_args, orch_sig, worker, child_args=child_args) - staged_outputs = [name for name in output_names if name not in child_args.tensors] + host_memory_outputs = [name for name in output_names if name not in child_args.tensors] child_memory_outputs = [name for name in output_names if name in child_args.tensors] golden_args = None if not skip_golden: golden_args = test_args.clone() with _golden_thread_cap(): - initial_golden = {name: getattr(golden_args, name).clone() for name in staged_outputs} + initial_golden = {name: getattr(golden_args, name).clone() for name in host_memory_outputs} for golden_round in range(rounds if child_memory_outputs else 1): if golden_round: for name, initial in initial_golden.items(): @@ -2041,7 +2041,7 @@ def _run_and_validate_l2( # noqa: PLR0913 -- threads CLI diagnostic flags + cas # Save initial output tensor values for reset between rounds initial_outputs = {} if rounds > 1: - for name in staged_outputs: + for name in host_memory_outputs: initial_outputs[name] = getattr(test_args, name).clone() # Execute rounds. The platform emits `[STRACE]` host/device markers to diff --git a/src/a2a3/runtime/host_build_graph/docs/SCALAR_DATA_ACCESS.md b/src/a2a3/runtime/host_build_graph/docs/SCALAR_DATA_ACCESS.md index 12f91a524f..f9539b6c84 100644 --- a/src/a2a3/runtime/host_build_graph/docs/SCALAR_DATA_ACCESS.md +++ b/src/a2a3/runtime/host_build_graph/docs/SCALAR_DATA_ACCESS.md @@ -2,14 +2,14 @@ `host_build_graph` runs the orchestration function synchronously on the host, before any AICPU scheduler or AICore kernel starts. `get_tensor_data` and -`set_tensor_data` therefore access the host view used to stage the graph's -external tensors; they do not interleave host code with device execution. +`set_tensor_data` therefore access the host view the graph's external tensors +were copied in from; they do not interleave host code with device execution. ## Supported Uses | Tensor state | `get_tensor_data` | `set_tensor_data` | | ------------ | ----------------- | ----------------- | -| External tensor with no submitted producer | Reads the staged host value | Updates the staged host value | +| External tensor with no submitted producer | Reads the copied-in host value | Updates the copied-in host value | | External control/output tensor not referenced by a task | Reads immediately | Writes immediately | | External tensor a submitted task writes (`OUTPUT`/`INOUT`) | Fails with `INVALID_ARGS` | Fails with `INVALID_ARGS` | | Output of a submitted task | Fails with `INVALID_ARGS` | Fails with `INVALID_ARGS` | @@ -17,7 +17,7 @@ external tensors; they do not interleave host code with device execution. | Tensor with an invalid or stale owner task ID | Fails with `INVALID_ARGS` | Fails with `INVALID_ARGS` | The supported write changes the data that will be copied to the device. Every -task in the graph observes that final staged value; submit order does not turn +task in the graph observes that final copied-in value; submit order does not turn the write into a barrier between kernels. ## API @@ -29,7 +29,7 @@ int32_t value = get_tensor_data(control, 1, index); set_tensor_data(layout, 1, index, value + 1); ``` -Both tensors in this example must be external tensors staged by the host. A +Both tensors in this example must be external tensors the host copied in. A common use is to read an input control value or publish runtime geometry into an external layout tensor that no submitted task owns. @@ -84,7 +84,7 @@ producer, so a forged ID cannot reach a task-table slot. A rejection latches ## Practical Rules -- Use scalar access only on external, host-staged tensors that no submitted task +- Use scalar access only on external, host-memory tensors that no submitted task produces. - Use tensor dependencies to order device tasks; do not use host scalar access as a device synchronization barrier. diff --git a/src/a2a3/runtime/host_build_graph/host/runtime_maker.cpp b/src/a2a3/runtime/host_build_graph/host/runtime_maker.cpp index 6164ad7b3b..346fd71a2e 100644 --- a/src/a2a3/runtime/host_build_graph/host/runtime_maker.cpp +++ b/src/a2a3/runtime/host_build_graph/host/runtime_maker.cpp @@ -693,7 +693,7 @@ int32_t run_host_orchestration( orchestrator.total_aiv_count = block_dim * PLATFORM_AIV_CORES_PER_BLOCKDIM; rt->mode = MODE_EXECUTE; // get_tensor_data/set_tensor_data resolve buffer.addr through the host - // views registered at staging time (host_build_graph/host_tensor_access.h), + // views registered at copy-in time (host_build_graph/host_tensor_access.h), // so the host orchestrator can read control tensors (e.g. paged_attention's // context_lens/block_table) whether or not the platform maps device memory // into the host address space. @@ -1095,7 +1095,7 @@ extern "C" int register_callable_impl(const ChipCallable *callable, const HostAp out->host_orch_func_ptr = eps; LOG_INFO("host-orch: loaded orchestration entry '%s' on host", orch_func_name); } - LOG_INFO("Orchestration SO: %zu bytes staged", orch_so_size); + LOG_INFO("Orchestration SO: %zu bytes uploaded", orch_so_size); return 0; } @@ -1171,8 +1171,8 @@ extern "C" int bind_callable_to_runtime_impl( HostTensorAccessor tensor_access(api); const BindPhaseMark args_phase = bind_phase_begin(); - uint64_t staged_bytes = 0; - int staged_tensors = 0; + uint64_t h2d_bytes = 0; + int h2d_tensors = 0; for (int i = 0; i < tensor_count; i++) { ChipTensor t = orch_args->tensor(i); @@ -1184,7 +1184,7 @@ extern "C" int bind_callable_to_runtime_impl( always_assert(t.buffer.addr < HEAP_VIRTUAL_BASE && "caller tensor reaches into the virtual heap window"); LOG_DEBUG(" ChipTensor %d: child memory, pass-through (0x%" PRIx64 ")", i, t.buffer.addr); // The bytes stay where the caller put them, so orchestration has no - // staged buffer to read them from. Claim the span now and let the + // copy-in buffer to read them from. Claim the span now and let the // platform resolve a means only if an access actually lands in it. if (!tensor_access.add_child_memory(t.buffer.addr, t.buffer.size)) { LOG_ERROR("host-orch: could not claim child-memory tensor %d (0x%" PRIx64 ")", i, t.buffer.addr); @@ -1204,19 +1204,19 @@ extern "C" int bind_callable_to_runtime_impl( } // Pure write-only OUTPUT buffers are never read by the kernel and hold - // no meaningful host content, so they need no device staging — the + // no meaningful host content, so they need no copy-in — the // kernel defines what it writes and any unwritten bytes are undefined. - // IN / INOUT (read-before-write) are staged H2D. + // IN / INOUT (read-before-write) are copied in H2D. bool is_pure_output = (signature != nullptr && i < sig_count && signature[i] == ArgDirection::OUT); if (!is_pure_output) { int rc = api->copy_to_device(dev_ptr, host_ptr, size); if (rc != 0) { - LOG_ERROR("Failed to stage tensor %d to device", i); + LOG_ERROR("Failed to copy tensor %d in to the device", i); api->device_free(dev_ptr); return PTO_RUNTIME_ERR_INTERNAL; } - staged_bytes += static_cast(size); - ++staged_tensors; + h2d_bytes += static_cast(size); + ++h2d_tensors; } // Read-only INPUT tensors are never written by the kernel, so there is // no point copying them back D2H at the end. Index the signature @@ -1229,7 +1229,7 @@ extern "C" int bind_callable_to_runtime_impl( LOG_DEBUG(" ChipTensor %d: %zu bytes at %p", i, size, dev_ptr); // host_build_graph runs the orchestrator on the host, which may read - // staged control tensors (e.g. paged_attention's context_lens and + // host-memory control tensors (e.g. paged_attention's context_lens and // block_table) via get_tensor_data to shape the graph. A pure output // has no valid readable bytes before execution, and a5 cannot map it; // exposing its caller buffer would therefore make reads unsafe. Leave @@ -1249,9 +1249,7 @@ extern "C" int bind_callable_to_runtime_impl( } { char attrs[kBindAttrsCapacity]; - snprintf( - attrs, sizeof(attrs), "ntensor=%d staged=%d bytes=%" PRIu64, tensor_count, staged_tensors, staged_bytes - ); + snprintf(attrs, sizeof(attrs), "ntensor=%d h2d=%d bytes=%" PRIu64, tensor_count, h2d_tensors, h2d_bytes); record_bind_phase(HostPhaseKind::BindArgs, args_phase, attrs); } diff --git a/src/a2a3/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp b/src/a2a3/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp index 74987b1f84..22c78480c5 100644 --- a/src/a2a3/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp +++ b/src/a2a3/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp @@ -366,7 +366,7 @@ extern "C" int register_callable_impl(const ChipCallable *callable, const HostAp out->orch_so_size = orch_so_size; out->func_name = callable->func_name(); out->config_name = callable->config_name(); - LOG_INFO("Orchestration SO: %zu bytes staged (host-only)", orch_so_size); + LOG_INFO("Orchestration SO: %zu bytes uploaded (host-only)", orch_so_size); return 0; } @@ -491,14 +491,14 @@ static bool derive_arena_static_sizes(const ArenaSizingConfig &sizing, ArenaStat } // per-run: the only signature-aware step. Copy the orch args, replacing each -// host tensor pointer with a freshly staged device pointer (H2D copy-in, or an +// host tensor pointer with a freshly allocated device pointer (H2D copy-in, or an // on-device zero for pure-OUTPUT buffers), and record the host/device pair for // copy-back. Read-only INPUT tensors skip copy-back. When `bump` is non-null, // ordinary non-child tensors are sliced from the runner's retained temporary // buffer (released as a no-op — the buffer is reused across runs); otherwise // each is device_malloc'd and freed in validate. On failure the partially -// staged device_args / tensor_leases_ stay owned by the caller's Runtime. -static bool stage_device_args( +// copied-in device_args / tensor_leases_ stay owned by the caller's Runtime. +static bool copy_in_device_args( Runtime *runtime, const HostApi *api, const ChipStorageTaskArgs *orch_args, const ArgDirection *signature, int sig_count, RetainedTempBump *bump, ChipStorageTaskArgs *out ) { @@ -542,14 +542,14 @@ static bool stage_device_args( } // Pure write-only OUTPUT buffers are never read by the kernel and hold - // no meaningful host content, so they need no device staging — the + // no meaningful host content, so they need no copy-in — the // kernel defines what it writes and any unwritten bytes are undefined. - // IN / INOUT (read-before-write) are staged H2D. + // IN / INOUT (read-before-write) are copied in H2D. bool is_pure_output = (signature != nullptr && i < sig_count && signature[i] == ArgDirection::OUT); if (!is_pure_output) { int rc = api->copy_to_device(dev_ptr, host_ptr, size); if (rc != 0) { - LOG_ERROR("Failed to stage tensor %d to device", i); + LOG_ERROR("Failed to copy tensor %d in to the device", i); if (release_kind == TensorReleaseKind::Free) { api->device_free(dev_ptr); } @@ -769,7 +769,7 @@ static bool build_and_cache_prebuilt_arena( * half runs only once per callable_id. * * Orchestrates the three lifecycles behind the bind: per-config arena sizing - * (resolve_arena_sizing) + per-run args (stage_device_args) + the prebuilt + * (resolve_arena_sizing) + per-run args (copy_in_device_args) + the prebuilt * runtime-arena image (build_and_cache_prebuilt_arena on a cache miss, then * bind_cached_runtime_image wires the pointers onto the runtime). * @@ -829,7 +829,7 @@ extern "C" int bind_callable_to_runtime_impl( }); ChipStorageTaskArgs device_args; - if (!stage_device_args(runtime, api, orch_args, signature, sig_count, &bump, &device_args)) { + if (!copy_in_device_args(runtime, api, orch_args, signature, sig_count, &bump, &device_args)) { return PTO_RUNTIME_ERR_INTERNAL; } diff --git a/src/a5/runtime/host_build_graph/docs/SCALAR_DATA_ACCESS.md b/src/a5/runtime/host_build_graph/docs/SCALAR_DATA_ACCESS.md index 12f91a524f..f9539b6c84 100644 --- a/src/a5/runtime/host_build_graph/docs/SCALAR_DATA_ACCESS.md +++ b/src/a5/runtime/host_build_graph/docs/SCALAR_DATA_ACCESS.md @@ -2,14 +2,14 @@ `host_build_graph` runs the orchestration function synchronously on the host, before any AICPU scheduler or AICore kernel starts. `get_tensor_data` and -`set_tensor_data` therefore access the host view used to stage the graph's -external tensors; they do not interleave host code with device execution. +`set_tensor_data` therefore access the host view the graph's external tensors +were copied in from; they do not interleave host code with device execution. ## Supported Uses | Tensor state | `get_tensor_data` | `set_tensor_data` | | ------------ | ----------------- | ----------------- | -| External tensor with no submitted producer | Reads the staged host value | Updates the staged host value | +| External tensor with no submitted producer | Reads the copied-in host value | Updates the copied-in host value | | External control/output tensor not referenced by a task | Reads immediately | Writes immediately | | External tensor a submitted task writes (`OUTPUT`/`INOUT`) | Fails with `INVALID_ARGS` | Fails with `INVALID_ARGS` | | Output of a submitted task | Fails with `INVALID_ARGS` | Fails with `INVALID_ARGS` | @@ -17,7 +17,7 @@ external tensors; they do not interleave host code with device execution. | Tensor with an invalid or stale owner task ID | Fails with `INVALID_ARGS` | Fails with `INVALID_ARGS` | The supported write changes the data that will be copied to the device. Every -task in the graph observes that final staged value; submit order does not turn +task in the graph observes that final copied-in value; submit order does not turn the write into a barrier between kernels. ## API @@ -29,7 +29,7 @@ int32_t value = get_tensor_data(control, 1, index); set_tensor_data(layout, 1, index, value + 1); ``` -Both tensors in this example must be external tensors staged by the host. A +Both tensors in this example must be external tensors the host copied in. A common use is to read an input control value or publish runtime geometry into an external layout tensor that no submitted task owns. @@ -84,7 +84,7 @@ producer, so a forged ID cannot reach a task-table slot. A rejection latches ## Practical Rules -- Use scalar access only on external, host-staged tensors that no submitted task +- Use scalar access only on external, host-memory tensors that no submitted task produces. - Use tensor dependencies to order device tasks; do not use host scalar access as a device synchronization barrier. diff --git a/src/a5/runtime/host_build_graph/host/runtime_maker.cpp b/src/a5/runtime/host_build_graph/host/runtime_maker.cpp index 3f78735172..d1fe658dc4 100644 --- a/src/a5/runtime/host_build_graph/host/runtime_maker.cpp +++ b/src/a5/runtime/host_build_graph/host/runtime_maker.cpp @@ -1343,7 +1343,7 @@ int32_t run_host_orchestration( orchestrator.total_aiv_count = block_dim * PLATFORM_AIV_CORES_PER_BLOCKDIM; rt->mode = MODE_EXECUTE; // get_tensor_data/set_tensor_data resolve buffer.addr through the host - // views registered at staging time (host_build_graph/host_tensor_access.h), + // views registered at copy-in time (host_build_graph/host_tensor_access.h), // so the host orchestrator can read control tensors (e.g. paged_attention's // context_lens/block_table) whether or not the platform maps device memory // into the host address space. @@ -1750,7 +1750,7 @@ extern "C" int register_callable_impl(const ChipCallable *callable, const HostAp out->host_orch_func_ptr = eps; LOG_INFO("host-orch: loaded orchestration entry '%s' on host", orch_func_name); } - LOG_INFO("Orchestration SO: %zu bytes staged", orch_so_size); + LOG_INFO("Orchestration SO: %zu bytes uploaded", orch_so_size); return 0; } @@ -1826,8 +1826,8 @@ extern "C" int bind_callable_to_runtime_impl( HostTensorAccessor tensor_access(api); const BindPhaseMark args_phase = bind_phase_begin(); - uint64_t staged_bytes = 0; - int staged_tensors = 0; + uint64_t h2d_bytes = 0; + int h2d_tensors = 0; for (int i = 0; i < tensor_count; i++) { ChipTensor t = orch_args->tensor(i); @@ -1839,7 +1839,7 @@ extern "C" int bind_callable_to_runtime_impl( always_assert(t.buffer.addr < HEAP_VIRTUAL_BASE && "caller tensor reaches into the virtual heap window"); LOG_DEBUG(" ChipTensor %d: child memory, pass-through (0x%" PRIx64 ")", i, t.buffer.addr); // The bytes stay where the caller put them, so orchestration has no - // staged buffer to read them from. Claim the span now and let the + // copy-in buffer to read them from. Claim the span now and let the // platform resolve a means only if an access actually lands in it. if (!tensor_access.add_child_memory(t.buffer.addr, t.buffer.size)) { LOG_ERROR("host-orch: could not claim child-memory tensor %d (0x%" PRIx64 ")", i, t.buffer.addr); @@ -1859,19 +1859,19 @@ extern "C" int bind_callable_to_runtime_impl( } // Pure write-only OUTPUT buffers are never read by the kernel and hold - // no meaningful host content, so they need no device staging — the + // no meaningful host content, so they need no copy-in — the // kernel defines what it writes and any unwritten bytes are undefined. - // IN / INOUT (read-before-write) are staged H2D. + // IN / INOUT (read-before-write) are copied in H2D. bool is_pure_output = (signature != nullptr && i < sig_count && signature[i] == ArgDirection::OUT); if (!is_pure_output) { int rc = api->copy_to_device(dev_ptr, host_ptr, size); if (rc != 0) { - LOG_ERROR("Failed to stage tensor %d to device", i); + LOG_ERROR("Failed to copy tensor %d in to the device", i); api->device_free(dev_ptr); return PTO_RUNTIME_ERR_INTERNAL; } - staged_bytes += static_cast(size); - ++staged_tensors; + h2d_bytes += static_cast(size); + ++h2d_tensors; } // Read-only INPUT tensors are never written by the kernel, so there is // no point copying them back D2H at the end. Index the signature @@ -1884,7 +1884,7 @@ extern "C" int bind_callable_to_runtime_impl( LOG_DEBUG(" ChipTensor %d: %zu bytes at %p", i, size, dev_ptr); // host_build_graph runs the orchestrator on the host, which may read - // staged control tensors (e.g. paged_attention's context_lens and + // host-memory control tensors (e.g. paged_attention's context_lens and // block_table) via get_tensor_data to shape the graph. A pure output // has no valid readable bytes before execution, and a5 cannot map it; // exposing its caller buffer would therefore make reads unsafe. Leave @@ -1904,9 +1904,7 @@ extern "C" int bind_callable_to_runtime_impl( } { char attrs[kBindAttrsCapacity]; - snprintf( - attrs, sizeof(attrs), "ntensor=%d staged=%d bytes=%" PRIu64, tensor_count, staged_tensors, staged_bytes - ); + snprintf(attrs, sizeof(attrs), "ntensor=%d h2d=%d bytes=%" PRIu64, tensor_count, h2d_tensors, h2d_bytes); record_bind_phase(HostPhaseKind::BindArgs, args_phase, attrs); } diff --git a/src/a5/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp b/src/a5/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp index ff02a39bb2..68c311b88d 100644 --- a/src/a5/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp +++ b/src/a5/runtime/tensormap_and_ringbuffer/host/runtime_maker.cpp @@ -366,7 +366,7 @@ extern "C" int register_callable_impl(const ChipCallable *callable, const HostAp out->orch_so_size = orch_so_size; out->func_name = callable->func_name(); out->config_name = callable->config_name(); - LOG_INFO("Orchestration SO: %zu bytes staged (host-only)", orch_so_size); + LOG_INFO("Orchestration SO: %zu bytes uploaded (host-only)", orch_so_size); return 0; } @@ -491,14 +491,14 @@ static bool derive_arena_static_sizes(const ArenaSizingConfig &sizing, ArenaStat } // per-run: the only signature-aware step. Copy the orch args, replacing each -// host tensor pointer with a freshly staged device pointer (H2D copy-in, or an +// host tensor pointer with a freshly allocated device pointer (H2D copy-in, or an // on-device zero for pure-OUTPUT buffers), and record the host/device pair for // copy-back. Read-only INPUT tensors skip copy-back. When `bump` is non-null, // ordinary non-child tensors are sliced from the runner's retained temporary // buffer (released as a no-op — the buffer is reused across runs); otherwise // each is device_malloc'd and freed in validate. On failure the partially -// staged device_args / tensor_leases_ stay owned by the caller's Runtime. -static bool stage_device_args( +// copied-in device_args / tensor_leases_ stay owned by the caller's Runtime. +static bool copy_in_device_args( Runtime *runtime, const HostApi *api, const ChipStorageTaskArgs *orch_args, const ArgDirection *signature, int sig_count, RetainedTempBump *bump, ChipStorageTaskArgs *out ) { @@ -542,14 +542,14 @@ static bool stage_device_args( } // Pure write-only OUTPUT buffers are never read by the kernel and hold - // no meaningful host content, so they need no device staging — the + // no meaningful host content, so they need no copy-in — the // kernel defines what it writes and any unwritten bytes are undefined. - // IN / INOUT (read-before-write) are staged H2D. + // IN / INOUT (read-before-write) are copied in H2D. bool is_pure_output = (signature != nullptr && i < sig_count && signature[i] == ArgDirection::OUT); if (!is_pure_output) { int rc = api->copy_to_device(dev_ptr, host_ptr, size); if (rc != 0) { - LOG_ERROR("Failed to stage tensor %d to device", i); + LOG_ERROR("Failed to copy tensor %d in to the device", i); if (release_kind == TensorReleaseKind::Free) { api->device_free(dev_ptr); } @@ -769,7 +769,7 @@ static bool build_and_cache_prebuilt_arena( * half runs only once per callable_id. * * Orchestrates the three lifecycles behind the bind: per-config arena sizing - * (resolve_arena_sizing) + per-run args (stage_device_args) + the prebuilt + * (resolve_arena_sizing) + per-run args (copy_in_device_args) + the prebuilt * runtime-arena image (build_and_cache_prebuilt_arena on a cache miss, then * bind_cached_runtime_image wires the pointers onto the runtime). * @@ -829,7 +829,7 @@ extern "C" int bind_callable_to_runtime_impl( }); ChipStorageTaskArgs device_args; - if (!stage_device_args(runtime, api, orch_args, signature, sig_count, &bump, &device_args)) { + if (!copy_in_device_args(runtime, api, orch_args, signature, sig_count, &bump, &device_args)) { return PTO_RUNTIME_ERR_INTERNAL; } diff --git a/src/common/host_build_graph/host/host_tensor_access.cpp b/src/common/host_build_graph/host/host_tensor_access.cpp index f8a7604f96..5ce9efeabf 100644 --- a/src/common/host_build_graph/host/host_tensor_access.cpp +++ b/src/common/host_build_graph/host/host_tensor_access.cpp @@ -27,7 +27,7 @@ enum class AccessMeans : uint8_t { // Not yet decided. A child-memory region starts here and resolves on the // first access that lands in it. Unresolved, - // The caller's staged host buffer, or a mapping this accessor installed. + // The caller's host buffer the bind copied in, or a mapping this accessor installed. // `needs_push_back` decides whether a write must also reach the device. HostView, // No host mapping was available for this allocation, so every access is a @@ -45,7 +45,7 @@ struct HostTensorRegion { AccessMeans means; }; -// One entry per tensor staged for the run being orchestrated. A run stages a +// One entry per caller tensor of the run being orchestrated. A run has a // handful of tensors and orchestration reads are cold-path, so a linear scan // costs less than the map that would replace it. struct HostTensorAccessor::Impl { diff --git a/src/common/host_build_graph/host/runtime_core.cpp b/src/common/host_build_graph/host/runtime_core.cpp index c833ecf211..8ed0d6481f 100644 --- a/src/common/host_build_graph/host/runtime_core.cpp +++ b/src/common/host_build_graph/host/runtime_core.cpp @@ -166,9 +166,9 @@ get_tensor_data(RuntimeContext *rt, const simpler::hbg::Tensor &tensor, uint32_t if (!host_tensor_read(rt->tensor_access, elem_addr, &result, elem_size)) { rt->orchestrator->report_fatal( SIMPLER_ERROR_INVALID_ARGS, __FUNCTION__, - "no host view for device address %#llx (%llu bytes): during host orchestration only tensors the " - "runtime staged and child-memory tensors the caller passed in are readable, not runtime-created " - "buffers", + "no host view for device address %#llx (%llu bytes): during host orchestration only host-memory " + "tensors the runtime copied in and child-memory tensors the caller passed in are readable, not " + "runtime-created buffers", (unsigned long long)elem_addr, (unsigned long long)elem_size ); return 0; @@ -197,9 +197,9 @@ void set_tensor_data( if (!host_tensor_write(rt->tensor_access, elem_addr, &value, elem_size)) { rt->orchestrator->report_fatal( SIMPLER_ERROR_INVALID_ARGS, __FUNCTION__, - "no writable host view for device address %#llx (%llu bytes): during host orchestration only tensors " - "the runtime staged and child-memory tensors the caller passed in are writable, not runtime-created " - "buffers", + "no writable host view for device address %#llx (%llu bytes): during host orchestration only " + "host-memory tensors the runtime copied in and child-memory tensors the caller passed in are " + "writable, not runtime-created buffers", (unsigned long long)elem_addr, (unsigned long long)elem_size ); } diff --git a/src/common/host_build_graph/host_tensor_access.h b/src/common/host_build_graph/host_tensor_access.h index 18d4b06e61..1938fdd9f0 100644 --- a/src/common/host_build_graph/host_tensor_access.h +++ b/src/common/host_build_graph/host_tensor_access.h @@ -20,8 +20,8 @@ * that capability is resolved, so the orchestrator core never dereferences a * device address itself. * - * The current bind path registers one region per staged tensor, backed by the - * caller's host tensor buffer: + * The current bind path registers one region per host-memory tensor, backed by + * the caller's host tensor buffer, which the bind has just copied in H2D: * * - A read observes that caller buffer. * - A write mutates that caller buffer, then uses the device-copy hook so the @@ -49,7 +49,7 @@ * `add` also retains a null-fallback platform path: it asks the platform for a * host-readable mapping whose address may equal or differ from `dev_base`, and * always accesses the returned address. The current runtime-maker path cannot - * reach it: staged tensors always have the caller buffer, while pure outputs + * reach it: host-memory tensors always have the caller buffer, while pure outputs * are deliberately left unregistered. The path remains as an explicit * platform-capability escape hatch in `add` and is covered directly by unit * tests; no current production caller reaches it. @@ -59,8 +59,8 @@ * reads and writes resolve to nothing. * * Regions and any optional mappings are owned by one orchestration run — the - * window between staging and the first dispatched task. A caller-buffer view - * holds the staged bytes, and nothing has executed yet to make it stale; once + * window between copy-in and the first dispatched task. A caller-buffer view + * holds the copied-in bytes, and nothing has executed yet to make it stale; once * tasks run, that view would be indistinguishable from live device memory. * `HostTensorAccessor` bounds the window and releases its mappings on every * exit path. A child-memory mapping is the exception it does not own: the @@ -113,7 +113,7 @@ class HostTensorAccessor { * Register `[dev_base, dev_base + size)`, using `fallback_host_view` (the * caller's host tensor buffer) when available and asking the platform for a * host mapping otherwise. The current runtime-maker always supplies the - * fallback for staged tensors and skips pure outputs, so its bind path does + * fallback for host-memory tensors and skips pure outputs, so its bind path does * not install mappings. * * @return false for an empty region, a null `api`, or when neither a @@ -142,7 +142,7 @@ class HostTensorAccessor { /** Mappings installed by `add` and not yet dropped by `close`. */ size_t mapping_count() const noexcept; - /** Total bytes covered by those mappings; excludes fallback staging views. */ + /** Total bytes covered by those mappings; excludes caller-buffer views. */ uint64_t mapped_bytes() const noexcept; /** diff --git a/tests/st/a2a3/host_build_graph/bgemm/test_bgemm.py b/tests/st/a2a3/host_build_graph/bgemm/test_bgemm.py index 2d8448cb1f..130398f360 100644 --- a/tests/st/a2a3/host_build_graph/bgemm/test_bgemm.py +++ b/tests/st/a2a3/host_build_graph/bgemm/test_bgemm.py @@ -69,8 +69,8 @@ def generate_args(self, params): A = torch.randn(BATCH, GRID_M, GRID_K, TILE_M, TILE_K, dtype=torch.float32) * 0.01 B = torch.randn(BATCH, GRID_K, GRID_N, TILE_K, TILE_N, dtype=torch.float32) * 0.01 # C is an INOUT accumulator: the k=0 tile_add reads it before anything - # writes it. A non-zero base makes the host->device staging of C - # observable — a zeroed or unstaged device buffer fails the compare. + # writes it. A non-zero base makes the host->device copy-in of C + # observable — a zeroed or never-copied device buffer fails the compare. C = torch.full((BATCH, GRID_M, GRID_N, TILE_M, TILE_N), C_BASE, dtype=torch.float32) return TaskArgsBuilder( diff --git a/tests/ut/cpp/a2a3/test_hbg_tensor_access.cpp b/tests/ut/cpp/a2a3/test_hbg_tensor_access.cpp index 10588a46fc..7fc4bdec9d 100644 --- a/tests/ut/cpp/a2a3/test_hbg_tensor_access.cpp +++ b/tests/ut/cpp/a2a3/test_hbg_tensor_access.cpp @@ -13,7 +13,7 @@ * Host-view resolution for the host orchestrator's tensor reads and writes, * and the per-run ownership of the mappings that serve them. * - * The fallback path serves staged tensors without mapping their device + * The fallback path serves host-memory tensors without mapping their device * allocations. `g_registered_view` is what the fake * `register_device_memory_to_host` hands back when no fallback is available. */ @@ -179,9 +179,8 @@ TEST_F(HostTensorAccessTest, FallbackWriteReportsCopyFailure) { EXPECT_FALSE(host_tensor_write(&accessor, kFakeDeviceBase, &written, sizeof(written))); } -// The fail-closed contract: an address outside every registered region — a -// GM-heap tensor the orchestrator created or a pass-through child-memory -// buffer — resolves to nothing instead of being dereferenced. +// The fail-closed contract: an address outside every region — a GM-heap tensor +// the orchestrator created — resolves to nothing instead of being dereferenced. TEST_F(HostTensorAccessTest, UnregisteredSpanFailsClosed) { int32_t fallback[2] = {1, 2}; HostTensorAccessor accessor(&kHostApi); diff --git a/tests/ut/py/test_scene_test_child_memory.py b/tests/ut/py/test_scene_test_child_memory.py index f500a639da..a8b434a33a 100644 --- a/tests/ut/py/test_scene_test_child_memory.py +++ b/tests/ut/py/test_scene_test_child_memory.py @@ -86,14 +86,14 @@ def test_child_memory_directions_empty_and_lifo(): TensorArg("y", torch.ones(4), True), TensorArg("z", torch.zeros(4), True), TensorArg("empty", torch.empty(0), True), - TensorArg("staged", torch.ones(4)), + TensorArg("host_memory", torch.ones(4)), ) worker = FakeWorker() with scene._child_memory_args(worker, args, [D.IN, D.INOUT, D.OUT, D.IN, D.IN]) as child_args: assert len(worker.created) == 3 assert worker.uploads == worker.created[:2] # Zero-shaped wire Tensors are rejected by the existing transport; - # the owner skips their device allocation and leaves them host-staged. + # the owner skips their device allocation and leaves them on host memory. assert "empty" not in child_args.tensors nonempty = TaskArgsBuilder(*(spec for spec in args.specs if spec.name != "empty")) _chip_args, outputs = scene._build_l2_ref_args(nonempty, [D.IN, D.INOUT, D.OUT, D.IN], worker, child_args)