| # Annotated review companion for justinchuby/onnx-genai-example-qwen2-5-0-5b-cuda-gqa-f16. | |
| # Model form: CUDA FP16 grouped-query attention. | |
| # Canonical authority: inference_metadata.yaml at revision ec8046b051a8f11e6d339a7d9d85dd1235053989. | |
| # Comments explain authored semantics, derived structure, and fail-closed omissions. | |
| # After YAML parsing this file is mechanically equal to the canonical metadata; comments do not create a second contract. | |
| # Backend implementation, recorded execution, and performance evidence remain separate from this portable metadata. | |
| # Schema spelling `1.0` normalizes under the current reader; versioning gates unknown fields before typed parsing. No component authors batch_capacity, so independent requests must execute per item at every component. | |
| schema_version: 'v1.2' | |
| # The pipeline contains executable package semantics. Deployment/QoS policy and backend-derived execution plans live outside this portable contract. | |
| pipeline: | |
| # This typed SSA workflow is the authoritative execution contract for the package. | |
| workflow: | |
| # Manifest vocabulary declares interpreter/adapter ABI requirements. It is not ORT/native readiness or performance evidence. | |
| manifest: | |
| # Current v1 capability strings mirror typed workflow structure; the structure remains the semantic source of truth. | |
| capabilities: | |
| - workflow_ssa | |
| - linear_effects | |
| - nested_control_flow | |
| - typed_emit | |
| - emit_valid_length | |
| - loop_induction_values | |
| - serving_service_contract | |
| - bounded_state_recurrence | |
| # Workflow inputs are the complete typed boundary. Roles and sources are authored; absent required values must fail before execution. | |
| inputs: | |
| # Input `request.input_ids` is required, provided by the request through a typed runtime role, and carries runtime role prompt_tokens@1.0. Unsupported required roles must fail closed. | |
| request.input_ids: | |
| # Authored tensor ABI: dtype int64, rank 2, shape [batch, sequence_len]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - sequence_len | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: runtime | |
| version: '1.0' | |
| role: prompt_tokens | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: request | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: true | |
| # Input `request.max_iterations` is required, provided by the request through a typed runtime role, and carries runtime role max_output_tokens@1.0. Unsupported required roles must fail closed. | |
| request.max_iterations: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [1]. No grouping permission is implied by this shape. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - 1 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: runtime | |
| version: '1.0' | |
| role: max_output_tokens | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: request | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: true | |
| package.one_token: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: opaque | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: literal | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: 1 | |
| # Input `package.one_step` is optional; omission uses semantic default 1, a package-authored literal/default, and carries opaque. Unsupported required roles must fail closed. | |
| package.one_step: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [1]. No grouping permission is implied by this shape. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - 1 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: opaque | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: literal | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: 1 | |
| # Input `package.max_context` is optional; omission uses semantic default 32768, a package-authored literal/default, and carries opaque. Unsupported required roles must fail closed. | |
| package.max_context: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [1]. No grouping permission is implied by this shape. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - 1 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: opaque | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: literal | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: 32768 | |
| # Input `request.prompt_lengths` is optional; omission uses semantic default -1, provided by the application as prompt_lengths, and carries opaque. Unsupported required roles must fail closed. | |
| request.prompt_lengths: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: opaque | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: application | |
| name: prompt_lengths | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: -1 | |
| # Input `request.eos_ids` is a required resolved runtime value with role eos_token_ids@1.0. An explicit request replaces the package EOS default; otherwise the runtime materializes the package fact. Unsupported roles fail closed. | |
| request.eos_ids: | |
| # The tensor contract validates request shape and dtype at the workflow boundary; it does not duplicate an ONNX component port ABI. | |
| contract: | |
| # Element type required from the request adapter. | |
| dtype: int64 | |
| # Tensor rank required before workflow execution. | |
| rank: 2 | |
| # Symbolic dimensions checked and unified across workflow values. | |
| shape: | |
| - batch | |
| - num_eos | |
| # Request-row identity layout; not independent batching permission. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Typed runtime meaning; unknown required role versions fail admission. | |
| role: | |
| # Role namespace. | |
| kind: runtime | |
| # Role ABI version. | |
| version: '1.0' | |
| # Runtime EOS tensor meaning. | |
| role: eos_token_ids | |
| # The request adapter supplies this value; the package does not author it. | |
| source: | |
| # Value provider class. | |
| kind: request | |
| # Missing required request inputs fail before component execution. | |
| required: true | |
| # Input `request.eos_lengths` is a required resolved runtime value with role eos_token_lengths@1.0. An explicit request replaces the package EOS default; otherwise the runtime materializes the package fact. Unsupported roles fail closed. | |
| request.eos_lengths: | |
| # The tensor contract validates request shape and dtype at the workflow boundary; it does not duplicate an ONNX component port ABI. | |
| contract: | |
| # Element type required from the request adapter. | |
| dtype: int64 | |
| # Tensor rank required before workflow execution. | |
| rank: 1 | |
| # Symbolic dimensions checked and unified across workflow values. | |
| shape: | |
| - batch | |
| # Request-row identity layout; not independent batching permission. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Typed runtime meaning; unknown required role versions fail admission. | |
| role: | |
| # Role namespace. | |
| kind: runtime | |
| # Role ABI version. | |
| version: '1.0' | |
| # Runtime EOS tensor meaning. | |
| role: eos_token_lengths | |
| # The request adapter supplies this value; the package does not author it. | |
| source: | |
| # Value provider class. | |
| kind: request | |
| # Missing required request inputs fail before component execution. | |
| required: true | |
| request.row_max_iterations: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: opaque | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: application | |
| name: row_max_iterations | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: -1 | |
| # Input `request.temperature` is optional; omission uses semantic default 1.0, provided by the request through a typed runtime role, and carries runtime role sampling_temperature@1.0. Unsupported required roles must fail closed. | |
| request.temperature: | |
| # Authored tensor ABI: dtype float32, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float32 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: runtime | |
| version: '1.0' | |
| role: sampling_temperature | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: request | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: 1.0 | |
| # Input `request.top_k` is optional; omission uses semantic default 1, provided by the request through a typed runtime role, and carries runtime role sampling_top_k@1.0. Unsupported required roles must fail closed. | |
| request.top_k: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: runtime | |
| version: '1.0' | |
| role: sampling_top_k | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: request | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: 1 | |
| # Input `request.top_p` is optional; omission uses semantic default 1.0, provided by the request through a typed runtime role, and carries runtime role sampling_top_p@1.0. Unsupported required roles must fail closed. | |
| request.top_p: | |
| # Authored tensor ABI: dtype float32, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float32 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: runtime | |
| version: '1.0' | |
| role: sampling_top_p | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: request | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: 1.0 | |
| # Input `request.min_p` is optional; omission uses semantic default 0.0, provided by the request through a typed runtime role, and carries runtime role sampling_min_p@1.0. Unsupported required roles must fail closed. | |
| request.min_p: | |
| # Authored tensor ABI: dtype float32, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float32 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: runtime | |
| version: '1.0' | |
| role: sampling_min_p | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: request | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: 0.0 | |
| # Input `request.seed` is optional; omission uses semantic default 0, provided by the request through a typed runtime role, and carries runtime role seed@1.0. Unsupported required roles must fail closed. | |
| request.seed: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: runtime | |
| version: '1.0' | |
| role: seed | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: request | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: 0 | |
| # Input `request.rng_counter` is optional; omission uses semantic default 0, provided by the application as rng_counter, and carries opaque. Unsupported required roles must fail closed. | |
| request.rng_counter: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: opaque | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: application | |
| name: rng_counter | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: 0 | |
| # Input `package.active` is optional; omission uses semantic default True, a package-authored literal/default, and carries opaque. Unsupported required roles must fail closed. | |
| package.active: | |
| # Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: opaque | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: literal | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: true | |
| # Input `package.not_done` is optional; omission uses semantic default False, a package-authored literal/default, and carries opaque. Unsupported required roles must fail closed. | |
| package.not_done: | |
| # Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: opaque | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: literal | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: false | |
| # Input `package.cache_lengths` is optional; omission uses semantic default 0, a package-authored literal/default, and carries opaque. Unsupported required roles must fail closed. | |
| package.cache_lengths: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: opaque | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: literal | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: 0 | |
| # Input `package.zero_batch` is optional; omission uses semantic default 0, a package-authored literal/default, and carries opaque. Unsupported required roles must fail closed. | |
| package.zero_batch: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Role identifies runtime semantics without relying on a model-family name. | |
| role: | |
| kind: opaque | |
| # Source states who must provide this value; the runtime must not invent application data. | |
| source: | |
| kind: literal | |
| # Required values fail closed when absent; optional values may use only their authored default/presence rule. | |
| required: false | |
| # This is a semantic tensor default used only when the optional input is omitted. | |
| default: 0 | |
| # Workflow outputs are the complete public result boundary and retain explicit tensor contracts. | |
| outputs: | |
| # Output `tokens` publishes the authored tokens result at the pre_adapter stage; it is not backend-readiness evidence. | |
| tokens: | |
| # Authored tensor ABI: dtype int64, rank 2, shape [batch, generated_sequence]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - generated_sequence | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| role: tokens | |
| # Stage records whether adapters/postprocessing have run before this value is exposed. | |
| stage: pre_adapter | |
| # Components are the executable artifacts and semantic policy programs. Their presence does not prove ORT/native readiness or performance. | |
| components: | |
| # Component `model` is a primary model compute graph using onnx from `model.onnx`. No batch_capacity is authored, so the runtime must not group independent requests at this component. | |
| model: | |
| # Selects a packaged implementation/artifact only; provider/backend readiness requires separate recorded evidence. | |
| implementation: | |
| kind: onnx | |
| artifact: model.onnx | |
| # Ports bind workflow names to the artifact's exact graph ABI; model-family inference is forbidden. | |
| ports: | |
| # Semantic port roles identify tokens, logits, positions, masks, caches, or media without architecture-name conditionals. | |
| roles: | |
| input_ids: token_ids | |
| attention_mask: attention_mask | |
| logits: logits | |
| # Component `token_sampler` is an executable semantic policy component implementing `onnx-genai.token-sampler` using onnx from `policies/token_sampler.onnx`. Policy graphs compute tensors and semantics; they are not deployment/QoS knobs or backend readiness claims. No batch_capacity is authored, so the runtime must not group independent requests at this component. | |
| token_sampler: | |
| # Selects a packaged implementation/artifact only; provider/backend readiness requires separate recorded evidence. | |
| implementation: | |
| kind: onnx | |
| artifact: policies/token_sampler.onnx | |
| # Ports bind workflow names to the artifact's exact graph ABI; model-family inference is forbidden. | |
| ports: | |
| # Exact graph inputs; each tensor shape/dtype is authored. | |
| inputs: | |
| # Port `logits`: Authored tensor ABI: dtype float32, rank 2, shape [batch, vocabulary]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| logits: | |
| dtype: float32 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - vocabulary | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `temperature`: Authored tensor ABI: dtype float32, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| temperature: | |
| dtype: float32 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `top_k`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| top_k: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `top_p`: Authored tensor ABI: dtype float32, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| top_p: | |
| dtype: float32 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `min_p`: Authored tensor ABI: dtype float32, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| min_p: | |
| dtype: float32 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `seed`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| seed: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `counter`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| counter: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `active`: Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| active: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `done`: Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| done: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Exact graph outputs; each tensor shape/dtype is authored. | |
| outputs: | |
| # Port `token`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| token: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `next_counter`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| next_counter: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Semantic component ABI `onnx-genai.token-sampler`; its bindings/parameters affect output correctness. | |
| contract: | |
| id: onnx-genai.token-sampler | |
| version: '2' | |
| bindings: | |
| logits: logits | |
| token: token | |
| temperature: temperature | |
| top_k: top_k | |
| top_p: top_p | |
| min_p: min_p | |
| active: active | |
| done: done | |
| seed: seed | |
| counter: counter | |
| next_counter: next_counter | |
| parameters: | |
| mode: seeded_stochastic | |
| batching: per_row | |
| inactive_rows: preserve | |
| # An override may replace this component only through the same validated semantic contract. | |
| # Application replacement is allowed only through the same typed component contract. | |
| application_overridable: true | |
| # Component `termination` is an executable semantic policy component implementing `onnx-genai.termination-predicate` using onnx from `policies/termination.onnx`. Policy graphs compute tensors and semantics; they are not deployment/QoS knobs or backend readiness claims. No batch_capacity is authored, so the runtime must not group independent requests at this component. | |
| termination: | |
| # Selects a packaged implementation/artifact only; provider/backend readiness requires separate recorded evidence. | |
| implementation: | |
| kind: onnx | |
| artifact: policies/termination.onnx | |
| # Ports bind workflow names to the artifact's exact graph ABI; model-family inference is forbidden. | |
| ports: | |
| # Exact graph inputs; each tensor shape/dtype is authored. | |
| inputs: | |
| # Port `tokens`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| tokens: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `eos_ids`: Authored tensor ABI: dtype int64, rank 2, shape [batch, num_eos]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| eos_ids: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - num_eos | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `eos_lengths`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| eos_lengths: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `iteration`: Authored tensor ABI: dtype int64, rank 1, shape [1]. No grouping permission is implied by this shape. | |
| iteration: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - 1 | |
| # Port `max_iterations`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| max_iterations: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `active`: Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| active: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Exact graph outputs; each tensor shape/dtype is authored. | |
| outputs: | |
| # Port `done`: Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| done: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `next_active`: Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| next_active: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `continue`: Authored tensor ABI: dtype bool, rank 1, shape [1]. No grouping permission is implied by this shape. | |
| continue: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - 1 | |
| # Semantic component ABI `onnx-genai.termination-predicate`; its bindings/parameters affect output correctness. | |
| contract: | |
| id: onnx-genai.termination-predicate | |
| version: '2' | |
| bindings: | |
| tokens: tokens | |
| eos_ids: eos_ids | |
| iteration: iteration | |
| max_iterations: max_iterations | |
| eos_lengths: eos_lengths | |
| active: active | |
| done: done | |
| next_active: next_active | |
| continue: continue | |
| parameters: | |
| batching: per_row | |
| inactive_rows: preserve | |
| # Component `token_state_update` is an executable semantic policy component implementing `onnx-genai.state-update` using onnx from `policies/token_state_update.onnx`. Policy graphs compute tensors and semantics; they are not deployment/QoS knobs or backend readiness claims. No batch_capacity is authored, so the runtime must not group independent requests at this component. | |
| token_state_update: | |
| # Selects a packaged implementation/artifact only; provider/backend readiness requires separate recorded evidence. | |
| implementation: | |
| kind: onnx | |
| artifact: policies/token_state_update.onnx | |
| # Ports bind workflow names to the artifact's exact graph ABI; model-family inference is forbidden. | |
| ports: | |
| # Exact graph inputs; each tensor shape/dtype is authored. | |
| inputs: | |
| # Port `current`: Authored tensor ABI: dtype int64, rank 2, shape [batch, 1]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| current: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - 1 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `update`: Authored tensor ABI: dtype int64, rank 2, shape [batch, 1]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| update: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - 1 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `active`: Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| active: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `done`: Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| done: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Exact graph outputs; each tensor shape/dtype is authored. | |
| outputs: | |
| # Port `next`: Authored tensor ABI: dtype int64, rank 2, shape [batch, 1]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| next: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - 1 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Semantic component ABI `onnx-genai.state-update`; its bindings/parameters affect output correctness. | |
| contract: | |
| id: onnx-genai.state-update | |
| version: '2' | |
| bindings: | |
| current: current | |
| update: update | |
| active: active | |
| done: done | |
| next: next | |
| parameters: | |
| batching: per_row | |
| inactive_rows: preserve | |
| # Component `last_token_logits` is a logit-selection policy graph using onnx from `policies/last_token_logits.onnx`. Policy graphs compute tensors and semantics; they are not deployment/QoS knobs or backend readiness claims. No batch_capacity is authored, so the runtime must not group independent requests at this component. | |
| last_token_logits: | |
| # Selects a packaged implementation/artifact only; provider/backend readiness requires separate recorded evidence. | |
| implementation: | |
| kind: onnx | |
| artifact: policies/last_token_logits.onnx | |
| # Ports bind workflow names to the artifact's exact graph ABI; model-family inference is forbidden. | |
| ports: | |
| # Exact graph inputs; each tensor shape/dtype is authored. | |
| inputs: | |
| # Port `logits`: Authored tensor ABI: dtype float16, rank 3, shape [batch, sequence, vocabulary]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| logits: | |
| dtype: float16 | |
| rank: 3 | |
| shape: | |
| - batch | |
| - sequence | |
| - vocabulary | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Exact graph outputs; each tensor shape/dtype is authored. | |
| outputs: | |
| # Port `last_logits`: Authored tensor ABI: dtype float32, rank 2, shape [batch, vocabulary]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| last_logits: | |
| dtype: float32 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - vocabulary | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Component `decoder_state_initializer` is a primary model compute graph using onnx from `policies/decoder_state_initializer.onnx`. No batch_capacity is authored, so the runtime must not group independent requests at this component. | |
| decoder_state_initializer: | |
| # Selects a packaged implementation/artifact only; provider/backend readiness requires separate recorded evidence. | |
| implementation: | |
| kind: onnx | |
| artifact: policies/decoder_state_initializer.onnx | |
| # Ports bind workflow names to the artifact's exact graph ABI; model-family inference is forbidden. | |
| ports: | |
| # Exact graph inputs; each tensor shape/dtype is authored. | |
| inputs: | |
| # Port `prompt_tokens`: Authored tensor ABI: dtype int64, rank 2, shape [batch, prompt_sequence]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| prompt_tokens: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - prompt_sequence | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `prompt_lengths`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| prompt_lengths: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `max_iterations`: Authored tensor ABI: dtype int64, rank 1, shape [1]. No grouping permission is implied by this shape. | |
| max_iterations: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - 1 | |
| # Exact graph outputs; each tensor shape/dtype is authored. | |
| outputs: | |
| # Port `attention_mask`: Authored tensor ABI: dtype int64, rank 2, shape [batch, capacity]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| attention_mask: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - capacity | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `body_attention_mask`: Authored tensor ABI: dtype int64, rank 2, shape [batch, capacity]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| body_attention_mask: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - capacity | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `token_slot`: Authored tensor ABI: dtype int64, rank 2, shape [batch, 1]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| token_slot: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - 1 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `generated_lengths`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| generated_lengths: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `cache_lengths`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| cache_lengths: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.0.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.0.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.0.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.0.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.1.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.1.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.1.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.1.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.10.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.10.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.10.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.10.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.11.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.11.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.11.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.11.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.12.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.12.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.12.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.12.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.13.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.13.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.13.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.13.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.14.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.14.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.14.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.14.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.15.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.15.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.15.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.15.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.16.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.16.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.16.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.16.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.17.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.17.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.17.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.17.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.18.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.18.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.18.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.18.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.19.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.19.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.19.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.19.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.2.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.2.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.2.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.2.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.20.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.20.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.20.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.20.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.21.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.21.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.21.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.21.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.22.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.22.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.22.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.22.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.23.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.23.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.23.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.23.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.3.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.3.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.3.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.3.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.4.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.4.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.4.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.4.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.5.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.5.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.5.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.5.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.6.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.6.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.6.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.6.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.7.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.7.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.7.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.7.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.8.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.8.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.8.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.8.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.9.key`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.9.key: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `past_key_values.9.value`: Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, capacity, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| past_key_values.9.value: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - capacity | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Component `decoder_step_update` is a primary model compute graph using onnx from `policies/decoder_step_update.onnx`. No batch_capacity is authored, so the runtime must not group independent requests at this component. | |
| decoder_step_update: | |
| # Selects a packaged implementation/artifact only; provider/backend readiness requires separate recorded evidence. | |
| implementation: | |
| kind: onnx | |
| artifact: policies/decoder_step_update.onnx | |
| # Ports bind workflow names to the artifact's exact graph ABI; model-family inference is forbidden. | |
| ports: | |
| # Exact graph inputs; each tensor shape/dtype is authored. | |
| inputs: | |
| # Port `attention_mask`: Authored tensor ABI: dtype int64, rank 2, shape [batch, context]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| attention_mask: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - context | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `logical_length`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| logical_length: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Exact graph outputs; each tensor shape/dtype is authored. | |
| outputs: | |
| # Port `next_attention_mask`: Authored tensor ABI: dtype int64, rank 2, shape [batch, context]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| next_attention_mask: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - context | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Component `cache_length_update` is a state-transition policy graph using onnx from `policies/cache_length_update.onnx`. Policy graphs compute tensors and semantics; they are not deployment/QoS knobs or backend readiness claims. No batch_capacity is authored, so the runtime must not group independent requests at this component. | |
| cache_length_update: | |
| # Selects a packaged implementation/artifact only; provider/backend readiness requires separate recorded evidence. | |
| implementation: | |
| kind: onnx | |
| artifact: policies/cache_length_update.onnx | |
| # Ports bind workflow names to the artifact's exact graph ABI; model-family inference is forbidden. | |
| ports: | |
| # Exact graph inputs; each tensor shape/dtype is authored. | |
| inputs: | |
| # Port `left`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| left: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `right`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| right: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `active`: Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| active: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `done`: Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| done: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Exact graph outputs; each tensor shape/dtype is authored. | |
| outputs: | |
| # Port `total`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| total: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Component `termination_batch_initializer` is a termination policy graph using onnx from `policies/termination_batch_initializer.onnx`. Policy graphs compute tensors and semantics; they are not deployment/QoS knobs or backend readiness claims. No batch_capacity is authored, so the runtime must not group independent requests at this component. | |
| termination_batch_initializer: | |
| # Selects a packaged implementation/artifact only; provider/backend readiness requires separate recorded evidence. | |
| implementation: | |
| kind: onnx | |
| artifact: policies/termination_batch_initializer.onnx | |
| # Ports bind workflow names to the artifact's exact graph ABI; model-family inference is forbidden. | |
| ports: | |
| # Exact graph inputs; each tensor shape/dtype is authored. | |
| inputs: | |
| # Port `input_eos_ids`: Authored tensor ABI: dtype int64, rank 2, shape [batch, num_eos]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| input_eos_ids: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - num_eos | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `input_eos_lengths`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| input_eos_lengths: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `input_max_iterations`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| input_max_iterations: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `fallback_max_iterations`: Authored tensor ABI: dtype int64, rank 1, shape [1]. No grouping permission is implied by this shape. | |
| fallback_max_iterations: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - 1 | |
| # Port `active`: Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| active: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Exact graph outputs; each tensor shape/dtype is authored. | |
| outputs: | |
| # Port `row_eos_ids`: Authored tensor ABI: dtype int64, rank 2, shape [batch, num_eos]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| row_eos_ids: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - num_eos | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `eos_lengths`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| eos_lengths: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `max_iterations`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| max_iterations: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Component `token_to_slot` is an authored workflow compute component using onnx from `policies/token_to_slot.onnx`. No batch_capacity is authored, so the runtime must not group independent requests at this component. | |
| token_to_slot: | |
| # Selects a packaged implementation/artifact only; provider/backend readiness requires separate recorded evidence. | |
| implementation: | |
| kind: onnx | |
| artifact: policies/token_to_slot.onnx | |
| # Ports bind workflow names to the artifact's exact graph ABI; model-family inference is forbidden. | |
| ports: | |
| # Exact graph inputs; each tensor shape/dtype is authored. | |
| inputs: | |
| # Port `token`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| token: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Exact graph outputs; each tensor shape/dtype is authored. | |
| outputs: | |
| # Port `slot`: Authored tensor ABI: dtype int64, rank 2, shape [batch, 1]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| slot: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - 1 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Component `generated_length_update` is a state-transition policy graph using onnx from `policies/generated_length_update.onnx`. Policy graphs compute tensors and semantics; they are not deployment/QoS knobs or backend readiness claims. No batch_capacity is authored, so the runtime must not group independent requests at this component. | |
| generated_length_update: | |
| # Selects a packaged implementation/artifact only; provider/backend readiness requires separate recorded evidence. | |
| implementation: | |
| kind: onnx | |
| artifact: policies/generated_length_update.onnx | |
| # Ports bind workflow names to the artifact's exact graph ABI; model-family inference is forbidden. | |
| ports: | |
| # Exact graph inputs; each tensor shape/dtype is authored. | |
| inputs: | |
| # Port `left`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| left: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `right`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| right: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `active`: Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| active: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Port `done`: Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| done: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # Exact graph outputs; each tensor shape/dtype is authored. | |
| outputs: | |
| # Port `total`: Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| total: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # State cells make recurrence, cache mutation, lifetime, and ownership explicit. Physical paging/tiering/placement are not encoded by tensor names. | |
| state: | |
| # State `token` is semantic tensor carry between workflow steps. Scope=invocation, recurrence=invariant. The authored recurrence determines how it evolves. | |
| token: | |
| # Authored tensor ABI: dtype int64, rank 2, shape [batch, 1]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - 1 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: initializer.token_slot | |
| # Recurrence defines the next-value rule: {kind=invariant}. | |
| recurrence: | |
| kind: invariant | |
| # State `logits` is semantic tensor carry between workflow steps. Scope=invocation, recurrence=invariant. The authored recurrence determines how it evolves. | |
| logits: | |
| # Authored tensor ABI: dtype float32, rank 2, shape [batch, 151936]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float32 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - 151936 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.last_logits | |
| # Recurrence defines the next-value rule: {kind=invariant}. | |
| recurrence: | |
| kind: invariant | |
| # State `generated_lengths` is loop/scheduler control state. Scope=invocation, recurrence=invariant. The values affect correctness, not merely performance. | |
| generated_lengths: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # State class distinguishes semantic state from advisory/runtime-only bookkeeping. | |
| class: semantic | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: initializer.generated_lengths | |
| # Recurrence defines the next-value rule: {kind=invariant}. | |
| recurrence: | |
| kind: invariant | |
| # State `active` is loop/scheduler control state. Scope=invocation, recurrence=invariant. The values affect correctness, not merely performance. | |
| active: | |
| # Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # State class distinguishes semantic state from advisory/runtime-only bookkeeping. | |
| class: semantic | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: package.active | |
| # Recurrence defines the next-value rule: {kind=invariant}. | |
| recurrence: | |
| kind: invariant | |
| # State `done` is loop/scheduler control state. Scope=invocation, recurrence=invariant. The values affect correctness, not merely performance. | |
| done: | |
| # Authored tensor ABI: dtype bool, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: bool | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # State class distinguishes semantic state from advisory/runtime-only bookkeeping. | |
| class: semantic | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: package.not_done | |
| # Recurrence defines the next-value rule: {kind=invariant}. | |
| recurrence: | |
| kind: invariant | |
| # State `accepted_len` is loop/scheduler control state. Scope=invocation, recurrence=invariant. The values affect correctness, not merely performance. | |
| accepted_len: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # State class distinguishes semantic state from advisory/runtime-only bookkeeping. | |
| class: semantic | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: package.zero_batch | |
| # Recurrence defines the next-value rule: {kind=invariant}. | |
| recurrence: | |
| kind: invariant | |
| # State `cache_lengths` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=invariant. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_lengths: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # State class distinguishes semantic state from advisory/runtime-only bookkeeping. | |
| class: semantic | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: initializer.cache_lengths | |
| # Recurrence defines the next-value rule: {kind=invariant}. | |
| recurrence: | |
| kind: invariant | |
| # State `rng_counter` is deterministic random-stream state. Scope=invocation, recurrence=invariant. Its recurrence preserves reproducibility across loop steps. | |
| rng_counter: | |
| # Authored tensor ABI: dtype int64, rank 1, shape [batch]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - batch | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # State class distinguishes semantic state from advisory/runtime-only bookkeeping. | |
| class: semantic | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: request.rng_counter | |
| # Recurrence defines the next-value rule: {kind=invariant}. | |
| recurrence: | |
| kind: invariant | |
| # State `attention_mask` is semantic tensor carry between workflow steps. Scope=invocation, recurrence=invariant. The authored recurrence determines how it evolves. | |
| attention_mask: | |
| # Authored tensor ABI: dtype int64, rank 2, shape [batch, context]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: int64 | |
| rank: 2 | |
| shape: | |
| - batch | |
| - context | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: initializer.attention_mask | |
| # Recurrence defines the next-value rule: {kind=invariant}. | |
| recurrence: | |
| kind: invariant | |
| # State `cache_0` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_0: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.0.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_1` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_1: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.0.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_2` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_2: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.1.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_3` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_3: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.1.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_4` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_4: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.2.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_5` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_5: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.2.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_6` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_6: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.3.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_7` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_7: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.3.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_8` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_8: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.4.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_9` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_9: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.4.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_10` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_10: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.5.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_11` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_11: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.5.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_12` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_12: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.6.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_13` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_13: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.6.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_14` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_14: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.7.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_15` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_15: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.7.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_16` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_16: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.8.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_17` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_17: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.8.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_18` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_18: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.9.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_19` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_19: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.9.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_20` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_20: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.10.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_21` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_21: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.10.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_22` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_22: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.11.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_23` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_23: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.11.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_24` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_24: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.12.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_25` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_25: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.12.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_26` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_26: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.13.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_27` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_27: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.13.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_28` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_28: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.14.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_29` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_29: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.14.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_30` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_30: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.15.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_31` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_31: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.15.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_32` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_32: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.16.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_33` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_33: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.16.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_34` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_34: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.17.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_35` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_35: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.17.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_36` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_36: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.18.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_37` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_37: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.18.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_38` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_38: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.19.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_39` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_39: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.19.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_40` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_40: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.20.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_41` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_41: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.20.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_42` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_42: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.21.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_43` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_43: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.21.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_44` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_44: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.22.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_45` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_45: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.22.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_46` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_46: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.23.key | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # State `cache_47` is semantic key/value or recurrent cache state. Scope=invocation, recurrence=bounded. Physical paging, tiering, prefix storage, and mirror dtype remain runtime policy. | |
| cache_47: | |
| # Authored tensor ABI: dtype float16, rank 4, shape [batch, 2, past_sequence_len, 64]. Request rows stay aligned on axis 0; this preserves identity but does not by itself authorize grouping independent requests. | |
| contract: | |
| dtype: float16 | |
| rank: 4 | |
| shape: | |
| - batch | |
| - 2 | |
| - past_sequence_len | |
| - 64 | |
| # Tensor row layout is structural. Only component batch_capacity can authorize grouping independent requests. | |
| batch_layout: | |
| kind: request_aligned | |
| axis: 0 | |
| # `invocation` defines the semantic lifetime boundary; it is not a worker-placement policy. | |
| scope: invocation | |
| # Initializer names the exact first SSA value for this state cell. | |
| initializer: decoder.setup.present.23.value | |
| # Recurrence defines the next-value rule: {kind=bounded, axis=2, max=package.max_context}. | |
| recurrence: | |
| kind: bounded | |
| axis: 2 | |
| max: package.max_context | |
| # Runtime management means the engine preserves this semantic state; storage layout remains backend policy. | |
| management: runtime | |
| # The state may be released only at this correctness boundary. | |
| release_boundary: invocation | |
| # Service-group membership links this cell to the typed cache/state service below. | |
| service_group: decoder_cache | |
| # Serving binds workflow tensors to scheduler/session semantics. It does not choose worker count, provider, QoS, or claim thread safety. | |
| serving: | |
| # Active-row binding tells the scheduler which request rows still execute. | |
| active: active | |
| # Done-row binding is the semantic termination signal. | |
| done: done | |
| # Accepted-length binding controls cache/state advancement after verification. | |
| accepted_len: accepted_len | |
| # State service groups connect semantic cells to cache/recurrent update contracts used by serving admission. | |
| state_service: | |
| # Each group declares semantic update/rollback behavior; physical cache storage remains deployment/runtime policy. | |
| groups: | |
| # Group `decoder_cache` is semantic `full_attention` state in `bnsh` layout with update None; rollback/fork/snapshot are correctness capabilities. | |
| decoder_cache: | |
| kind: full_attention | |
| # Sequence axis identifies logical history growth for append/rollback validation. | |
| sequence_axis: 2 | |
| # Layout names tensor axis order, not physical paging or device placement. | |
| layout: bnsh | |
| logical_lengths: cache_lengths | |
| aliasing: permitted | |
| reuse: | |
| prefix_reusable: true | |
| evictable_prefix: false | |
| # Port bindings connect service state to exact component input/output names. | |
| ports: | |
| model: | |
| cache_0: | |
| input: past_key_values.0.key | |
| output: present.0.key | |
| role: key | |
| layer: 0 | |
| cache_1: | |
| input: past_key_values.0.value | |
| output: present.0.value | |
| role: value | |
| layer: 0 | |
| cache_2: | |
| input: past_key_values.1.key | |
| output: present.1.key | |
| role: key | |
| layer: 1 | |
| cache_3: | |
| input: past_key_values.1.value | |
| output: present.1.value | |
| role: value | |
| layer: 1 | |
| cache_4: | |
| input: past_key_values.2.key | |
| output: present.2.key | |
| role: key | |
| layer: 2 | |
| cache_5: | |
| input: past_key_values.2.value | |
| output: present.2.value | |
| role: value | |
| layer: 2 | |
| cache_6: | |
| input: past_key_values.3.key | |
| output: present.3.key | |
| role: key | |
| layer: 3 | |
| cache_7: | |
| input: past_key_values.3.value | |
| output: present.3.value | |
| role: value | |
| layer: 3 | |
| cache_8: | |
| input: past_key_values.4.key | |
| output: present.4.key | |
| role: key | |
| layer: 4 | |
| cache_9: | |
| input: past_key_values.4.value | |
| output: present.4.value | |
| role: value | |
| layer: 4 | |
| cache_10: | |
| input: past_key_values.5.key | |
| output: present.5.key | |
| role: key | |
| layer: 5 | |
| cache_11: | |
| input: past_key_values.5.value | |
| output: present.5.value | |
| role: value | |
| layer: 5 | |
| cache_12: | |
| input: past_key_values.6.key | |
| output: present.6.key | |
| role: key | |
| layer: 6 | |
| cache_13: | |
| input: past_key_values.6.value | |
| output: present.6.value | |
| role: value | |
| layer: 6 | |
| cache_14: | |
| input: past_key_values.7.key | |
| output: present.7.key | |
| role: key | |
| layer: 7 | |
| cache_15: | |
| input: past_key_values.7.value | |
| output: present.7.value | |
| role: value | |
| layer: 7 | |
| cache_16: | |
| input: past_key_values.8.key | |
| output: present.8.key | |
| role: key | |
| layer: 8 | |
| cache_17: | |
| input: past_key_values.8.value | |
| output: present.8.value | |
| role: value | |
| layer: 8 | |
| cache_18: | |
| input: past_key_values.9.key | |
| output: present.9.key | |
| role: key | |
| layer: 9 | |
| cache_19: | |
| input: past_key_values.9.value | |
| output: present.9.value | |
| role: value | |
| layer: 9 | |
| cache_20: | |
| input: past_key_values.10.key | |
| output: present.10.key | |
| role: key | |
| layer: 10 | |
| cache_21: | |
| input: past_key_values.10.value | |
| output: present.10.value | |
| role: value | |
| layer: 10 | |
| cache_22: | |
| input: past_key_values.11.key | |
| output: present.11.key | |
| role: key | |
| layer: 11 | |
| cache_23: | |
| input: past_key_values.11.value | |
| output: present.11.value | |
| role: value | |
| layer: 11 | |
| cache_24: | |
| input: past_key_values.12.key | |
| output: present.12.key | |
| role: key | |
| layer: 12 | |
| cache_25: | |
| input: past_key_values.12.value | |
| output: present.12.value | |
| role: value | |
| layer: 12 | |
| cache_26: | |
| input: past_key_values.13.key | |
| output: present.13.key | |
| role: key | |
| layer: 13 | |
| cache_27: | |
| input: past_key_values.13.value | |
| output: present.13.value | |
| role: value | |
| layer: 13 | |
| cache_28: | |
| input: past_key_values.14.key | |
| output: present.14.key | |
| role: key | |
| layer: 14 | |
| cache_29: | |
| input: past_key_values.14.value | |
| output: present.14.value | |
| role: value | |
| layer: 14 | |
| cache_30: | |
| input: past_key_values.15.key | |
| output: present.15.key | |
| role: key | |
| layer: 15 | |
| cache_31: | |
| input: past_key_values.15.value | |
| output: present.15.value | |
| role: value | |
| layer: 15 | |
| cache_32: | |
| input: past_key_values.16.key | |
| output: present.16.key | |
| role: key | |
| layer: 16 | |
| cache_33: | |
| input: past_key_values.16.value | |
| output: present.16.value | |
| role: value | |
| layer: 16 | |
| cache_34: | |
| input: past_key_values.17.key | |
| output: present.17.key | |
| role: key | |
| layer: 17 | |
| cache_35: | |
| input: past_key_values.17.value | |
| output: present.17.value | |
| role: value | |
| layer: 17 | |
| cache_36: | |
| input: past_key_values.18.key | |
| output: present.18.key | |
| role: key | |
| layer: 18 | |
| cache_37: | |
| input: past_key_values.18.value | |
| output: present.18.value | |
| role: value | |
| layer: 18 | |
| cache_38: | |
| input: past_key_values.19.key | |
| output: present.19.key | |
| role: key | |
| layer: 19 | |
| cache_39: | |
| input: past_key_values.19.value | |
| output: present.19.value | |
| role: value | |
| layer: 19 | |
| cache_40: | |
| input: past_key_values.20.key | |
| output: present.20.key | |
| role: key | |
| layer: 20 | |
| cache_41: | |
| input: past_key_values.20.value | |
| output: present.20.value | |
| role: value | |
| layer: 20 | |
| cache_42: | |
| input: past_key_values.21.key | |
| output: present.21.key | |
| role: key | |
| layer: 21 | |
| cache_43: | |
| input: past_key_values.21.value | |
| output: present.21.value | |
| role: value | |
| layer: 21 | |
| cache_44: | |
| input: past_key_values.22.key | |
| output: present.22.key | |
| role: key | |
| layer: 22 | |
| cache_45: | |
| input: past_key_values.22.value | |
| output: present.22.value | |
| role: value | |
| layer: 22 | |
| cache_46: | |
| input: past_key_values.23.key | |
| output: present.23.key | |
| role: key | |
| layer: 23 | |
| cache_47: | |
| input: past_key_values.23.value | |
| output: present.23.value | |
| role: value | |
| layer: 23 | |
| # The ordered SSA program defines actual execution and dataflow; capability strings are not a substitute for these steps. | |
| steps: | |
| # Run a bounded semantic loop (request.max_iterations); carried state and termination values make progress explicit and fail closed if unsupported. | |
| - kind: loop | |
| setup: | |
| # Invoke component `decoder_state_initializer` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: decoder_state_initializer | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| prompt_tokens: request.input_ids | |
| prompt_lengths: request.prompt_lengths | |
| max_iterations: request.max_iterations | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| attention_mask: initializer.attention_mask | |
| body_attention_mask: initializer.body_attention_mask | |
| token_slot: initializer.token_slot | |
| generated_lengths: initializer.generated_lengths | |
| cache_lengths: initializer.cache_lengths | |
| past_key_values.0.key: initializer.past_key_values.0.key | |
| past_key_values.0.value: initializer.past_key_values.0.value | |
| past_key_values.1.key: initializer.past_key_values.1.key | |
| past_key_values.1.value: initializer.past_key_values.1.value | |
| past_key_values.10.key: initializer.past_key_values.10.key | |
| past_key_values.10.value: initializer.past_key_values.10.value | |
| past_key_values.11.key: initializer.past_key_values.11.key | |
| past_key_values.11.value: initializer.past_key_values.11.value | |
| past_key_values.12.key: initializer.past_key_values.12.key | |
| past_key_values.12.value: initializer.past_key_values.12.value | |
| past_key_values.13.key: initializer.past_key_values.13.key | |
| past_key_values.13.value: initializer.past_key_values.13.value | |
| past_key_values.14.key: initializer.past_key_values.14.key | |
| past_key_values.14.value: initializer.past_key_values.14.value | |
| past_key_values.15.key: initializer.past_key_values.15.key | |
| past_key_values.15.value: initializer.past_key_values.15.value | |
| past_key_values.16.key: initializer.past_key_values.16.key | |
| past_key_values.16.value: initializer.past_key_values.16.value | |
| past_key_values.17.key: initializer.past_key_values.17.key | |
| past_key_values.17.value: initializer.past_key_values.17.value | |
| past_key_values.18.key: initializer.past_key_values.18.key | |
| past_key_values.18.value: initializer.past_key_values.18.value | |
| past_key_values.19.key: initializer.past_key_values.19.key | |
| past_key_values.19.value: initializer.past_key_values.19.value | |
| past_key_values.2.key: initializer.past_key_values.2.key | |
| past_key_values.2.value: initializer.past_key_values.2.value | |
| past_key_values.20.key: initializer.past_key_values.20.key | |
| past_key_values.20.value: initializer.past_key_values.20.value | |
| past_key_values.21.key: initializer.past_key_values.21.key | |
| past_key_values.21.value: initializer.past_key_values.21.value | |
| past_key_values.22.key: initializer.past_key_values.22.key | |
| past_key_values.22.value: initializer.past_key_values.22.value | |
| past_key_values.23.key: initializer.past_key_values.23.key | |
| past_key_values.23.value: initializer.past_key_values.23.value | |
| past_key_values.3.key: initializer.past_key_values.3.key | |
| past_key_values.3.value: initializer.past_key_values.3.value | |
| past_key_values.4.key: initializer.past_key_values.4.key | |
| past_key_values.4.value: initializer.past_key_values.4.value | |
| past_key_values.5.key: initializer.past_key_values.5.key | |
| past_key_values.5.value: initializer.past_key_values.5.value | |
| past_key_values.6.key: initializer.past_key_values.6.key | |
| past_key_values.6.value: initializer.past_key_values.6.value | |
| past_key_values.7.key: initializer.past_key_values.7.key | |
| past_key_values.7.value: initializer.past_key_values.7.value | |
| past_key_values.8.key: initializer.past_key_values.8.key | |
| past_key_values.8.value: initializer.past_key_values.8.value | |
| past_key_values.9.key: initializer.past_key_values.9.key | |
| past_key_values.9.value: initializer.past_key_values.9.value | |
| # Invoke component `model` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: model | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| input_ids: request.input_ids | |
| past_key_values.0.key: initializer.past_key_values.0.key | |
| past_key_values.0.value: initializer.past_key_values.0.value | |
| past_key_values.1.key: initializer.past_key_values.1.key | |
| past_key_values.1.value: initializer.past_key_values.1.value | |
| past_key_values.2.key: initializer.past_key_values.2.key | |
| past_key_values.2.value: initializer.past_key_values.2.value | |
| past_key_values.3.key: initializer.past_key_values.3.key | |
| past_key_values.3.value: initializer.past_key_values.3.value | |
| past_key_values.4.key: initializer.past_key_values.4.key | |
| past_key_values.4.value: initializer.past_key_values.4.value | |
| past_key_values.5.key: initializer.past_key_values.5.key | |
| past_key_values.5.value: initializer.past_key_values.5.value | |
| past_key_values.6.key: initializer.past_key_values.6.key | |
| past_key_values.6.value: initializer.past_key_values.6.value | |
| past_key_values.7.key: initializer.past_key_values.7.key | |
| past_key_values.7.value: initializer.past_key_values.7.value | |
| past_key_values.8.key: initializer.past_key_values.8.key | |
| past_key_values.8.value: initializer.past_key_values.8.value | |
| past_key_values.9.key: initializer.past_key_values.9.key | |
| past_key_values.9.value: initializer.past_key_values.9.value | |
| past_key_values.10.key: initializer.past_key_values.10.key | |
| past_key_values.10.value: initializer.past_key_values.10.value | |
| past_key_values.11.key: initializer.past_key_values.11.key | |
| past_key_values.11.value: initializer.past_key_values.11.value | |
| past_key_values.12.key: initializer.past_key_values.12.key | |
| past_key_values.12.value: initializer.past_key_values.12.value | |
| past_key_values.13.key: initializer.past_key_values.13.key | |
| past_key_values.13.value: initializer.past_key_values.13.value | |
| past_key_values.14.key: initializer.past_key_values.14.key | |
| past_key_values.14.value: initializer.past_key_values.14.value | |
| past_key_values.15.key: initializer.past_key_values.15.key | |
| past_key_values.15.value: initializer.past_key_values.15.value | |
| past_key_values.16.key: initializer.past_key_values.16.key | |
| past_key_values.16.value: initializer.past_key_values.16.value | |
| past_key_values.17.key: initializer.past_key_values.17.key | |
| past_key_values.17.value: initializer.past_key_values.17.value | |
| past_key_values.18.key: initializer.past_key_values.18.key | |
| past_key_values.18.value: initializer.past_key_values.18.value | |
| past_key_values.19.key: initializer.past_key_values.19.key | |
| past_key_values.19.value: initializer.past_key_values.19.value | |
| past_key_values.20.key: initializer.past_key_values.20.key | |
| past_key_values.20.value: initializer.past_key_values.20.value | |
| past_key_values.21.key: initializer.past_key_values.21.key | |
| past_key_values.21.value: initializer.past_key_values.21.value | |
| past_key_values.22.key: initializer.past_key_values.22.key | |
| past_key_values.22.value: initializer.past_key_values.22.value | |
| past_key_values.23.key: initializer.past_key_values.23.key | |
| past_key_values.23.value: initializer.past_key_values.23.value | |
| attention_mask: initializer.attention_mask | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| logits: decoder.setup.logits | |
| present.0.key: decoder.setup.present.0.key | |
| present.0.value: decoder.setup.present.0.value | |
| present.1.key: decoder.setup.present.1.key | |
| present.1.value: decoder.setup.present.1.value | |
| present.2.key: decoder.setup.present.2.key | |
| present.2.value: decoder.setup.present.2.value | |
| present.3.key: decoder.setup.present.3.key | |
| present.3.value: decoder.setup.present.3.value | |
| present.4.key: decoder.setup.present.4.key | |
| present.4.value: decoder.setup.present.4.value | |
| present.5.key: decoder.setup.present.5.key | |
| present.5.value: decoder.setup.present.5.value | |
| present.6.key: decoder.setup.present.6.key | |
| present.6.value: decoder.setup.present.6.value | |
| present.7.key: decoder.setup.present.7.key | |
| present.7.value: decoder.setup.present.7.value | |
| present.8.key: decoder.setup.present.8.key | |
| present.8.value: decoder.setup.present.8.value | |
| present.9.key: decoder.setup.present.9.key | |
| present.9.value: decoder.setup.present.9.value | |
| present.10.key: decoder.setup.present.10.key | |
| present.10.value: decoder.setup.present.10.value | |
| present.11.key: decoder.setup.present.11.key | |
| present.11.value: decoder.setup.present.11.value | |
| present.12.key: decoder.setup.present.12.key | |
| present.12.value: decoder.setup.present.12.value | |
| present.13.key: decoder.setup.present.13.key | |
| present.13.value: decoder.setup.present.13.value | |
| present.14.key: decoder.setup.present.14.key | |
| present.14.value: decoder.setup.present.14.value | |
| present.15.key: decoder.setup.present.15.key | |
| present.15.value: decoder.setup.present.15.value | |
| present.16.key: decoder.setup.present.16.key | |
| present.16.value: decoder.setup.present.16.value | |
| present.17.key: decoder.setup.present.17.key | |
| present.17.value: decoder.setup.present.17.value | |
| present.18.key: decoder.setup.present.18.key | |
| present.18.value: decoder.setup.present.18.value | |
| present.19.key: decoder.setup.present.19.key | |
| present.19.value: decoder.setup.present.19.value | |
| present.20.key: decoder.setup.present.20.key | |
| present.20.value: decoder.setup.present.20.value | |
| present.21.key: decoder.setup.present.21.key | |
| present.21.value: decoder.setup.present.21.value | |
| present.22.key: decoder.setup.present.22.key | |
| present.22.value: decoder.setup.present.22.value | |
| present.23.key: decoder.setup.present.23.key | |
| present.23.value: decoder.setup.present.23.value | |
| # Invoke component `termination_batch_initializer` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: termination_batch_initializer | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| input_eos_ids: request.eos_ids | |
| input_eos_lengths: request.eos_lengths | |
| input_max_iterations: request.row_max_iterations | |
| fallback_max_iterations: request.max_iterations | |
| active: package.active | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| row_eos_ids: termination.eos_ids | |
| eos_lengths: termination.eos_lengths | |
| max_iterations: termination.max_iterations | |
| # Invoke component `last_token_logits` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: last_token_logits | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| logits: decoder.setup.logits | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| last_logits: decoder.setup.last_logits | |
| # Nested steps remain part of the typed executable workflow. | |
| steps: | |
| # Invoke component `token_sampler` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: token_sampler | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| logits: logits | |
| temperature: request.temperature | |
| top_k: request.top_k | |
| top_p: request.top_p | |
| min_p: request.min_p | |
| seed: request.seed | |
| counter: rng_counter | |
| active: active | |
| done: done | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| token: sample.body | |
| next_counter: sample.next_counter | |
| # Invoke component `token_to_slot` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: token_to_slot | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| token: sample.body | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| slot: sample.slot | |
| # Invoke component `generated_length_update` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: generated_length_update | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| left: generated_lengths | |
| right: package.one_token | |
| active: active | |
| done: done | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| total: token.next_lengths | |
| # Invoke component `generated_length_update` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: generated_length_update | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| left: package.zero_batch | |
| right: package.one_token | |
| active: active | |
| done: done | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| total: token.emitted_length | |
| # Invoke component `token_state_update` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: token_state_update | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| current: token | |
| update: sample.slot | |
| active: active | |
| done: done | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| next: token.body | |
| # Invoke component `termination` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: termination | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| tokens: sample.body | |
| eos_ids: termination.eos_ids | |
| eos_lengths: termination.eos_lengths | |
| iteration: loop.iteration | |
| max_iterations: termination.max_iterations | |
| active: active | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| done: loop.done | |
| continue: loop.continue | |
| next_active: loop.next_active | |
| # Invoke component `cache_length_update` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: cache_length_update | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| left: cache_lengths | |
| right: package.one_token | |
| active: active | |
| done: done | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| total: cache_lengths.next | |
| # Invoke component `cache_length_update` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: cache_length_update | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| left: package.zero_batch | |
| right: package.one_token | |
| active: active | |
| done: done | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| total: accepted_len.next | |
| # Publish `token.body` as output `tokens` using append semantics. | |
| - kind: emit | |
| value: token.body | |
| output: tokens | |
| # Emit/update mode is semantic replacement/append behavior, not a buffering optimization. | |
| mode: append | |
| # Valid length prevents padded/unused elements from becoming observable output. | |
| valid_length: token.emitted_length | |
| when: active | |
| # Invoke component `decoder_step_update` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: decoder_step_update | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| attention_mask: attention_mask | |
| logical_length: cache_lengths | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| next_attention_mask: decoder_step.body_attention_mask | |
| # Invoke component `model` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: model | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| input_ids: token.body | |
| past_key_values.0.key: cache_0 | |
| past_key_values.0.value: cache_1 | |
| past_key_values.1.key: cache_2 | |
| past_key_values.1.value: cache_3 | |
| past_key_values.2.key: cache_4 | |
| past_key_values.2.value: cache_5 | |
| past_key_values.3.key: cache_6 | |
| past_key_values.3.value: cache_7 | |
| past_key_values.4.key: cache_8 | |
| past_key_values.4.value: cache_9 | |
| past_key_values.5.key: cache_10 | |
| past_key_values.5.value: cache_11 | |
| past_key_values.6.key: cache_12 | |
| past_key_values.6.value: cache_13 | |
| past_key_values.7.key: cache_14 | |
| past_key_values.7.value: cache_15 | |
| past_key_values.8.key: cache_16 | |
| past_key_values.8.value: cache_17 | |
| past_key_values.9.key: cache_18 | |
| past_key_values.9.value: cache_19 | |
| past_key_values.10.key: cache_20 | |
| past_key_values.10.value: cache_21 | |
| past_key_values.11.key: cache_22 | |
| past_key_values.11.value: cache_23 | |
| past_key_values.12.key: cache_24 | |
| past_key_values.12.value: cache_25 | |
| past_key_values.13.key: cache_26 | |
| past_key_values.13.value: cache_27 | |
| past_key_values.14.key: cache_28 | |
| past_key_values.14.value: cache_29 | |
| past_key_values.15.key: cache_30 | |
| past_key_values.15.value: cache_31 | |
| past_key_values.16.key: cache_32 | |
| past_key_values.16.value: cache_33 | |
| past_key_values.17.key: cache_34 | |
| past_key_values.17.value: cache_35 | |
| past_key_values.18.key: cache_36 | |
| past_key_values.18.value: cache_37 | |
| past_key_values.19.key: cache_38 | |
| past_key_values.19.value: cache_39 | |
| past_key_values.20.key: cache_40 | |
| past_key_values.20.value: cache_41 | |
| past_key_values.21.key: cache_42 | |
| past_key_values.21.value: cache_43 | |
| past_key_values.22.key: cache_44 | |
| past_key_values.22.value: cache_45 | |
| past_key_values.23.key: cache_46 | |
| past_key_values.23.value: cache_47 | |
| attention_mask: decoder_step.body_attention_mask | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| logits: decoder.body.logits | |
| present.0.key: decoder.body.present.0.key | |
| present.0.value: decoder.body.present.0.value | |
| present.1.key: decoder.body.present.1.key | |
| present.1.value: decoder.body.present.1.value | |
| present.2.key: decoder.body.present.2.key | |
| present.2.value: decoder.body.present.2.value | |
| present.3.key: decoder.body.present.3.key | |
| present.3.value: decoder.body.present.3.value | |
| present.4.key: decoder.body.present.4.key | |
| present.4.value: decoder.body.present.4.value | |
| present.5.key: decoder.body.present.5.key | |
| present.5.value: decoder.body.present.5.value | |
| present.6.key: decoder.body.present.6.key | |
| present.6.value: decoder.body.present.6.value | |
| present.7.key: decoder.body.present.7.key | |
| present.7.value: decoder.body.present.7.value | |
| present.8.key: decoder.body.present.8.key | |
| present.8.value: decoder.body.present.8.value | |
| present.9.key: decoder.body.present.9.key | |
| present.9.value: decoder.body.present.9.value | |
| present.10.key: decoder.body.present.10.key | |
| present.10.value: decoder.body.present.10.value | |
| present.11.key: decoder.body.present.11.key | |
| present.11.value: decoder.body.present.11.value | |
| present.12.key: decoder.body.present.12.key | |
| present.12.value: decoder.body.present.12.value | |
| present.13.key: decoder.body.present.13.key | |
| present.13.value: decoder.body.present.13.value | |
| present.14.key: decoder.body.present.14.key | |
| present.14.value: decoder.body.present.14.value | |
| present.15.key: decoder.body.present.15.key | |
| present.15.value: decoder.body.present.15.value | |
| present.16.key: decoder.body.present.16.key | |
| present.16.value: decoder.body.present.16.value | |
| present.17.key: decoder.body.present.17.key | |
| present.17.value: decoder.body.present.17.value | |
| present.18.key: decoder.body.present.18.key | |
| present.18.value: decoder.body.present.18.value | |
| present.19.key: decoder.body.present.19.key | |
| present.19.value: decoder.body.present.19.value | |
| present.20.key: decoder.body.present.20.key | |
| present.20.value: decoder.body.present.20.value | |
| present.21.key: decoder.body.present.21.key | |
| present.21.value: decoder.body.present.21.value | |
| present.22.key: decoder.body.present.22.key | |
| present.22.value: decoder.body.present.22.value | |
| present.23.key: decoder.body.present.23.key | |
| present.23.value: decoder.body.present.23.value | |
| # Invoke component `last_token_logits` with explicit SSA input/output bindings; this is executable tensor computation, not a host-side capability flag. | |
| - kind: invoke | |
| component: last_token_logits | |
| # SSA bindings supply exact values to this step; no implicit tensor discovery occurs. | |
| inputs: | |
| logits: decoder.body.logits | |
| # SSA output names become the only values subsequent steps may consume. | |
| outputs: | |
| last_logits: decoder.body.last_logits | |
| continue_when: active | |
| max_iterations: request.max_iterations | |
| carried: | |
| - cell: token | |
| next: token.body | |
| - cell: logits | |
| next: decoder.body.last_logits | |
| - cell: generated_lengths | |
| next: token.next_lengths | |
| - cell: active | |
| next: loop.next_active | |
| - cell: done | |
| next: loop.done | |
| - cell: cache_lengths | |
| next: cache_lengths.next | |
| - cell: accepted_len | |
| next: accepted_len.next | |
| - cell: rng_counter | |
| next: sample.next_counter | |
| - cell: attention_mask | |
| next: decoder_step.body_attention_mask | |
| - cell: cache_0 | |
| next: decoder.body.present.0.key | |
| - cell: cache_1 | |
| next: decoder.body.present.0.value | |
| - cell: cache_2 | |
| next: decoder.body.present.1.key | |
| - cell: cache_3 | |
| next: decoder.body.present.1.value | |
| - cell: cache_4 | |
| next: decoder.body.present.2.key | |
| - cell: cache_5 | |
| next: decoder.body.present.2.value | |
| - cell: cache_6 | |
| next: decoder.body.present.3.key | |
| - cell: cache_7 | |
| next: decoder.body.present.3.value | |
| - cell: cache_8 | |
| next: decoder.body.present.4.key | |
| - cell: cache_9 | |
| next: decoder.body.present.4.value | |
| - cell: cache_10 | |
| next: decoder.body.present.5.key | |
| - cell: cache_11 | |
| next: decoder.body.present.5.value | |
| - cell: cache_12 | |
| next: decoder.body.present.6.key | |
| - cell: cache_13 | |
| next: decoder.body.present.6.value | |
| - cell: cache_14 | |
| next: decoder.body.present.7.key | |
| - cell: cache_15 | |
| next: decoder.body.present.7.value | |
| - cell: cache_16 | |
| next: decoder.body.present.8.key | |
| - cell: cache_17 | |
| next: decoder.body.present.8.value | |
| - cell: cache_18 | |
| next: decoder.body.present.9.key | |
| - cell: cache_19 | |
| next: decoder.body.present.9.value | |
| - cell: cache_20 | |
| next: decoder.body.present.10.key | |
| - cell: cache_21 | |
| next: decoder.body.present.10.value | |
| - cell: cache_22 | |
| next: decoder.body.present.11.key | |
| - cell: cache_23 | |
| next: decoder.body.present.11.value | |
| - cell: cache_24 | |
| next: decoder.body.present.12.key | |
| - cell: cache_25 | |
| next: decoder.body.present.12.value | |
| - cell: cache_26 | |
| next: decoder.body.present.13.key | |
| - cell: cache_27 | |
| next: decoder.body.present.13.value | |
| - cell: cache_28 | |
| next: decoder.body.present.14.key | |
| - cell: cache_29 | |
| next: decoder.body.present.14.value | |
| - cell: cache_30 | |
| next: decoder.body.present.15.key | |
| - cell: cache_31 | |
| next: decoder.body.present.15.value | |
| - cell: cache_32 | |
| next: decoder.body.present.16.key | |
| - cell: cache_33 | |
| next: decoder.body.present.16.value | |
| - cell: cache_34 | |
| next: decoder.body.present.17.key | |
| - cell: cache_35 | |
| next: decoder.body.present.17.value | |
| - cell: cache_36 | |
| next: decoder.body.present.18.key | |
| - cell: cache_37 | |
| next: decoder.body.present.18.value | |
| - cell: cache_38 | |
| next: decoder.body.present.19.key | |
| - cell: cache_39 | |
| next: decoder.body.present.19.value | |
| - cell: cache_40 | |
| next: decoder.body.present.20.key | |
| - cell: cache_41 | |
| next: decoder.body.present.20.value | |
| - cell: cache_42 | |
| next: decoder.body.present.21.key | |
| - cell: cache_43 | |
| next: decoder.body.present.21.value | |
| - cell: cache_44 | |
| next: decoder.body.present.22.key | |
| - cell: cache_45 | |
| next: decoder.body.present.22.value | |
| - cell: cache_46 | |
| next: decoder.body.present.23.key | |
| - cell: cache_47 | |
| next: decoder.body.present.23.value | |
| termination: generation_eos | |
| iteration: | |
| value: loop.iteration | |
| contract: | |
| dtype: int64 | |
| rank: 1 | |
| shape: | |
| - 1 | |
| # Package facts are portable execution defaults and identities. They are not deployment policy, backend support claims, or request overrides. | |
| package: | |
| # Tokenizer facts interpret request/model token IDs. Numeric special-token facts are package defaults; token spellings and chat templates remain in the tokenizer assets. | |
| tokenizer: | |
| # Execution-relevant token IDs have one authority here. Request EOS values are optional overrides; workflow inputs and termination components do not own another package-default copy. | |
| special_tokens: | |
| # Ordered set: generation stops on any listed package-default ID. | |
| eos_token_id: | |
| - 151645 | |
| # Padding ID used by package-authored tensor contracts. | |
| pad_token_id: 151643 | |
| # Beginning-of-sequence ID. | |
| bos_token_id: 151643 | |