Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
25 commits
Select commit Hold shift + click to select a range
caa11e6
[Feature][Operator] Add DeepSeek V4.1 sparse attention operators
GDzhu01 Sep 12, 2026
fd87bf4
[Docs] Fix Aurora compilation matrix link
GDzhu01 Sep 12, 2026
e53d0b0
[Docs] Avoid unresolved operator source link
GDzhu01 Sep 12, 2026
e791a85
[Fix] Preserve hc pre v2 operator compatibility
GDzhu01 Sep 12, 2026
db04d32
[Fix] Preserve QLI metadata device argument
GDzhu01 Sep 12, 2026
99aa88c
Preserve quant lightning indexer v2 compatibility
GDzhu01 Sep 12, 2026
001bb52
Route legacy QLI calls through extended backend
GDzhu01 Sep 12, 2026
64da806
[Feature][Model] Add DeepSeek V4.1 serving support
GDzhu01 Sep 12, 2026
dc38b8d
Fix latest-main compatibility regressions
GDzhu01 Sep 12, 2026
e898ee2
Align framework calls with versioned operators
GDzhu01 Sep 13, 2026
5dd4bcf
Fix DeepSeek V4.1 mypy errors
GDzhu01 Sep 13, 2026
1a87614
Fix DeepSeek V4.1 test typing
GDzhu01 Sep 13, 2026
6226799
Fix DeepSeek V4.1 CPU regressions
GDzhu01 Sep 13, 2026
c2e12a2
Avoid NPU initialization in CPU Engram paths
GDzhu01 Sep 13, 2026
6f6b613
Add DeepSeek V4 vision test timing
GDzhu01 Sep 13, 2026
3096c82
Defer Ascend MoE ops during global patching
GDzhu01 Sep 13, 2026
96fa872
Keep lazy MoE router factory patchable
GDzhu01 Sep 13, 2026
5c4187d
Defer EPLB operator import during global patching
GDzhu01 Sep 13, 2026
4e4317d
Preserve legacy DSA CP exchange semantics
GDzhu01 Sep 13, 2026
c8e468b
Preserve eager DP dummy request metadata
GDzhu01 Sep 13, 2026
d86e1ed
Keep gathered DSA projection weights alive
GDzhu01 Sep 13, 2026
3be9d21
Fix DSA CP sequence-parallel MoE handoff
GDzhu01 Sep 13, 2026
3f4b620
Format DSA CP regression fix
GDzhu01 Sep 13, 2026
777325a
Preserve upstream speculative decode state updates
GDzhu01 Sep 13, 2026
e67ab64
Preserve singleton MoE context for ordinary models
GDzhu01 Sep 13, 2026
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
The table of contents is too big for display.
Diff view
Diff view
  •  
  •  
  •  
1 change: 1 addition & 0 deletions .github/workflows/scripts/estimated_times.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -49,6 +49,7 @@ estimated_times:
tests/e2e/pull_request/one_card/test_completion_with_prompt_embeds.py: 170
tests/e2e/pull_request/one_card/test_cpu_offloading.py: 30
tests/e2e/pull_request/one_card/test_cpu_weight_offload.py: 890
tests/e2e/pull_request/one_card/test_deepseek_v4_vision_precision.py: 600
tests/e2e/pull_request/one_card/test_guided_decoding.py: 670
tests/e2e/pull_request/one_card/test_minicpm.py: 300
tests/e2e/pull_request/one_card/test_minimax_m3_sparse_attn.py: 440
Expand Down
112 changes: 112 additions & 0 deletions benchmarks/prepare_indexer_indices.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,112 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
"""Compare fused indexer postprocessing in an NPU graph.

Run: python benchmarks/prepare_indexer_indices.py
Times exclude compilation, graph capture and host tensor allocation. Repetition
counts adapt to a warmup measurement so slow INT32-sort baselines stay bounded.
"""

import argparse
import json
import statistics
from functools import partial

import torch
import torch_npu # noqa: F401

from vllm_ascend.ops.triton.prepare_indexer_indices import prepare_indexer_indices
from vllm_ascend.ops.triton.triton_utils import init_device_properties_triton


def reference_indices(selected, positions, compress_ratio):
visible = ((positions + 1) // compress_ratio).unsqueeze(-1)
valid = (selected >= 0) & (selected < visible)
sentinel = torch.iinfo(torch.int32).max
selected = torch.where(valid, selected, sentinel).sort(dim=-1).values
return torch.where(selected == sentinel, -1, selected)


def graph_latency_us(fn, value):
for _ in range(3):
fn(value)
torch.npu.synchronize()
start = torch.npu.Event(enable_timing=True)
end = torch.npu.Event(enable_timing=True)
start.record()
fn(value)
end.record()
end.synchronize()
estimate_ms = max(start.elapsed_time(end), 0.001)
# Capture at most about 10 ms of work and measure about 50 ms per sample.
batch = min(32, max(1, int(10 / estimate_ms)))
repeats = min(20, max(1, int(50 / (batch * estimate_ms))))
graph = torch.npu.NPUGraph()
with torch.npu.graph(graph, capture_error_mode="thread_local", auto_dispatch_capture=True):
for _ in range(batch):
output = fn(value)
for _ in range(3):
graph.replay()
torch.npu.synchronize()
samples = []
for _ in range(5):
start = torch.npu.Event(enable_timing=True)
end = torch.npu.Event(enable_timing=True)
start.record()
for _ in range(repeats):
graph.replay()
end.record()
end.synchronize()
samples.append(start.elapsed_time(end) * 1000 / (batch * repeats))
# Keep the captured outputs alive until timing completes.
del output
return statistics.median(samples)


def benchmark(stage, value, reference, fused, **shape):
expected, actual = reference(value), fused(value)
if isinstance(expected, torch.Tensor):
expected, actual = (expected,), (actual,)
for output, ref in zip(actual, expected):
torch.testing.assert_close(output, ref, rtol=0, atol=0)
original_us = graph_latency_us(reference, value)
fused_us = graph_latency_us(fused, value)
print(
json.dumps(
{
"stage": stage,
**shape,
"reference_us": original_us,
"triton_us": fused_us,
"speedup": original_us / fused_us,
}
),
flush=True,
)


@torch.inference_mode()
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--tokens", type=int, nargs="+", default=[1, 32, 256, 4096])
parser.add_argument("--topk", type=int, nargs="+", default=[128, 2048])
args = parser.parse_args()
torch.npu.set_device(0)
init_device_properties_triton()
torch.manual_seed(41)
for tokens in args.tokens:
for topk in args.topk:
selected = torch.randint(-1, 4096, (tokens, topk), dtype=torch.int32, device="npu")
positions = torch.full((tokens,), 4095, dtype=torch.int64, device="npu")
benchmark(
"indices",
selected,
partial(reference_indices, positions=positions, compress_ratio=2),
partial(prepare_indexer_indices, positions=positions, compress_ratio=2),
tokens=tokens,
topk=topk,
)


if __name__ == "__main__":
main()
83 changes: 83 additions & 0 deletions benchmarks/quantize_indexer_query.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,83 @@
# SPDX-License-Identifier: Apache-2.0
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
"""Compare indexer query quantization latency inside an NPU graph.

Run: python benchmarks/quantize_indexer_query.py
Times exclude compilation, graph capture and host tensor allocation.
"""

import argparse
import json
import statistics

import torch
import torch_npu # noqa: F401

from vllm_ascend.ops.triton.quantize_indexer_query import quantize_indexer_query
from vllm_ascend.ops.triton.triton_utils import init_device_properties_triton


def reference(query):
scale = (query.float().abs().amax(-1) / 127.0).half().clamp_min_(2.0**-24)
quantized = (query.float() / scale.float().unsqueeze(-1)).round().clamp(-127, 127).to(torch.int8)
return quantized, scale


def graph_latency_us(fn, query, batch=32, repeats=20):
for _ in range(3):
fn(query)
torch.npu.synchronize()
graph = torch.npu.NPUGraph()
with torch.npu.graph(graph, capture_error_mode="thread_local", auto_dispatch_capture=True):
for _ in range(batch):
output = fn(query)
for _ in range(3):
graph.replay()
torch.npu.synchronize()
samples = []
for _ in range(5):
start = torch.npu.Event(enable_timing=True)
end = torch.npu.Event(enable_timing=True)
start.record()
for _ in range(repeats):
graph.replay()
end.record()
end.synchronize()
samples.append(start.elapsed_time(end) * 1000 / (batch * repeats))
# Keep the captured outputs alive until timing completes.
assert output[0].shape == query.shape
return statistics.median(samples)


@torch.inference_mode()
def main():
parser = argparse.ArgumentParser(description=__doc__)
parser.add_argument("--tokens", type=int, nargs="+", default=[1, 32, 256, 4096])
parser.add_argument("--heads", type=int, nargs="+", default=[32, 64])
args = parser.parse_args()
torch.npu.set_device(0)
init_device_properties_triton()
torch.manual_seed(41)
for tokens in args.tokens:
for heads in args.heads:
query = torch.randn(tokens, heads, 128, dtype=torch.bfloat16, device="npu")
for actual, expected in zip(quantize_indexer_query(query), reference(query)):
torch.testing.assert_close(actual, expected, rtol=0, atol=0)
original_us = graph_latency_us(reference, query)
fused_us = graph_latency_us(quantize_indexer_query, query)
print(
json.dumps(
{
"tokens": tokens,
"heads": heads,
"reference_us": original_us,
"triton_us": fused_us,
"speedup": original_us / fused_us,
}
),
flush=True,
)


if __name__ == "__main__":
main()
3 changes: 2 additions & 1 deletion csrc/attention/common/op_kernel/aicpu_common.h
Original file line number Diff line number Diff line change
Expand Up @@ -16,6 +16,7 @@
#ifndef AICPU_COMMON_H
#define AICPU_COMMON_H

#include <algorithm>
#include <cstdint>
#include <vector>
#include "log.h"
Expand Down Expand Up @@ -103,7 +104,7 @@ inline bool IsTensorExists(const Tensor *tensor)

inline std::vector<int64_t> GetTensorDataAsInt64(const Tensor *tensor)
{
std::vector<int64_t> result {};
std::vector<int64_t> result{};

if (!IsTensorExists(tensor)) {
return result;
Expand Down
Loading
Loading