Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
33 commits
Select commit Hold shift + click to select a range
ebf4e8a
Add Laya CUDA resource ownership
Sep 30, 2026
50591ae
Combine checkpoint and CUDA resource dependencies
Sep 30, 2026
ff8d017
Keep Laya checkpoint weights resident on CUDA
Sep 30, 2026
4b2c0d4
Allocate bounded Laya inference workspaces
Sep 30, 2026
ad1c7b8
Run Laya encoder and decision layers through native CUDA
Sep 30, 2026
5b5518b
Check encoder reuse without intermediate synchronization
Sep 30, 2026
0c50122
Clarify eager encoder inference scope
Sep 30, 2026
d37da99
Document separate CUDA operator loading
Sep 30, 2026
684470a
Validate distinct mixed requests at maximum encoder batch
Sep 30, 2026
0a349f3
Validate rotary sizes before loading kernels and honor GPU selection
Sep 30, 2026
506b4a1
Merge main into CUDA resource branch
Oct 1, 2026
90cb4a1
Merge checkpoint update into Laya GPU dependencies
Oct 1, 2026
fb4316f
Merge updated CUDA resources into Laya GPU dependencies
Oct 1, 2026
b8357d4
Merge updated dependencies into Laya workspace
Oct 1, 2026
7208c8b
Merge updated workspace dependencies into Laya encoder
Oct 1, 2026
a20b2da
Move CUDA resource tests to the root test directory
Oct 1, 2026
dbab441
Merge branch 'codex/laya-test-layout' into codex/laya-tests-resident-…
Oct 1, 2026
9ef493e
Merge branch 'codex/laya-tests-runtime' into codex/laya-tests-residen…
Oct 1, 2026
b8f9a33
Merge branch 'codex/laya-tests-resident-base' into codex/laya-tests-w…
Oct 1, 2026
2f5ff4b
Move Laya residency and workspace tests to the root test directory
Oct 1, 2026
dfb6eda
Merge branch 'codex/laya-tests-workspace' into codex/laya-tests-encoder
Oct 1, 2026
ff4d8fa
Move Laya encoder tests to the root test directory
Oct 1, 2026
6ae5e0f
Merge main into Laya PR #48
Oct 3, 2026
565e3fe
Merge updated Laya workspace branch into PR #49
Oct 3, 2026
df85135
docs(laya): publish native encoder validation steps
Oct 3, 2026
7574f19
docs(laya): publish residency and workspace validation
Oct 3, 2026
047d018
docs(laya): include parent resource validation steps
Oct 3, 2026
b7c9384
Merge resource validation docs into the encoder
Oct 3, 2026
d235783
perf(laya): resolve encoder resources and kernel handles at load
Levius-Fubuki Oct 4, 2026
8f58f15
feat(laya): port prepared execution plan onto native worker
Levius-Fubuki Oct 5, 2026
ecfdeaf
fix(laya): preserve sparse bundle attention compatibility
Levius-Fubuki Oct 5, 2026
5833efa
fix(laya): propagate missing attention shape errors
Levius-Fubuki Oct 5, 2026
6043294
ci: validate retargeted Laya host changes against main
Levius-Fubuki Oct 5, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
43 changes: 40 additions & 3 deletions src/backends/cuda/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -7,7 +7,7 @@ use std::{
rc::Rc,
};
pub type Ptr = *mut c_void;
type Kernel = unsafe extern "C" fn(*mut Ptr, i32, i32, i32, Ptr) -> i32;
type Launch = unsafe extern "C" fn(*mut Ptr, i32, i32, i32, Ptr) -> i32;
struct Context {
lib: Library,
stream: Ptr,
Expand Down Expand Up @@ -86,6 +86,15 @@ impl Cuda {
pub fn sync(&self) -> Result<()> {
self.ctx.sync()
}
/// Resolve a kernel once while retaining the owning runtime and native code.
pub fn resolve(&self, name: &str) -> Result<Kernel> {
Ok(Kernel {
ctx: self.ctx.clone(),
launch: self
.ctx
.symbol::<Launch>(format!("laya_{name}\0").as_bytes())?,
})
}
/// # Safety
/// Tensor shape, dtype, layout, aliasing and allocation sizes must match the generated kernel.
/// Buffers must belong to this context and stay alive until synchronization or graph destruction.
Expand All @@ -96,7 +105,7 @@ impl Cuda {
);
let k = self
.ctx
.symbol::<Kernel>(format!("laya_{name}\0").as_bytes())?;
.symbol::<Launch>(format!("laya_{name}\0").as_bytes())?;
self.ctx.check(unsafe {
k(
args.as_ptr() as *mut Ptr,
Expand Down Expand Up @@ -151,7 +160,7 @@ impl Cuda {
) -> Result<()> {
let k = self
.ctx
.symbol::<Kernel>(format!("laya_{name}\0").as_bytes())?;
.symbol::<Launch>(format!("laya_{name}\0").as_bytes())?;
self.ctx.check(unsafe {
k(
args.as_ptr() as *mut Ptr,
Expand Down Expand Up @@ -245,3 +254,31 @@ impl Drop for Graph {
}
}
}

/// A resolved entry point retaining its stream and native library.
#[derive(Clone)]
pub struct Kernel {
ctx: Rc<Context>,
launch: Launch,
}
impl Kernel {
/// # Safety
/// Shapes, dtype, layout, aliasing and pointer lifetimes must match this kernel.
/// Every pointer belongs to this context and stays alive through synchronization
/// or destruction of any graph that captures the launch.
pub unsafe fn launch(&self, args: &[Ptr], b: usize, l: usize) -> Result<()> {
ensure!(
b > 0 && b <= 16 && l > 0 && l <= 512 && l.is_multiple_of(16),
"invalid CUDA shape"
);
self.ctx.check(unsafe {
(self.launch)(
args.as_ptr() as *mut Ptr,
b as i32,
l as i32,
(b * l) as i32,
self.ctx.stream,
)
})
}
}
Loading
Loading