MCPcopy Create free account

hub / github.com/brontoguana/krasis / functions

Functions12,380 in github.com/brontoguana/krasis

↓ 3 callersFunction_unsupported_torch_devices
Return visible devices whose SM arch is missing from installed torch.
python/krasis/setup.py:366
↓ 3 callersFunction_variant_metric
(variant: Dict[str, Any], *keys: str, default: float = 0.0)
python/krasis/hqq_self_calibrate.py:4670
↓ 3 callersFunction_vector_delta_summary
( expected: Optional[Sequence[float]], actual: Optional[Sequence[float]], )
tests/trace_diff.py:307
↓ 3 callersMethod_visible_config_options
Return config options visible in the current TUI mode.
python/krasis/launcher.py:1495
↓ 3 callersFunction_visible_len
Length of string with ANSI escape codes stripped.
python/krasis/launcher.py:71
↓ 3 callersMethodadd_pointer_offset
Adds a pointer offset in units of Element
src/cuda/flash_attn/cutlass/cutlass/transform/threadblock/ell_predicated_tile_access_iterator.h:870
↓ 3 callersMethodadd_pointer_offset
Adds a pointer offset in units of Element
src/cuda/flash_attn/cutlass/cutlass/transform/threadblock/regular_tile_access_iterator_tensor_op.h:293
↓ 3 callersMethodadd_tile_offset
Advances an iterator along logical dimensions of matrix in units of whole tiles
src/cuda/flash_attn/cutlass/cutlass/transform/threadblock/ell_predicated_tile_access_iterator.h:877
↓ 3 callersMethodadd_tile_offset
Adds a tile offset
src/cuda/flash_attn/cutlass/cutlass/transform/threadblock/regular_tile_access_iterator_tensor_op.h:305
↓ 3 callersFunctionaggregate_candidate_groups
(groups: list[dict[str, Any]])
tests/hqq_attention_diff.py:1787
↓ 3 callersFunctionany_of
src/cuda/flash_attn/cutlass/cute/algorithm/tuple_algorithms.hpp:304
↓ 3 callersFunctionappend_safety_limit_dump
( kind: &str, device_id: i32, free_mb: u64, safety_margin_mb: u64, deficit_mb: u64, )
src/vram_monitor.rs:242
↓ 3 callersMethodapply_saved
Apply loaded config values.
python/krasis/launcher.py:607
↓ 3 callersMethodarrive_and_expect_tx
Performs an arrive operation + expected transaction bytes increment
src/cuda/flash_attn/cutlass/cutlass/arch/barrier.h:541
↓ 3 callersFunctionarrive_internal
src/cuda/flash_attn/cutlass/cutlass/arch/barrier.h:301
↓ 3 callersMethodas_bf16
Return as BF16 ref (panics if not BF16).
src/weights/mod.rs:311
↓ 3 callersFunctionattention_quant_label
(attention_quant: str)
python/krasis/attention_backend.py:255
↓ 3 callersFunctionbasis_value
src/cuda/flash_attn/cutlass/cute/numeric/arithmetic_tuple.hpp:260
↓ 3 callersMethodbegin_row
src/cuda/flash_attn/cutlass/cutlass/epilogue/threadblock/fusion/visitor_store.hpp:315
↓ 3 callersFunctionbenchmark_decode_detailed
Detailed decode benchmark with per-token timings.
tests/archive/test_v2lite_timing.py:77
↓ 3 callersFunctionblock
src/cuda/flash_attn/cutlass/cute/util/debug.hpp:121
↓ 3 callersFunctionbuild_contract
( model_name: str, model_path: str, tokenizer: Any, *, max_new_tokens: int, add_genera
tests/reference_contract.py:533
↓ 3 callersFunctionbuild_prefill_diagnostic_from_logits
( tokenizer: Any, prompt_input_ids: Any, generated_token_ids: List[int], step_logits: Any,
tests/generate_reference.py:1017
↓ 3 callersFunctionbuild_prompt
Build a prompt of approximately target_tokens length using Gutenberg text.
tests/archive/test_qwen235b_bench.py:46
↓ 3 callersFunctionbuild_reference_artifact_metadata
( reference: Dict[str, Any], reference_contract: Dict[str, Any], reference_path: Optional[str], )
tests/reference_contract.py:667
↓ 3 callersFunctionbuild_teacher_forced_per_token_data
( model: Any, tokenizer: Any, prompt_input_ids: Any, generated_token_ids: List[int], *,
tests/generate_reference.py:959
↓ 3 callersFunctionbuild_token_diagnostic_entry
( tokenizer: Any, step_logits: Any, *, expected_token_id: int, logit_pos: int, prev_to
tests/generate_reference.py:913
↓ 3 callersFunctioncapture_prefill_logits
(model: Any, prompt_input_ids: Any, *, use_cache: Optional[bool] = None)
tests/generate_reference.py:1007
↓ 3 callersFunctioncapture_settings_for_profile
(profile_id: str)
tests/reference_contract.py:462
↓ 3 callersMethodclear_mask
Clears the predicate set efficiently
src/cuda/flash_attn/cutlass/cutlass/transform/threadblock/ell_predicated_tile_access_iterator.h:924
↓ 3 callersFunctioncommand_output
(args: list[str])
scripts/build_sidecars.py:111
↓ 3 callersFunctioncompare
(name, cpu_t, gpu_t, n=8)
tests/test_gemv_verify.py:34
↓ 3 callersFunctioncompare
(name, ref_t, test_t, n=8)
tests/archive/test_moe_accum_verify.py:31
↓ 3 callersFunctioncompare_init
(name: str, chunk: torch.Tensor)
tests/test_hqq_rust_quantizer.py:94
↓ 3 callersFunctioncompare_real_artifact_registration_layer
( name: str, real_case: dict, selected_tensor_names: list[str] | None = None, )
tests/test_hqq_rust_quantizer.py:1558
↓ 3 callersFunctioncomplement
src/cuda/flash_attn/cutlass/cute/layout.hpp:1178
↓ 3 callersFunctioncompute_hqq4_rmse_rust
Rust shadow RMSE path for one HQQ4 group chunk.
python/krasis/attention_backend.py:1421
↓ 3 callersMethodconvert
src/cuda/flash_attn/cutlass/cutlass/half.h:195
↓ 3 callersMethodconvert_from_float
src/cuda/flash_attn/cutlass/cutlass/float8.h:1071
↓ 3 callersFunctioncooperative_gemm_predication
src/cuda/flash_attn/cutlass/cute/algorithm/cooperative_gemm.hpp:166
↓ 3 callersFunctioncopy_unpack
src/cuda/flash_attn/cutlass/cute/atom/copy_traits.hpp:111
↓ 3 callersFunctioncrd2crd
src/cuda/flash_attn/cutlass/cute/stride.hpp:254
↓ 3 callersFunctioncreate_bundle
(path: Path)
scripts/build_sidecars.py:477
↓ 3 callersFunctioncreate_run_dir
(run_type: str, parent: Optional[os.PathLike] = None)
python/krasis/run_paths.py:35
↓ 3 callersFunctioncuda_alloc_f32
(count: usize)
src/hqq.rs:442
↓ 3 callersFunctioncuda_free
(ptr: cuda_sys::CUdeviceptr)
src/hqq.rs:459
↓ 3 callersFunctiondequant_bf16
(data: &[u8], n: usize)
src/gguf.rs:559
↓ 3 callersFunctiondequant_ct
Vectorized dequantize compressed-tensors INT4 to FP32.
tests/test_rust_vs_python.py:40
↓ 3 callersFunctiondequant_ct_vectorized
Vectorized dequantization of compressed-tensors INT4. Each int32 word contains 8 nibbles: nibble_j at bits [4j, 4j+3]. Dequantized value = (n
tests/archive/test_prequant_compare.py:44
↓ 3 callersFunctiondequant_f32
(data: &[u8], n: usize)
src/gguf.rs:536
↓ 3 callersFunctiondequantize_raw_data
Dequantize raw GGUF-format bytes to FP32. Accepts raw byte data + GGML type + element count. Used by the GGUF→AVX2 cache builder to dequantize indivi
src/gguf.rs:872
↓ 3 callersMethoddispatch_matmul
Dispatch matmul to correct INT4/INT8 kernel.
src/decode.rs:1502
↓ 3 callersMethoddiv
Computes integer division using precomputed values. This is computationally inexpensive.
src/cuda/flash_attn/cutlass/cutlass/fast_math.h:389
↓ 3 callersFunctiondtype_name
(tensor: torch.Tensor)
tests/test_hqq_fused_branch_runtime.py:33
↓ 3 callersMethodell_add_mask
add mask for small tiles in ELL
src/cuda/flash_attn/cutlass/cutlass/transform/threadblock/ell_predicated_tile_access_iterator.h:940
↓ 3 callersFunctionemit_contract_trace
( context: str, reference_contract: Dict[str, Any], runtime_contract: Dict[str, Any], validati
tests/reference_contract.py:1142
↓ 3 callersMethodenable
< CUTLASS_HOST_DEVICE enables all accesses guarded by mask
src/cuda/flash_attn/cutlass/cutlass/epilogue/threadblock/predicated_tile_iterator.h:179
↓ 3 callersMethodenable_mask
Clears the predicate set efficiently
src/cuda/flash_attn/cutlass/cutlass/transform/threadblock/ell_predicated_tile_access_iterator.h:928
↓ 3 callersMethodend_row
src/cuda/flash_attn/cutlass/cutlass/epilogue/threadblock/fusion/visitor_store.hpp:341
↓ 3 callersMethodend_step
Called after all accumulator elements have been visited
src/cuda/flash_attn/cutlass/cutlass/epilogue/threadblock/fusion/visitor_2x.hpp:111
↓ 3 callersMethodevict_page_cache
Evict all pages from page cache. Call after all data has been copied out.
src/weights/safetensors_io.rs:210
↓ 3 callersFunctionevidence_trace_log_path
(output_path: str)
python/krasis/hqq_self_calibrate.py:387
↓ 3 callersFunctionexpected_marlin_cache_size
Expected total Marlin cache file size.
src/weights/mod.rs:1719
↓ 3 callersFunctionexpert_matmul_int4
( packed: *const u32, // [N, K/8] packed INT4 scales: *const u16, // [N, K/group_size] BF16 sc
src/kernel/avx2.rs:105
↓ 3 callersFunctionexpert_matmul_int4_integer
( packed: *const u32, // [N, K/8] packed INT4 weight_scales: *const u16, // [N, K/group_size] B
src/kernel/avx2.rs:366
↓ 3 callersFunctionexpert_matmul_int4_marlin
AVX2 Marlin-native INT4 matmul — production kernel. Processes Marlin tiles: for each (K-group × N-chunk-of-64), unpacks 1024 INT4 values from 128 u32
src/kernel/avx2.rs:1965
↓ 3 callersFunctionexpert_matmul_int4_transposed
( packed: *const u32, // [K/8, n_stride] scales: *const u16, // [K/group_size, n_stride] a
src/kernel/avx2.rs:868
↓ 3 callersFunctionexpert_matmul_int8_integer
( data: *const i8, // [N, K] raw INT8 weights weight_scales: *const u16, // [N, K/group_size
src/kernel/avx2.rs:656
↓ 3 callersFunctionextract_server_error
Extract the tail of the server log for error reporting.
tests/release_test.py:938
↓ 3 callersFunctionf32_to_bf16
(v: f32)
tests/test_marlin_attn_shapes.rs:22
↓ 3 callersFunctionfast_exp_avx2_inline
(x: __m256)
src/decode.rs:2004
↓ 3 callersFunctionfill_workspace
src/cuda/flash_attn/cutlass/cutlass/workspace.h:97
↓ 3 callersFunctionfind
src/cuda/flash_attn/cutlass/cute/container/tuple.hpp:254
↓ 3 callersFunctionfind_krasis_command
Find the installed krasis command. Fails if not found. Prefers the conda env krasis because that's the environment with GPU dependencies (PyT
tests/release_test.py:541
↓ 3 callersFunctionfind_nvcc
()
build.rs:300
↓ 3 callersMethodfinish
(self)
build.rs:62
↓ 3 callersFunctionflatten
src/cuda/flash_attn/cutlass/cute/tensor_impl.hpp:582
↓ 3 callersFunctionfnv1a
FNV-1a hash for cache invalidation.
src/weights/mod.rs:1215
↓ 3 callersMethodforward
Forward pass for one layer. Args: hidden: [M, hidden_size] BF16 residual: [M, hidden_size] BF16 or None (first layer)
python/krasis/layer.py:313
↓ 3 callersFunctiongemm
src/cuda/flash_attn/fa2/utils.h:139
↓ 3 callersFunctiongen
(prompt, num_tokens=10)
tests/test_gguf_native.py:33
↓ 3 callersFunctiongenerate
(model, prompt, max_new_tokens=64, temperature=0.6, top_k=50, top_p=0.95)
tests/archive/run_length_sweep.py:20
↓ 3 callersFunctionget
src/cuda/flash_attn/cutlass/cute/layout.hpp:497
↓ 3 callersMethodget
Returns a pointer
src/cuda/flash_attn/cutlass/cutlass/transform/threadblock/ell_predicated_tile_access_iterator.h:883
↓ 3 callersMethodget
Returns a pointer
src/cuda/flash_attn/cutlass/cutlass/transform/threadblock/regular_tile_access_iterator_tensor_op.h:299
↓ 3 callersFunctiongetParams
src/cuda/flash_attn/cutlass/cutlass/conv/threadblock/conv2d_fprop_activation_tile_access_iterator_analytic.h:177
↓ 3 callersMethodget_batch_idx
Obtains the batch index
src/cuda/flash_attn/cutlass/cutlass/gemm/threadblock/threadblock_swizzle_streamk.h:653
↓ 3 callersFunctionget_grid_shape
src/cuda/flash_attn/cutlass/cutlass/transform/kernel/sm90_sparse_gemm_compressor.hpp:183
↓ 3 callersMethodget_grid_shape
src/cuda/flash_attn/cutlass/cutlass/gemm/kernel/sm100_tile_scheduler.hpp:204
↓ 3 callersMethodget_mask
Gets the mask
src/cuda/flash_attn/cutlass/cutlass/transform/threadblock/ell_predicated_tile_access_iterator.h:936
↓ 3 callersMethodget_mask
< Sets the mask
src/cuda/flash_attn/cutlass/cutlass/epilogue/threadblock/predicated_tile_iterator.h:1060
↓ 3 callersFunctionget_max_cta_occupancy
src/cuda/flash_attn/cutlass/cutlass/gemm/kernel/tile_scheduler_params.h:54
↓ 3 callersFunctionget_scalar_type
Get ScalarType for Marlin quantization config.
python/krasis/marlin_utils.py:66
↓ 3 callersFunctionget_shard_for_tensor
(model_path, tensor_name)
tests/archive/test_prequant_compare.py:36
↓ 3 callersMethodget_stride
src/cuda/flash_attn/cutlass/cutlass/transform/threadblock/ell_predicated_tile_access_iterator.h:893
↓ 3 callersFunctiongguf_matvec_int
INT16-activation matvec dispatch (uses AVX2 for Q4_K, Q8_0, Q4_0).
src/gguf_kernels.rs:177
↓ 3 callersFunctiongptq_marlin_gemm
Call vendored Marlin GEMM kernel via ctypes. Matches sgl_kernel.gptq_marlin_gemm API for drop-in replacement.
python/krasis/marlin_utils.py:137
↓ 3 callersFunctiongpu_mem_mb
Current GPU memory usage per device in MB.
tests/archive/meta_optimiser.py:76
↓ 3 callersMethodhas_cpu_weights
Whether CPU decode weights (transposed format) have been populated.
src/weights/mod.rs:4250
↓ 3 callersFunctionhf_login
(token: str)
python/krasis/hf_downloader.py:271
← previousnext →1,201–1,300 of 12,380, ranked by callers