pypto-lib
Tensor-level kernels and end-to-end LLM model implementations built on the PyPTO programming framework
File Explorer
- skills
- benchmarking.md
- language-policy.md
- problem-handling.md
- report.py
- SKILL.md
- SKILL.md
- SKILL.md
- report.py
- SKILL.md
- hint_l1_tile.py
- SKILL.md
- tile_budget.py
- report.py
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- gen_profiling_case.py
- incore_profile.py
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- CLAUDE.md
- action.yml
- action.yml
- action.yml
- bug_report.yml
- config.yml
- documentation.yml
- feature_request.yml
- detect_changes.py
- ci.yml
- daily_ci.yml
- docs.yml
- __init__.py
- base.py
- registry.py
- repo_links.py
- cce-incore-profiling.md
- cube-tile-tuning.md
- debugging.md
- dependency-and-scheduling.md
- incore-simulator-profiling.md
- index.md
- performance-tuning.md
- precision-tuning.md
- ring-heap-and-scope-stats.md
- advanced.md
- beginner.md
- index.md
- intermediate.md
- first-kernel.md
- index.md
- installation.md
- platforms.md
- index.md
- decode_optimization.md
- index.md
- index.md
- index.md
- optimization.md
- paged_attention_pypto.md
- index.md
- cce-extern-kernel.md
- distributed-programming.md
- index.md
- pypto-coding-style.md
- compile-runtime-workflow.md
- golden-harness.md
- index.md
- save-and-replay.md
- index.md
- requirements.txt
- allreduce.py
- gemm_eltwise.py
- multi_proj.py
- topk.py
- hello_world.py
- matmul.py
- gemm.py
- layer_norm.py
- rms_norm.py
- rope.py
- softmax.py
- __init__.py
- runner.py
- spec.py
- validation.py
- config.py
- decode_compressor_ratio128.py
- decode_compressor_ratio4.py
- decode_cp_token_allgather.py
- decode_csa.py
- decode_fwd.py
- decode_hca.py
- decode_indexer.py
- decode_indexer_compressor.py
- decode_layer.py
- decode_metadata.py
- decode_o_proj.py
- decode_sparse_attn_csa.py
- decode_sparse_attn_hca.py
- decode_sparse_attn_swa.py
- decode_swa.py
- dspark_attention.py
- dspark_context_kv.py
- dspark_drafter.py
- dspark_markov.py
- dspark_prefill.py
- dspark_proj.py
- expert_routed.py
- expert_shared.py
- gate.py
- hc_head.py
- hc_post.py
- hc_pre.py
- lm_head.py
- lookup_embedding.py
- markov_head.py
- moe.py
- prefill_compressor_ratio128.py
- prefill_compressor_ratio4.py
- prefill_cp_token_allgather.py
- prefill_csa.py
- prefill_fwd.py
- prefill_hca.py
- prefill_indexer.py
- prefill_indexer_compressor.py
- prefill_layer.py
- prefill_metadata.py
- prefill_o_proj.py
- prefill_sparse_attn.py
- prefill_swa.py
- qkv_proj_rope.py
- rmsnorm.py
- rope_interleave.py
- utils.py
- config.py
- decode_compressor_ratio128.py
- decode_compressor_ratio4.py
- decode_csa.py
- decode_fwd.py
- decode_fwd_mtp.py
- decode_hca.py
- decode_indexer.py
- decode_indexer_compressor.py
- decode_layer.py
- decode_mtp.py
- decode_prepare.py
- decode_sparse_attn_csa.py
- decode_sparse_attn_hca.py
- decode_sparse_attn_swa.py
- decode_swa.py
- expert_routed.py
- expert_shared.py
- gate.py
- hc_head.py
- hc_post.py
- hc_pre.py
- lm_head.py
- lookup_embedding.py
- moe.py
- mtp_projection.py
- prefill_compressor_ratio128.py
- prefill_compressor_ratio4.py
- prefill_cp_csa_draft.py
- prefill_cp_exchange.py
- prefill_cp_fwd_draft.py
- prefill_cp_hca_draft.py
- prefill_cp_layer_draft.py
- prefill_cp_swa_draft.py
- prefill_cp_zigzag.py
- prefill_csa.py
- prefill_fwd.py
- prefill_hca.py
- prefill_indexer.py
- prefill_indexer_compressor.py
- prefill_layer.py
- prefill_mtp.py
- prefill_sparse_attn.py
- prefill_swa.py
- qkv_proj_rope.py
- rmsnorm.py
- rope_interleave.py
- sample.py
- serving_contract.py
- utils.py
- config.py
- decode_attention_csa.py
- decode_attention_hca.py
- decode_attention_swa.py
- decode_compressor_ratio128.py
- decode_compressor_ratio4.py
- decode_fwd.py
- decode_indexer.py
- decode_indexer_compressor.py
- decode_layer.py
- decode_metadata.py
- decode_mtp.py
- decode_sparse_attn.py
- decode_sparse_attn_hca.py
- decode_sparse_attn_swa.py
- expert_routed.py
- expert_shared.py
- gate.py
- hc_head.py
- hc_post.py
- hc_pre.py
- input_pack.py
- lm_head.py
- moe.py
- mtp_projection.py
- prefill_attention_csa.py
- prefill_attention_hca.py
- prefill_attention_swa.py
- prefill_compressor_ratio128.py
- prefill_compressor_ratio4.py
- prefill_fwd.py
- prefill_indexer.py
- prefill_indexer_compressor.py
- prefill_layer.py
- prefill_mtp.py
- prefill_sparse_attn.py
- qkv_proj_rope.py
- rmsnorm.py
- rope_tables.py
- synthetic_token_loop.py
- utils.py
- entry.cpp
- entry.cpp
- kernel_tiling.h
- fai_body.hpp
- metadata_layout.h
- rope_qkv_generated.hpp
- runtime_tensor_compat.hpp
- entry.cpp
- qwen_fai_runtime_tiler.hpp
- arch.hpp
- cross_core_sync.hpp
- local_tensor_buffer.hpp
- resource.hpp
- alignment.hpp
- dependent_false.hpp
- macros.hpp
- block_epilogue.hpp
- block_epilogue_init_outputs.hpp
- block_epilogue_online_softmax.hpp
- block_epilogue_online_softmax_low_prec.hpp
- block_epilogue_rescale_o.hpp
- block_epilogue_rescale_o_low_prec.hpp
- CombineScale.hpp
- copy_gm_to_ub.hpp
- copy_ub_to_gm.hpp
- tile_broadcast_inplace_by_column.hpp
- tile_broadcast_inplace_by_row.hpp
- tile_broadcast_mul.hpp
- tile_broadcast_one_blk.hpp
- tile_cast.hpp
- tile_copy.hpp
- tile_elemwise_add.hpp
- tile_elemwise_mul.hpp
- tile_elemwise_muls.hpp
- tile_swizzle.hpp
- dispatch_policy.hpp
- block_mmad.hpp
- block_mmad_pv.hpp
- block_mmad_pv_decode.hpp
- block_mmad_qk.hpp
- block_mmad_qk_decode.hpp
- copy_gm_to_l1.hpp
- copy_gm_to_ub.hpp
- copy_l0c_to_gm.hpp
- copy_l1_to_bt.hpp
- copy_l1_to_l0a.hpp
- copy_l1_to_l0b.hpp
- copy_ub_to_gm.hpp
- tile_copy.hpp
- tile_copy_tla.hpp
- tile_mmad.hpp
- dispatch_policy.hpp
- gemm_type.hpp
- helper.hpp
- layout.hpp
- matrix.hpp
- vector.hpp
- base_defs.hpp
- coord.hpp
- gemm_coord.hpp
- matrix_coord.hpp
- flash_attention_regular.h
- kernel_common.hpp
- config.py
- constants.py
- contract.py
- decode_fwd.py
- decode_layer_a8w8.py
- decode_ssn_draft.py
- decode_tq_draft.py
- greedy_sample.py
- paged_attention_cce.py
- paged_attention_pypto.py
- prefill_fwd.py
- prefill_fwd_a8w8.py
- prefill_tq_draft.py
- rms_lm_head.py
- rope_qkv_regen.py
- test_paged_attention_cce.py
- test_paged_attention_pypto.py
- topk_select.py
- turboquant_kv.py
- weights.py
- conftest.py
- test_deepseek_v4_flash_serving_contract.py
- test_deepseek_v4_pro_decode_rope.py
- test_deepseek_v4_pro_moe_protocol.py
- test_deepseek_v4_pro_token_loop.py
- test_qwen3_14b_contract.py
- test_repo_links.py
- conftest.py
- test_runner.py
- test_spec.py
- test_validation.py
- check_docs_nav.py
- check_english_only.py
- check_headers.py
- check_public_docs.py
- conftest.py
- .gitignore
- .pre-commit-config.yaml
- AGENTS.md
- mkdocs.yml
- README.md
- ruff.toml
# Use via CDN
jsDelivrjsDelivr serves any public GitHub repository as a CDN with zero setup. Pick a version and a file to get a ready-to-paste link and snippet.
Link
Example
Command Glossary
Commands referenced in this DOCs, explained below.
python
View Details ▼
python
Python language interpreter.
python
Start a REPL (interactive shell):
python {{path/to/file.py}}
Execute a specific Python file:
python -i {{path/to/file.py}}
Execute a specific Python file and start a REPL:
// repository documentation
Was this content helpful?
(0 ratings)
