pypto-serving
Provide LLM inference service based on PyPTO
File Explorer
- SKILL.md
- SKILL.md
- SKILL.md
- openai.yaml
- analyze_profile.py
- render_8lane.py
- run_profile.py
- run_profile.sh
- validate_artifact.py
- SKILL.md
- openai.yaml
- analyze_profile.py
- launch_server.py
- render_8lane.py
- run_profile.py
- run_profile.sh
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- SKILL.md
- README.md
- README.md
- skills
- README.md
- skills
- action.yml
- action.yml
- bug_report.yml
- config.yml
- documentation.yml
- feature_request.yml
- ci.yml
- docs.yml
- __init__.py
- repo_links.py
- deepseek-v4-conversion.md
- index.md
- pypto-prepack-deepseek-v4.md
- pypto-serving.md
- architecture.md
- deepseek-v4-dspark.md
- deepseek-v4-runtime.md
- model-integration.md
- weight-staging.md
- installation.md
- quickstart.md
- benchmarking.md
- deepseek-v4.md
- offline-inference.md
- online-serving.md
- parallel.md
- profile.md
- qwen.md
- index.md
- requirements.txt
- theme-revision.txt
- channels.md
- engine.md
- README.md
- README.md
- engine.cpp
- meson.build
- meson.build
- HiCR
- TaskR
- module.hpp
- subscription.hpp
- base.hpp
- input.hpp
- message.hpp
- messageTypeRegistry.hpp
- output.hpp
- engine.hpp
- .clang-format
- meson.build
- meson_options.txt
- __init__.py
- __main__.py
- main.py
- __init__.py
- parallel.py
- types.py
- __init__.py
- compiler.py
- l3_callable.py
- __init__.py
- executor.py
- pypto_executor.py
- sampler.py
- utils.py
- __init__.py
- buffer_set.py
- l3_dispatch.py
- model_runner.py
- task_args.py
- __init__.py
- packer.py
- pipeline.py
- shard.py
- spec.py
- stacker.py
- store.py
- __init__.py
- __init__.py
- encoding.py
- npu_executor.py
- npu_runner.py
- task_args.py
- weight_loader.py
- weight_spec.py
- __init__.py
- npu_executor.py
- npu_runner.py
- task_args.py
- weight_loader.py
- weight_spec.py
- __init__.py
- npu_executor.py
- npu_runner.py
- task_args.py
- weight_spec.py
- __init__.py
- model_family.py
- model_loader.py
- tokenizer.py
- __init__.py
- async_engine.py
- __init__.py
- kv_cache.py
- __init__.py
- scheduler.py
- __init__.py
- ipc.py
- server.py
- serving_worker.py
- streamer.py
- __init__.py
- env.py
- gc_utils.py
- prefill.py
- __init__.py
- __init__.py
- env.py
- merge.py
- recorder.py
- __init__.py
- prepack_deepseek_v4.py
- __init__.py
- worker.py
- __init__.py
- requirements.txt
- convert_deepseek_v4_to_w8a8.py
- merge_profile.sh
- check_docs_nav.py
- check_english_only.py
- check_headers.py
- check_public_docs.py
- test_parallel_options.py
- test_parallel_config.py
- test_task_args.py
- test_weight_pipeline.py
- test_weight_store.py
- conftest.py
- test_model_components.py
- test_weight_pack_parity.py
- test_weight_sidecar_parity.py
- test_dspark_model.py
- test_npu_executor.py
- test_npu_runner_inputs.py
- test_qwen_weight_staging.py
- conftest.py
- test_tokenizer.py
- __init__.py
- test_async_engine_replicas.py
- test_async_pipeline.py
- test_output_delivery.py
- test_request_cleanup.py
- test_async_scheduler.py
- __init__.py
- test_streaming_usage.py
- test_worker_sampling.py
- test_worker_step_protocol.py
- __init__.py
- device_sampling_fakes.py
- test_profiling.py
- __init__.py
- __init__.py
- test_deepseek_dspark_accuracy.py
- test_deepseek_v4_accuracy.py
- test_deepseek_v4_conversion.py
- test_kernel_compiler.py
- test_qwen3_accuracy.py
- test_qwen3_serving.py
- .gitignore
- .gitmodules
- .pre-commit-config.yaml
- AGENTS.md
- mkdocs.yml
- pyproject.toml
- pypto-lib
- README.md
- ruff.toml
# Use via CDN
jsDelivrjsDelivr serves any public GitHub repository as a CDN with zero setup. Pick a version and a file to get a ready-to-paste link and snippet.
Link
Example
// repository documentation
Was this content helpful?
(0 ratings)
