opsd-predictive-law
A Predictive Law for On-Policy Self-Distillation From World Feedback (ICML RLxF 2026)
File Explorer
- code.py
- gpqa.py
- math.py
- mmlu_pro.py
- prompts.py
- sciknoweval.py
- train.py
- utils.py
- code.py
- code_elo.py
- codeforces.py
- data_handling.py
- humaneval.py
- livecodebench.py
- math.py
- mbpp.py
- primeintellect.py
- taco.py
- load_dataset.py
- prepare_lcb.sh
- preprocess.py
- split_tests.py
- run_eval_self_teacher_gap_1step.sh
- run_sdpo_lcb.sh
- fig1_predictive_law.png
- fig2_convergence.png
- fig3_scaling.png
- verl_training.sh
- __init__.py
- base.py
- nccl_checkpoint_engine.py
- nixl_checkpoint_engine.py
- README.md
- __init__.py
- agent_loop.py
- prometheus_utils.py
- single_turn_agent_loop.py
- tool_agent_loop.py
- tool_parser.py
- utils.py
- __init__.py
- sampler.py
- __init__.py
- dynamicgen_dataset.py
- __init__.py
- agent_loop.py
- partial_single_turn_agent_loop.py
- partial_tool_agent_loop.py
- fully_async_ppo_megatron_trainer.yaml
- fully_async_ppo_trainer.yaml
- dapo_30b_a3b_base_math_fsdp.sh
- dapo_7b_async_retool.sh
- dapo_7b_math_fsdp2_16_16.sh
- dapo_7b_math_fsdp2_32_32.sh
- dapo_7b_math_fsdp2_4_12.sh
- dapo_7b_math_fsdp2_4_4.sh
- dapo_7b_math_fsdp2_64_64.sh
- dapo_7b_math_fsdp2_64_64_mis.sh
- dapo_7b_math_fsdp2_8_8.sh
- geo3k_qwen25vl_7b_megatron_4_4.sh
- grpo_30b_a3b_base_math_megatron_96_32.sh
- grpo_30b_a3b_base_math_megatron_96_32_mis.sh
- runtime_env.yaml
- simple_streaming_demo.py
- __init__.py
- vllm_async_server.py
- checkpoint_engine.py
- detach_utils.py
- fsdp2_utils.py
- fsdp_workers.py
- fully_async_main.py
- fully_async_rollouter.py
- fully_async_trainer.py
- megatron_utils.py
- megatron_worker.py
- message_queue.py
- param_sync.py
- ray_trainer.py
- README.md
- README_zh.md
- __init__.py
- agent_loop.py
- one_step_off_ppo_megatron_trainer.yaml
- one_step_off_ppo_trainer.yaml
- dapo_7b_math_fsdp2_4_12.sh
- dapo_7b_math_fsdp2_64_64.sh
- dapo_7b_math_fsdp2_64_64_ris.sh
- dapo_7b_math_fsdp2_colocate.sh
- dapo_7b_math_fsdp2_sglang_4_12.sh
- dapo_7b_math_fsdp2_sglang_colocate.sh
- dapo_7b_math_megatron_4_12.sh
- dapo_7b_math_megatron_colocate.sh
- grpo_0.6b_gsm8k_fsdp2_2_6.sh
- grpo_0.6b_gsm8k_fsdp2_sglang_2_6.sh
- grpo_3b_gsm8k_fsdp2_2_6.sh
- grpo_qwen3_8b_gsm8k_fsdp2_8_8_npu.sh
- distributed_utils.py
- fsdp_workers.py
- main_ppo.py
- megatron_workers.py
- ray_trainer.py
- README.md
- utils.py
- __init__.py
- base.py
- dapo.py
- limited.py
- naive.py
- registry.py
- remote.py
- inner_sglang_router.py
- naive_router.py
- __init__.py
- reward_loop.py
- reward_model.py
- transfer_queue_ppo_megatron_trainer.yaml
- transfer_queue_ppo_trainer.yaml
- agent_loop.py
- main_ppo.py
- ray_trainer.py
- run_qwen3-8b_transferqueue.sh
- rob_ppo_trainer.yaml
- __init__.py
- isaac_env.py
- __init__.py
- libero_env.py
- utils.py
- venv.py
- __init__.py
- action_utils.py
- __init__.py
- configuration_prismatic.py
- constants.py
- modeling_prismatic.py
- processing_prismatic.py
- train_utils.py
- dp_rob.py
- env_loop.py
- fsdp_workers.py
- main_ppo.py
- naive_rollout_rob.py
- prepare_libero_dataset.py
- README.md
- requirements_vla.txt
- rob_ray_trainer.py
- run_simpleVLA_isaac_disagg.sh
- run_simpleVLA_libero_grpo.sh
- __init__.py
- __init__.py
- interaction_registry.py
- __init__.py
- base.py
- gsm8k_interaction.py
- weather_interaction.py
- __init__.py
- __main__.py
- base_model_merger.py
- fsdp_model_merger.py
- megatron_model_merger.py
- __init__.py
- llama_loader.py
- llama_loader_depracated.py
- llama_saver.py
- __init__.py
- parallel_attention.py
- parallel_decoder.py
- parallel_linear.py
- parallel_mlp.py
- parallel_rmsnorm.py
- __init__.py
- modeling_llama_megatron.py
- __init__.py
- __init__.py
- attention.py
- model.py
- rope_utils.py
- vision_config.py
- vision_model.py
- vision_transformer_block.py
- __init__.py
- bridge.py
- config_converter.py
- loader.py
- mbridge.py
- model_forward.py
- model_forward_1f1b_overlap.py
- model_forward_fused.py
- model_initializer.py
- patch_v012.py
- readme.md
- registry.py
- saver.py
- util.py
- weight_converter.py
- __init__.py
- qwen2_loader.py
- qwen2_loader_depracated.py
- qwen2_saver.py
- __init__.py
- parallel_attention.py
- parallel_decoder.py
- parallel_linear.py
- parallel_mlp.py
- parallel_rmsnorm.py
- __init__.py
- modeling_qwen2_megatron.py
- __init__.py
- __init__.py
- apertus.py
- dense_common.py
- glm4v.py
- kimi_vl.py
- llama.py
- monkey_patch.py
- npu_patch.py
- qwen2.py
- qwen2_vl.py
- qwen3_vl.py
- tiled_mlp.py
- __init__.py
- README.md
- registry.py
- weight_loader_registry.py
- __init__.py
- decorator.py
- worker.py
- worker_group.py
- __init__.py
- base.py
- __init__.py
- __init__.py
- state_dict.py
- __init__.py
- _state_dict_utils.py
- __init__.py
- __init__.py
- __init__.py
- McpClientManager.py
- utils.py
- __init__.py
- search_r1_like_utils.py
- tool_registry.py
- __init__.py
- base_tool.py
- geo3k_tool.py
- gsm8k_tool.py
- image_zoom_in_tool.py
- mcp_base_tool.py
- mcp_search_tool.py
- sandbox_fusion_tools.py
- schemas.py
- search_tool.py
- actor.yaml
- dp_actor.yaml
- megatron_actor.yaml
- rollout_correction.yaml
- critic.yaml
- dp_critic.yaml
- megatron_critic.yaml
- legacy_data.yaml
- fsdp.yaml
- megatron.yaml
- veomni.yaml
- hf_model.yaml
- npu_profile.yaml
- fsdp.yaml
- megatron.yaml
- veomni.yaml
- profiler.yaml
- dp_ref.yaml
- megatron_ref.yaml
- ref.yaml
- dp_reward_loop.yaml
- dp_reward_model.yaml
- megatron_reward_loop.yaml
- megatron_reward_model.yaml
- reward_model.yaml
- rollout.yaml
- __init__.py
- _generated_ppo_megatron_trainer.yaml
- _generated_ppo_trainer.yaml
- algorithm.py
- baseline_grpo.yaml
- config.py
- evaluation.yaml
- generation.yaml
- ppo_megatron_trainer.yaml
- ppo_trainer.yaml
- reward_manager.yaml
- sdpo.yaml
- sft_trainer.yaml
- sft_trainer_engine.yaml
- user.yaml
- __init__.py
- core_algos.py
- metric_utils.py
- prefix_grouper_utils.py
- ray_trainer.py
- reward.py
- rollout_corr_helper.py
- utils.py
- __init__.py
- constants_ppo.py
- fsdp_sft_trainer.py
- main_eval.py
- main_generation.py
- main_generation_server.py
- main_ppo.py
- runtime_env.yaml
- sft_trainer.py
- sft_trainer_ray.py
- __init__.py
- checkpoint_handler.py
- checkpoint_manager.py
- fsdp_checkpoint_manager.py
- megatron_checkpoint_manager.py
- __init__.py
- dataset_utils.py
- multiturn_sft_dataset.py
- README.md
- rl_dataset.py
- rm_dataset.py
- sft_dataset.py
- vision_utils.py
- __init__.py
- metrics.py
- performance.py
- trajectory_tracker.py
- __init__.py
- torch_functional.py
- __init__.py
- kernels.py
- linear_cross_entropy.py
- __init__.py
- aggregate_logger.py
- __init__.py
- dist_checkpointing.py
- memory.py
- optimizer.py
- pipeline_parallel.py
- router_replay_patch.py
- router_replay_utils.py
- sequence_parallel.py
- tensor_parallel.py
- __init__.py
- utils.py
- __init__.py
- config.py
- empty_annotations.py
- mstx_profile.py
- nvtx_profile.py
- performance.py
- profile.py
- __init__.py
- ray_backend.py
- __init__.py
- code.py
- gpqa.py
- math.py
- mcq.py
- mmlu_pro.py
- tooluse.py
- __init__.py
- README.md
- testing_util.py
- utils.py
- __init__.py
- grader.py
- math_normalize.py
- __init__.py
- utils.py
- __init__.py
- geo3k.py
- gsm8k.py
- math_batch.py
- math_dapo.py
- math_reward.py
- math_verify.py
- search_r1_like_qa_em.py
- sglang_fp8_utils.py
- __init__.py
- patch.py
- utils.py
- vllm_fp8_utils.py
- __init__.py
- activation_offload.py
- attention_utils.py
- chat_template.py
- config.py
- device.py
- distributed.py
- flops_counter.py
- fs.py
- fsdp_utils.py
- groupwise.py
- hdfs_io.py
- import_utils.py
- logging_utils.py
- megatron_peft_utils.py
- megatron_utils.py
- memory_buffer.py
- memory_utils.py
- model.py
- net_utils.py
- npu_flash_attn_utils.py
- py_functional.py
- ray_utils.py
- rollout_skip.py
- rollout_trace.py
- seqlen_balancing.py
- tensordict_utils.py
- tokenizer.py
- torch_dtypes.py
- torch_functional.py
- tracking.py
- transferqueue_utils.py
- transformers_compat.py
- ulysses.py
- version
- __init__.py
- base.py
- dp_actor.py
- megatron_actor.py
- __init__.py
- actor.py
- critic.py
- engine.py
- megatron_peft.py
- model.py
- optimizer.py
- reward_model.py
- rollout.py
- __init__.py
- base.py
- dp_critic.py
- megatron_critic.py
- __init__.py
- transformer_impl.py
- utils.py
- __init__.py
- transformer_impl.py
- utils.py
- __init__.py
- transformer_impl.py
- __init__.py
- transformer_impl.py
- utils.py
- __init__.py
- base.py
- utils.py
- __init__.py
- abstract.py
- batch.py
- dapo.py
- naive.py
- prime.py
- registry.py
- __init__.py
- reward_model.py
- __init__.py
- base.py
- __init__.py
- naive_rollout.py
- __init__.py
- async_sglang_server.py
- http_server_engine.py
- sglang_rollout.py
- utils.py
- __init__.py
- utils.py
- vllm_async_server.py
- vllm_rollout.py
- __init__.py
- base.py
- hf_rollout.py
- replica.py
- schemas.py
- tokenizer.py
- utils.py
- __init__.py
- base.py
- fsdp_ulysses.py
- __init__.py
- losses.py
- padding.py
- __init__.py
- engine_workers.py
- fsdp_workers.py
- megatron_workers.py
- __init__.py
- base_config.py
- protocol.py
- py.typed
- .envrc
- .gitignore
- .python-version
- eval_self_teacher_gap_1step.sbatch
- LICENSE
- pyproject.toml
- README.md
- run.sbatch
- uv.lock
# Use via CDN
jsDelivrjsDelivr serves any public GitHub repository as a CDN with zero setup. Pick a version and a file to get a ready-to-paste link and snippet.
Link
Example
// repository documentation
Was this content helpful?
(0 ratings)
