orbit
Stable and Efficient Reinforcement Learning for Trillion-Parameter LLMs
File Explorer
Download Latest Version (.zip)- bot-slash-lint.yaml
- generate_github_workflows.py
- pr-test.yml
- pr-test.yml.j2
- pre-commit.yml
- release-docs.yaml
- .gitkeep
- license-138438922-4722503.pdf
- orbit-logo.png
- orbit_logo.png
- orbital.png
- orbital.svg
- kimi-memory.png
- kimi-rl-curves.png
- memory-scaling-lines.png
- memory-scaling.png
- orbit-logo.png
- peftarena-results.png
- qwen3-loravsoft.png
- v4flash-memory.png
- v4flash-rl-curves.png
- v4pro-validation.png
- blog-post.css
- common.css
- fonts.css
- syntax.css
- typography.css
- ui.css
- index.html
- orbit-adapter-async-db.html
- orbit_icon.png
- README.md
- run-qwen2_5-0_5b-bf16-math-lora.sh
- run-qwen2_5-0_5b-bf16-math-oft.sh
- run-qwen2_5-3b-bf16-math-lora.sh
- run-qwen2_5-3b-bf16-math-oft.sh
- run-qwen2_5-7b-bf16-openr1-full-b200.sh
- run-qwen2_5-7b-bf16-openr1-full.sh
- run-qwen2_5-7b-bf16-openr1-lora-b200.sh
- run-qwen2_5-7b-bf16-openr1-lora.sh
- run-qwen2_5-7b-bf16-openr1-oft-all.sh
- run-qwen2_5-7b-bf16-openr1-oft-b32-b200.sh
- run-qwen2_5-7b-bf16-openr1-oft-b32-kl-b200.sh
- run-qwen2_5-7b-bf16-openr1-oft-b32-kl.sh
- run-qwen2_5-7b-bf16-openr1-oft-b32.sh
- run-qwen2_5-7b-bf16-openr1-oft-b64-b200.sh
- run-qwen2_5-7b-bf16-openr1-oft-b64.sh
- run-qwen3-30b-a3b-bf16-openr1-full-lr1e6.sh
- run-qwen3-30b-a3b-bf16-openr1-full-lr3e6.sh
- run-qwen3-30b-a3b-bf16-openr1-lora.sh
- run-qwen3-30b-a3b-bf16-openr1-oft-b32.sh
- run-qwen3-30b-a3b-bf16-openr1-oft-b64.sh
- run-qwen3-30b-a3b-instruct-2507-bf16-openr1-lora.sh
- run-qwen3-30b-a3b-instruct-2507-bf16-openr1-oft.sh
- run-qwen3-4b-instruct-2507-bf16-math-oft-async.sh
- run-qwen3-4b-instruct-2507-bf16-math-oft-fully-async.sh
- run-qwen3-4b-instruct-2507-bf16-math-oft.sh
- dsv4-common.sh
- README.md
- run-dsv4-mxfp4-math-oft-flash-debug.sh
- run-dsv4-mxfp4-math-oft-pro-debug.sh
- run-dsv4-mxfp4-openr1-oft-flash.sh
- run-dsv4-mxfp4-openr1-oft-pro.sh
- run-kimi-k25-int4-math-oft-debug6.sh
- run-kimi-k25-int4-math-oft.sh
- run-kimi-k25-int4-openr1-oft.sh
- run-kimi-k25-nvfp4-math-oft.sh
- run-kimi-k26-int4-openr1-oft.sh
- run-qwen3-30b-a3b-fp8-math-oft.sh
- run-qwen3-30b-a3b-instruct-2507-fp8-math-oft.sh
- run-qwen3-30b-a3b-int4-math-oft.sh
- run-qwen3-30b-a3b-nvfp4-math-oft.sh
- run-qwen3-4b-fp8-math-oft.sh
- run-qwen3-4b-int4-math-oft.sh
- run-qwen3-4b-nvfp4-math-oft.sh
- eval_math.sh
- test.jsonl
- test.jsonl
- test.jsonl
- __init__.py
- PS.interp
- PS.tokens
- PSLexer.interp
- PSLexer.py
- PSLexer.tokens
- PSListener.py
- PSParser.py
- __init__.py
- latex2sympy2.py
- data_loader.py
- evaluate.py
- examples.py
- grader.py
- math_eval.py
- model_utils.py
- parser.py
- python_executor.py
- sglang_adapter_utils.py
- trajectory.py
- utils.py
- merge_peft.py
- prepare_eval_checkpoint.py
- LICENSE
- README.md
- eval-math-peft-arena.sh
- README.md
- __init__.py
- load_cuda13_2_orbit_env.sh
- README.md
- __init__.py
- peft_wrap.py
- fake_int4_quant_cuda.cu
- setup.py
- __init__.py
- padding_remover.py
- quantizer_compressed_tensors.py
- quantizer_fp8.py
- quantizer_mxfp8.py
- quantizer_nvfp4.py
- __init__.py
- deepseekv3.py
- glm4.py
- glm4moe.py
- llama.py
- mimo.py
- qwen2.py
- qwen3_5.py
- qwen3_next.py
- qwen3moe.py
- __init__.py
- ipc.py
- nccl.py
- __init__.py
- _gather.py
- _payload.py
- interface.py
- registry.py
- runtime.py
- __init__.py
- broadcast.py
- mixin.py
- p2p.py
- p2p_transfer_utils.py
- __init__.py
- common.py
- hf_weight_iterator_base.py
- hf_weight_iterator_bridge.py
- hf_weight_iterator_direct.py
- update_weight_from_tensor.py
- __init__.py
- actor.py
- arguments.py
- bridge_lora_helpers.py
- bridge_peft_helpers.py
- bridge_provider_overrides.py
- checkpoint.py
- ci_utils.py
- initialize.py
- lora_utils.py
- low_precision_bootstrap.py
- memory_attribution.py
- misc_utils.py
- model.py
- model_provider.py
- model_state_manager.py
- mtp_rl_patches.py
- oft_utils.py
- parallel.py
- peft_offload.py
- peft_utils.py
- replay_utils.py
- runtime_device.py
- sglang.py
- state_mode.py
- tensor_semantics.py
- __init__.py
- arguments.py
- sglang_config.py
- sglang_engine.py
- __init__.py
- ci_utils.py
- cp_utils.py
- data.py
- log_utils.py
- loss.py
- parallel.py
- __init__.py
- __init__.py
- actor_group.py
- placement_group.py
- ray_actor.py
- rollout.py
- train_actor.py
- utils.py
- __init__.py
- base_types.py
- dynamic_sampling_filters.py
- __init__.py
- agentic_tool_call.py
- benchmarkers.py
- multi_turn.py
- single_turn.py
- __init__.py
- generate_endpoint_utils.py
- openai_endpoint_utils.py
- sample_utils.py
- tool_call_utils.py
- __init__.py
- compatibility.py
- eval_logging.py
- inference_rollout_common.py
- inference_rollout_eval.py
- inference_rollout_train.py
- __init__.py
- deepscaler.py
- f1.py
- gpqa.py
- ifbench.py
- math_alignment.py
- math_dapo_utils.py
- math_utils.py
- peft_arena_reward.py
- __init__.py
- linear_trajectory.py
- session_errors.py
- session_server.py
- session_types.py
- sessions.py
- __init__.py
- base_types.py
- data_source.py
- fully_async_rollout.py
- sft_rollout.py
- sglang_rollout.py
- sleep_rollout.py
- __init__.py
- radix_tree.py
- radix_tree_middleware.py
- __init__.py
- router.py
- qwen3.5_fixed.jinja
- qwen3_fixed.jinja
- qwen3_thinking_2507_and_next_fixed.jinja
- __init__.py
- autofix.py
- deepseek_v4.py
- template.py
- tito_tokenizer.py
- token_seq_comparator.py
- __init__.py
- command_utils.py
- __init__.py
- chat_template_verify.py
- mock_sglang_server.py
- mock_tools.py
- mock_trajectories.py
- uvicorn_thread_server.py
- __init__.py
- arguments.py
- async_utils.py
- context_utils.py
- data.py
- distributed_utils.py
- dumper_utils.py
- env_report.py
- environ.py
- eval_config.py
- flops_utils.py
- fp8_kernel.py
- health_monitor.py
- http_utils.py
- iter_utils.py
- logging_utils.py
- mask_utils.py
- megatron_bridge_utils.py
- memory_utils.py
- metric_checker.py
- metric_utils.py
- misc.py
- ppo_utils.py
- processing_utils.py
- profile_utils.py
- prometheus_utils.py
- ray_utils.py
- reloadable_process_group.py
- replay_base.py
- reward_normalization.py
- rocm_checkpoint_writer.py
- seqlen_balancing.py
- tensor_backper.py
- tensorboard_utils.py
- timer.py
- tracking_utils.py
- train_dump_utils.py
- train_metric_utils.py
- training_eta.py
- typer_utils.py
- types.py
- wandb_utils.py
- __init__.py
- __init__.py
- deepseek_v32.py
- glm4.py
- glm4moe.py
- glm4moe_lite.py
- mimo.py
- qwen3_5.py
- qwen3_next.py
- __init__.py
- qwen3_fp8_bridge.py
- __init__.py
- convert_checkpoints.py
- convert_fp8_checkpoint_direct.py
- convert_int4_checkpoint_direct.py
- convert_nvfp4_checkpoint_direct.py
- quantize_to_int4.py
- __init__.py
- __init__.py
- __init__.py
- __init__.py
- README.md
- deepseek-v3-20layer.sh
- deepseek-v3-5layer.sh
- deepseek-v3.sh
- deepseek-v4-flash-debug.sh
- deepseek-v4-flash.sh
- deepseek-v4-pro.sh
- glm4-32B.sh
- glm4-9B.sh
- glm4.5-106B-A12B.sh
- glm4.5-355B-A32B.sh
- glm4.7-flash.sh
- glm5-744B-A40B.sh
- glm5-744B-A40B_20layer.sh
- glm5-744B-A40B_4layer.sh
- gpt-oss-20b.sh
- kimi-k2-thinking.sh
- kimi-k2.sh
- kimi-k25-debug-6layer.sh
- kimi-k25.sh
- kimi-k26.sh
- llama3-8B.sh
- llama3.1-8B-Instruct.sh
- llama3.2-3B-Instruct-amd.sh
- llama3.2-3B-Instruct.sh
- mimo-7B-rl.sh
- moonlight.sh
- qwen2.5-0.5B.sh
- qwen2.5-1.5B.sh
- qwen2.5-32B.sh
- qwen2.5-3B.sh
- qwen2.5-7B-4layer.sh
- qwen2.5-7B.sh
- qwen3-0.6B.sh
- qwen3-1.7B.sh
- qwen3-14B.sh
- qwen3-235B-A22B.sh
- qwen3-30B-A3B-4layer.sh
- qwen3-30B-A3B-5layer.sh
- qwen3-30B-A3B.sh
- qwen3-32B.sh
- qwen3-4B-Instruct-2507-w4a16.sh
- qwen3-4B-Instruct-2507.sh
- qwen3-4B.sh
- qwen3-8B.sh
- qwen3-next-80B-A3B.sh
- qwen3.5-27B.sh
- qwen3.5-35B-A3B.sh
- qwen3.5-4B.sh
- qwen3.5-9B.sh
- README.md
- __init__.py
- indexer.py
- sparse_mla.py
- tilelang_indexer_bwd.py
- tilelang_indexer_fwd.py
- tilelang_sparse_mla_bwd.py
- tilelang_sparse_mla_fwd.py
- __init__.py
- glm5.py
- __init__.py
- cp_utils.py
- glm4.py
- hf_attention.py
- qwen3_5.py
- qwen3_next.py
- __init__.py
- common.sh
- configuration_deepseek_v4.py
- convert_dsv4_hf_to_megatron.sh
- convert_fp8_checkpoint_direct.sh
- convert_int4_checkpoint_direct.sh
- convert_nvfp4_checkpoint_direct.sh
- deepseek_v4_chat_template.jinja
- README.md
- common.sh
- driver.sh
- launcher.sh
- load_cuda13_2_orbit_env.sh
- paths.sh
- preflight.sh
- ray.sh
- tool_env.sh
- wandb.sh
- clean_room_gate.sh
- check_oft_sharding_invariants.py
- inspect_oft_streamed_audit.py
- inspect_peft_wrap_audit.py
- README.md
- __init__.py
- run_compare.py
- test_run_compare.py
- cpu_memory_profiler_visualize.py
- __init__.py
- bench_kimi_int4_oft_gemm_latency.py
- check_checkpoint_parity.py
- check_dsv4_checkpoint_parity.py
- check_dsv4_deepgemm_cross_repo_parity.py
- check_fp8_checkpoint_parity.py
- check_fp8_runtime_parity.py
- check_int4_checkpoint_parity.py
- check_int4_runtime_parity.py
- check_nvfp4_checkpoint_parity.py
- check_nvfp4_runtime_parity.py
- check_runtime_step0_parity.py
- checkpoint_parity_core.py
- checkpoint_parity_utils.py
- convert_checkpoints.py
- convert_dsv4_hf_to_megatron.py
- convert_fp8_checkpoint_direct.py
- convert_fsdp_to_hf.py
- convert_hf_to_fp8.py
- convert_hf_to_hf_int4.py
- convert_hf_to_int4_legacy.py
- convert_hf_to_mxfp8.py
- convert_hf_to_torch_dist.py
- convert_int4_checkpoint_direct.py
- convert_k2_thinking_int4_to_bf16.py
- convert_math_eval_to_orbit.py
- convert_nvfp4_checkpoint_direct.py
- convert_peftarena_data.py
- convert_to_hf_legacy.py
- convert_torch_dist_to_hf.py
- cpu_memory_profiler.py
- eval_checkpoints_loop.sh
- eval_checkpoints_once.sh
- fp8_cast_bf16.py
- gpu_process_memory_sampler.py
- quantize_to_int4.py
- runtime_step0_parity_utils.py
- summarize_eval_results.py
- .gitignore
- .pre-commit-config.yaml
- CUDA-13-install.md
- env.sh
- LICENSE
- pyproject.toml
- README.md
- requirements.txt
- setup.py
- train.py
- train_async.py
- uv.lock
// repository documentation
Was this content helpful?
(0 ratings)
