SGLang-FluentLLM
No description available.
File Explorer
Download Latest Version (.zip)- 1-bug-report.yml
- 2-feature-request.yml
- cancel-pr-workflow.yml
- close-inactive-issues.yml
- execute-notebook.yml
- experiment-runner.yml
- nightly-test.yml
- pr-test-amd.yml
- pr-test-rust.yml
- pr-test-sgl-kernel.yml
- pr-test.yml
- release-docker-amd.yml
- release-docker-dev.yml
- release-docker.yml
- release-docs.yml
- release-fake-tag.yml
- release-pypi-kernel.yml
- release-pypi-router.yml
- release-pypi.yml
- release-whl-kernel.yml
- CODEOWNERS
- pull_request_template.md
- deep_ep
- deep_gemm
- deep_gemm_oss
- eps
- fast-hadamard-transform
- flash-attention
- flashinfer
- flashmla
- flashmla-fp8
- flashmla-swap
- triton_configs
- logo.png
- logo_square.png
- bench_long_context.py
- bench_mix.py
- bench_mix.sh
- bench_multiturn.py
- bench_serving.py
- data_processing.py
- download.sh
- nextqa.py
- perf.py
- README.md
- search_optimal_params.py
- test_hicache.py
- bench_other.py
- bench_sglang.py
- build_dataset.py
- dataset.txt
- README.md
- bench_sglang.py
- README.md
- triton_flashinfer_cudnn.py
- benchmark_deepgemm_fp8_gemm.py
- benchmark_deepgemm_fp8_group_gemm.py.py
- README.md
- benchmark_deepseekv3_moe_align_blocks.py
- benchmark_torch_compile_fused_moe.py
- benchmark_vllm_vs_sglang_fused_moe_triton.py
- README.md
- tuning_fused_moe_triton_sep.py
- benchmark_lightning_attention_decode.py
- benchmark_lightning_attention_prefill.py
- bench_int8_quant.py
- tuning_block_wise_kernel.py
- benchmark_rmsnorm.py
- benchmark_write_req_to_token_pool_triton.py
- bench_other.py
- bench_sglang.py
- download_data.sh
- README.md
- bench_hf.py
- bench_sglang.py
- data_utils.py
- eval_utils.py
- prompt_format.yaml
- README.md
- bench_other.py
- bench_sglang.py
- README.md
- bench_other.py
- bench_sglang.py
- README.md
- bench_other.py
- bench_sglang.py
- build_dataset.py
- README.md
- bench_other.py
- bench_sglang.py
- data_gen.py
- long_prompt_multi_turn.py
- README.md
- test_pd_mmlu.py
- tool_chat_template_deepseekv31.jinja
- bench.py
- run_bench.sh
- cuda_wait_value.cuh
- hicache.cuh
- tensor.h
- utils.cuh
- utils.h
- warp.cuh
- .clang-format
- cuda_wait_value.py
- hicache.py
- utils.py
- __init__.py
- configuration_deepseek_mha_nsa.py
- configuration_flash.py
- configuration_glm4_moe.py
- configuration_qwen3.py
- configuration_qwen3_moe.py
- configuration_shortcut.py
- device_config.py
- kimi_linear.py
- load_config.py
- model_config.py
- qwen3_next.py
- utils.py
- base_grammar_backend.py
- llguidance_backend.py
- outlines_backend.py
- outlines_jump_forward.py
- xgrammar_backend.py
- __init__.py
- conn.py
- __init__.py
- conn.py
- __init__.py
- conn.py
- __init__.py
- conn.py
- transfer_engine.py
- __init__.py
- conn.py
- decode.py
- decode_schedule_batch_mixin.py
- kv_events.py
- mini_lb.py
- prefill.py
- utils.py
- test_triton_rsag.py
- triton_barrier.py
- triton_rsag.py
- triton_utils.py
- utils.py
- cuda_wrapper.py
- custom_all_reduce.py
- custom_all_reduce_utils.py
- hpu_communicator.py
- npu_communicator.py
- pynccl.py
- pynccl_wrapper.py
- shm_broadcast.py
- xpu_communicator.py
- __init__.py
- communication_op.py
- decoder_comm_manager.py
- model_tensor_tracer.py
- naive_distributed.py
- parallel_state.py
- parallel_strategy.py
- utils.py
- __init__.py
- encoding_dsv32.py
- protocol.py
- serving_base.py
- serving_chat.py
- serving_completions.py
- serving_embedding.py
- serving_rerank.py
- serving_responses.py
- serving_score.py
- tool_server.py
- usage_processor.py
- utils.py
- context.py
- engine.py
- EngineBase.py
- harmony_utils.py
- http_server.py
- http_server_engine.py
- tool.py
- base_format_detector.py
- core_types.py
- deepseekv31_detector.py
- deepseekv32_detector.py
- deepseekv3_detector.py
- ebnf_composer.py
- function_call_parser.py
- glm4_moe_detector.py
- gpt_oss_detector.py
- kimik2_detector.py
- llama32_detector.py
- longcat_detector.py
- longcat_xml_detector.py
- mistral_detector.py
- pythonic_detector.py
- qwen25_detector.py
- qwen3_coder_detector.py
- step3_detector.py
- utils.py
- dequant_k_cache.py
- index_buf_accessor.py
- nsa_indexer.py
- quant_k_cache.py
- readme.txt
- topk_utils.py
- transform_index.py
- triton_kernel.py
- utils.py
- chunk.py
- chunk_delta_h.py
- chunk_o.py
- chunk_scaled_dot_kkt.py
- cumsum.py
- fused_recurrent.py
- fused_sigmoid_gating_recurrent.py
- index.py
- kda.py
- l2norm.py
- layernorm_gated.py
- op.py
- solve_tril.py
- utils.py
- wy_fast.py
- causal_conv1d.py
- causal_conv1d_triton.py
- mamba.py
- compress_attn.py
- compress_attn_v0.py
- compress_attn_v1.py
- compress_kv.py
- nsa_mla.py
- select_attn.py
- decode_attention.py
- double_sparsity_attention.py
- extend_attention.py
- prefill_attention.py
- rocm_mla_decode_rope.py
- base_attn_backend.py
- chunker.py
- dsa_backend.py
- duo_attn_backend.py
- duo_attn_triton.py
- flash_attention_backend.py
- flashinfer_backend.py
- flashinfer_mla_backend.py
- flashmla_backend.py
- hybrid_linear_attn_backend.py
- npu_mla_backend.py
- torch_native_backend.py
- torch_native_mla_backend.py
- triton_backend.py
- utils.py
- vision.py
- __init__.py
- common.py
- cutlass.py
- deep_geem.py
- flashinfer.py
- fp8_kernel.py
- fp8_utils.py
- triton.py
- int8_kernel.py
- int8_utils.py
- base.py
- fp8.py
- unquant.py
- w8a8_fp8.py
- w8a8_int8.py
- __init__.py
- deep_ep.py
- fast_ep.py
- deep_ep_executor.py
- eps_executor.py
- eps_mixin.py
- fp8_eps_executor.py
- triton_executor.py
- wna16_executor.py
- aok.py
- triton.py
- fire.py
- triton.py
- triton.py
- triton.py
- __init__.py
- __init__.py
- triton_common.py
- triton_config.py
- common.py
- compress_tensor_load_functions.py
- fp8.py
- load_functions.py
- mapping.py
- unquant.py
- w8a8.py
- w8a8_fp8.py
- w8a8_int8.py
- wna16.py
- __init__.py
- config.py
- fused_moe_native.py
- layer.py
- topk.py
- utils.py
- __init__.py
- compressed_tensors_scheme.py
- compressed_tensors_w8a8_int8.py
- compressed_tensors_wNa16.py
- __init__.py
- compressed_tensors.py
- fused_moe.py
- gptq_marlin_moe.py
- scalar_type.py
- __init__.py
- base_config.py
- fp8.py
- utils.py
- w8a8_fp8.py
- w8a8_int8.py
- activation.py
- dp_attention.py
- flashinfer_comm_fusion.py
- layernorm.py
- linear.py
- logits_processor.py
- over_embedding.py
- parameter.py
- radix_attention.py
- rotary_embedding.py
- sampler.py
- utils.py
- vocab_parallel_embedding.py
- __init__.py
- deepseek.py
- cache_controller.py
- configure_logging.py
- data_parallel_controller.py
- detokenizer_manager.py
- eplb_manager.py
- expert_distribution.py
- expert_location.py
- expert_location_dispatch.py
- io_struct.py
- req.py
- schedule_batch.py
- schedule_policy.py
- scheduler.py
- scheduler_post_process_mixin.py
- scheduler_profiler_mixin.py
- scheduler_stats_mixin.py
- session_controller.py
- template_manager.py
- tokenizer_communicator_mixin.py
- tokenizer_manager.py
- tp_worker.py
- tp_worker_overlap_thread.py
- utils.py
- mooncake_store.py
- README.md
- test_mooncake_store.py
- __init__.py
- backend_factory.py
- allocator.py
- base_prefix_cache.py
- chunk_cache.py
- evict_policy.py
- flush_cache.py
- hicache_storage.py
- hiradix_cache.py
- memory_pool.py
- memory_pool_host.py
- radix_cache.py
- utils.py
- collector.py
- func_timer.py
- utils.py
- attn_initializer.py
- cuda_graph_runner.py
- eplb_mixin.py
- forward_batch_info.py
- model_runner.py
- prefill_cuda_graph_runner.py
- weight_mixin.py
- __init__.py
- loader.py
- utils.py
- weight_utils.py
- deepseek_mha_nsa.py
- deepseek_nextn.py
- deepseek_v2.py
- deepseek_v2_overlap.py
- deepseek_v32.py
- extensible.py
- flash_nextn.py
- gemma.py
- gemma2.py
- glm4_moe.py
- glm4_moe_nextn.py
- gpt2.py
- gpt_oss.py
- grok.py
- kimi_linear.py
- llama.py
- llama_eagle.py
- llama_eagle3.py
- llama_nextn.py
- longcat_eagle3.py
- longcat_flash.py
- longcat_flash_overlap.py
- longcat_large.py
- longcat_ultra.py
- qwen.py
- qwen2.py
- qwen2_eagle.py
- qwen2_moe.py
- qwen3.py
- qwen3_moe.py
- qwen3_next.py
- qwen3_next_mtp.py
- qwen3_nsa.py
- registry.py
- torch_native_llama.py
- utils.py
- longcat_prompt_builder.py
- code_completion_parser.py
- conversation.py
- harmony_parser.py
- jinja_template_utils.py
- reasoning_parser.py
- __init__.py
- frequency_penalty.py
- min_new_tokens.py
- orchestrator.py
- presence_penalty.py
- repetition_penalty.py
- custom_logit_processor.py
- sampling_batch_info.py
- sampling_params.py
- base_spec_worker.py
- eagle_utils.py
- eagle_worker.py
- eagle_worker_overlap.py
- pld_cuda_graph_runner.py
- pld_worker.py
- pld_worker_overlap.py
- spec_decoding_cuda_graph_runner.py
- spec_info.py
- __init__.py
- tbo_executor.py
- trace.py
- __init__.py
- common.py
- rpd_utils.py
- slow_rank_detector.py
- _custom_ops.py
- aio_rwlock.py
- custom_op.py
- env.py
- hf_transformers_utils.py
- host_shared_memory.py
- model_parallel.py
- oe_utils.py
- offloader.py
- patch_torch.py
- server.py
- server_args.py
- server_args_config_parser.py
- torch_memory_saver_adapter.py
- triton_complie_monitor.py
- utils.py
- warmup.py
- long_prompt.txt
- run_eval.py
- send_one.py
- simple_eval_common.py
- simple_eval_gpqa.py
- simple_eval_humaneval.py
- simple_eval_math.py
- simple_eval_mgsm.py
- simple_eval_mmlu.py
- test_activation.py
- test_block_fp8.py
- test_block_fp8_ep.py
- test_layernorm.py
- test_thinking_budgets.py
- __init__.py
- bench_chunker.py
- bench_one_batch.py
- bench_one_batch_server.py
- bench_serving.py
- check_env.py
- global_config.py
- launch_server.py
- llama3_eval.py
- utils.py
- version.py
- pyproject.toml
- upload_pypi.sh
- random_config.yaml
- random_flashinfer_vs_triton_config.yaml
- sharegpt_config.yaml
- hf_nsa_mla_v1_1.py
- test_compress_kv.py
- test_compute_slc_probs_prefill.py
- test_nsa_mla.py
- test_select_attn.py
- test_dsa_bf16.py
- test_lora.py
- test_lora_backend.py
- test_multi_lora_backend.py
- utils.py
- compare.py
- test_embedding_models.py
- test_generation_models.py
- test_qwen_models.py
- test_reward_models.py
- __init__.py
- double-sparsity-config-Llama-3.1-8B-Instruct.json
- experiment_runner.py
- kv_cache_scales_llama3_1_8b.json
- kv_cache_scales_llama3_8b.json
- kv_cache_scales_qwen2_1_5b.json
- run_suite.py
- test_abort.py
- test_bench_one_batch.py
- test_bench_serving.py
- test_block_int8.py
- test_cache_report.py
- test_chunked_prefill.py
- test_create_kvindices.py
- test_custom_allreduce.py
- test_data_parallelism.py
- test_double_sparsity.py
- test_dp_attention.py
- test_eagle_infer.py
- test_ebnf_constrained.py
- test_embedding_openai_server.py
- test_eval_accuracy_large.py
- test_eval_accuracy_large_chunked_prefill.py
- test_eval_accuracy_large_mixed_chunked_prefill.py
- test_eval_accuracy_mini.py
- test_eval_fp8_accuracy.py
- test_fp8_kernel.py
- test_fp8_kvcache.py
- test_function_calling.py
- test_fused_moe.py
- test_get_weights_by_name.py
- test_gguf.py
- test_gptqmodel_dynamic.py
- test_health_check.py
- test_hicache.py
- test_hidden_states.py
- test_input_embeddings.py
- test_json_constrained.py
- test_kv_events.py
- test_large_max_new_tokens.py
- test_matched_stop.py
- test_metrics.py
- test_mla.py
- test_mla_flashinfer.py
- test_mla_fp8.py
- test_mla_tp.py
- test_modelopt_fp8kvcache.py
- test_models_from_modelscope.py
- test_moe_ep.py
- test_moe_eval_accuracy_large.py
- test_nightly_gsm8k_eval.py
- test_nightly_human_eval.py
- test_nightly_math_eval.py
- test_no_chunked_prefill.py
- test_no_overlap_scheduler.py
- test_openai_server.py
- test_penalty.py
- test_pytorch_sampling_backend.py
- test_radix_attention.py
- test_rearrange.py
- test_release_memory_occupation.py
- test_request_length_validation.py
- test_retract_decode.py
- test_sagemaker_server.py
- test_schedule_policy.py
- test_server_args.py
- test_session_control.py
- test_skip_tokenizer_init.py
- test_srt_endpoint.py
- test_srt_engine.py
- test_srt_engine_with_quant_args.py
- test_torch_compile.py
- test_torch_compile_moe.py
- test_torch_native_attention_backend.py
- test_torch_tp.py
- test_torchao.py
- test_triton_attention_backend.py
- test_triton_attention_kernels.py
- test_triton_attention_rocm_mla.py
- test_update_weights_from_disk.py
- test_update_weights_from_distributed.py
- test_update_weights_from_tensor.py
- test_verl_engine.py
- test_vertex_endpoint.py
- test_vision_chunked_prefill.py
- test_vision_llm.py
- test_vision_openai_server.py
- test_w8a8_quantization.py
- __init__.py
- README.md
- test_generate_attn_args.py
- .editorconfig
- .gitignore
- .gitmodules
- .isort.cfg
- .pre-commit-config.yaml
- clean_setup.sh
- LICENSE
- Makefile
- Quick_Start.md
- README.md
# Installation Guide
1. Get the code
git clone https://github.com/meituan-longcat/SGLang-FluentLLM
Downloads the entire project code from GitHub to your computer.
cd SGLang-FluentLLM
Moves into the project folder you just downloaded.
2. Python
Easy RecommendedPrerequisites
pip install .
Installs the package published on PyPI directly โ no need to clone the source.
python <์คํํ ํ์ผ๋ช
>.py # README์์ ์ ํํ ์คํ ํ์ผ๋ช
์ ํ์ธํ์ธ์
Runs the Python script (or module).
If it runs without errors and prints output in the terminal, it worked.
3. Make
MediumPrerequisites
- Git Needed to download the project code from GitHub.
- Make Usually pre-installed on Linux/macOS. On Windows, install separately (e.g. via MSYS2 or WSL).
make
Compiles the code based on the generated build configuration to produce an executable.
If it finishes without errors, it worked. Try running the generated executable directly.
// repository documentation
Was this content helpful?
(0 ratings)
