STRING
[ICLR'25] Data and code for our paper "Why Does the Effective Context Length of LLMs Fall Short?"
파일 탐색기
최종 버전 다운로드 (.zip)- comp.png
- difference.png
- string.png
- download_dataset.sh
- args.py
- compute_scores.py
- eval_utils.py
- prompt.py
- test_infbench_llama.py
- PaulGrahamEssays.json
- test_niah_llama.py
- download_paulgraham_essay.py
- download_qa_dataset.sh
- hotpotqa.json
- PaulGrahamEssays.json
- PaulGrahamEssays_URLs.txt
- squad.json
- common_words_extraction.py
- constants.py
- freq_words_extraction.py
- niah.py
- qa.py
- variable_tracking.py
- prepare.py
- template.py
- tokenizer.py
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- string-<S=0.33L>-<W=128>.jsonl
- auto_prepare_data.py
- synthetic.yaml
- test_ruler_llama.py
- test_ruler_qwen2.py
- utils.py
- Meta-Llama-3-70B-Instruct.yaml
- Meta-Llama-3-8B-Instruct-FP8.yaml
- Meta-Llama-3-8B-Instruct.yaml
- Mixtral-8x22B-Instruct-v0.1-FP8-Dynamic.yaml
- Mixtral-8x7B-Instruct-v0.1-FP8.yaml
- Mixtral-8x7B-Instruct-v0.1.yaml
- models-large.txt
- models-small.txt
- Qwen2-57B-A14-Instruct.yaml
- run-lm-eval-gsm-hf-baseline.sh
- run-lm-eval-gsm-vllm-baseline.sh
- run-tests.sh
- test_lm_eval_correctness.py
- convert-results-json-to-markdown.py
- wait-for-image.sh
- descriptions.md
- benchmark-pipeline.yaml
- kickoff-pipeline.sh
- README.md
- run-benchmarks-suite.sh
- check-wheel-size.py
- download-images.sh
- release-pipeline.yaml
- run-amd-test.sh
- run-benchmarks.sh
- run-cpu-test.sh
- run-neuron-test.sh
- run-openvino-test.sh
- run-xpu-test.sh
- test-pipeline.yaml
- 100-documentation.yml
- 200-installation.yml
- 300-usage.yml
- 400-bug report.yml
- 500-feature request.yml
- 600-new model.yml
- 700-performance discussion.yml
- 750-RFC.yml
- 800-misc discussion.yml
- config.yml
- build.sh
- create_release.js
- cuda-install.sh
- env.sh
- pytorch-install.sh
- clang-format.yml
- mypy.yaml
- publish.yml
- ruff.yml
- yapf.yml
- PULL_REQUEST_TEMPLATE.md
- w8a8_benchmarks.py
- weight_shapes.py
- benchmark_aqlm.py
- benchmark_marlin.py
- benchmark_moe.py
- benchmark_paged_attention.py
- benchmark_rope.py
- benchmark_shapes.py
- benchmark_hashing.py
- backend_request_func.py
- benchmark_latency.py
- benchmark_prefix_caching.py
- benchmark_serving.py
- benchmark_throughput.py
- launch_tgi_server.sh
- README.md
- sonnet.txt
- cpu_extension.cmake
- hipify.py
- utils.cmake
- attention_dtypes.h
- attention_generic.cuh
- attention_kernels.cu
- attention_utils.cuh
- dtype_bfloat16.cuh
- dtype_float16.cuh
- dtype_float32.cuh
- dtype_fp8.cuh
- activation.cpp
- attention.cpp
- cache.cpp
- cpu_types.hpp
- cpu_types_vsx.hpp
- cpu_types_x86.hpp
- layernorm.cpp
- pos_encoding.cpp
- torch_bindings.cpp
- moe_ops.h
- topk_softmax_kernels.cu
- torch_bindings.cpp
- bgmv_bf16_bf16_bf16.cu
- bgmv_bf16_fp32_bf16.cu
- bgmv_config.h
- bgmv_fp16_fp16_fp16.cu
- bgmv_fp16_fp32_fp16.cu
- bgmv_fp32_bf16_bf16.cu
- bgmv_fp32_fp16_fp16.cu
- bgmv_impl.cuh
- generator.py
- vec_dtypes.cuh
- LICENSE
- punica_ops.cu
- punica_ops.h
- torch_bindings.cpp
- type_convert.h
- gemm_kernels.cu
- dequantize.cuh
- gemm_kernels.cu
- int8_quant_kernels.cu
- broadcast_load_epilogue_c2x.hpp
- broadcast_load_epilogue_c3x.hpp
- common.hpp
- scaled_mm_c2x.cu
- scaled_mm_c3x.cu
- scaled_mm_entry.cu
- hip_float8.h
- hip_float8_impl.h
- quant_utils.cuh
- quant_utils.cuh
- common.cu
- fp8_marlin.cu
- compat.cuh
- matrix_view.cuh
- q_gemm.cu
- qdq_2.cuh
- qdq_3.cuh
- qdq_4.cuh
- qdq_8.cuh
- qdq_util.cuh
- gptq_marlin.cu
- gptq_marlin.cuh
- gptq_marlin_dtypes.cuh
- gptq_marlin_repack.cu
- LICENSE
- marlin_cuda_kernel.cu
- base.h
- mem.h
- mma.h
- LICENSE
- marlin_24_cuda_kernel.cu
- quant_cuda_kernel.cu
- activation_kernels.cu
- cache.h
- cache_kernels.cu
- cuda_compat.h
- cuda_utils.h
- cuda_utils_kernels.cu
- custom_all_reduce.cu
- custom_all_reduce.cuh
- custom_all_reduce_test.cu
- dispatch_utils.h
- layernorm_kernels.cu
- moe_align_block_size_kernels.cu
- ops.h
- pos_encoding_kernels.cu
- reduction_utils.cuh
- registration.h
- torch_bindings.cpp
- dockerfile-stages-dependency.png
- k_vecs.png
- key.png
- logits_vec.png
- q_vecs.png
- query.png
- v_vec.png
- value.png
- vllm-logo-only-light.png
- vllm-logo-text-dark.png
- vllm-logo-text-light.png
- apc.rst
- details.md
- meetups.rst
- sponsors.md
- dockerfile.rst
- async_llm_engine.rst
- engine_index.rst
- llm_engine.rst
- input_processing_pipeline.rst
- model_inputs_index.rst
- paged_attention.rst
- adding_multimodal_model.rst
- multimodal_index.rst
- llm.rst
- llm_inputs.rst
- offline_index.rst
- sampling_params.rst
- examples_index.template.rst
- amd-installation.rst
- cpu-installation.rst
- debugging.rst
- installation.rst
- neuron-installation.rst
- openvino-installation.rst
- quickstart.rst
- tpu-installation.rst
- xpu-installation.rst
- adding_model.rst
- engine_args.rst
- lora.rst
- performance.rst
- spec_decode.rst
- supported_models.rst
- vlm.rst
- auto_awq.rst
- fp8.rst
- fp8_e4m3_kvcache.rst
- fp8_e5m2_kvcache.rst
- supported_hardware.rst
- deploying_with_bentoml.rst
- deploying_with_cerebrium.rst
- deploying_with_docker.rst
- deploying_with_dstack.rst
- deploying_with_kserve.rst
- deploying_with_lws.rst
- deploying_with_triton.rst
- distributed_serving.rst
- env_vars.rst
- faq.rst
- integrations.rst
- metrics.rst
- openai_compatible_server.md
- run_on_sky.rst
- serving_with_langchain.rst
- tensorizer.rst
- usage_stats.md
- conf.py
- generate_examples.py
- index.rst
- make.bat
- Makefile
- README.md
- requirements-docs.txt
- quantize.py
- README.md
- extract_scales.py
- README.md
- docker-compose.yaml
- dummy_client.py
- Otel.md
- prometheus.yaml
- README.md
- api_client.py
- aqlm_example.py
- gradio_openai_chatbot_webserver.py
- gradio_webserver.py
- llava_example.py
- llava_next_example.py
- llm_engine_example.py
- logging_configuration.md
- lora_with_quantization_inference.py
- multilora_inference.py
- offline_inference.py
- offline_inference_arctic.py
- offline_inference_distributed.py
- offline_inference_embedding.py
- offline_inference_mlpspeculator.py
- offline_inference_neuron.py
- offline_inference_openai.md
- offline_inference_with_prefix.py
- openai_chat_completion_client.py
- openai_completion_client.py
- openai_embedding_client.py
- openai_example_batch.jsonl
- openai_vision_api_client.py
- phi3v_example.py
- save_sharded_state.py
- template_alpaca.jinja
- template_baichuan.jinja
- template_chatglm.jinja
- template_chatglm2.jinja
- template_chatml.jinja
- template_falcon.jinja
- template_falcon_180b.jinja
- template_inkbot.jinja
- template_llava.jinja
- tensorize_vllm_model.py
- rocm_bf16.patch
- __init__.py
- api_server_async_engine.py
- test_api_server.py
- test_async_llm_engine.py
- test_chat_template.py
- test_openapi_server_ray.py
- test_request_tracker.py
- __init__.py
- test_basic_correctness.py
- test_chunked_prefill.py
- test_preemption.py
- __init__.py
- conftest.py
- test_correctness.py
- test_correctness_sliding_window.py
- __init__.py
- conftest.py
- test_block_manager_v2.py
- test_block_table.py
- test_common.py
- test_cpu_gpu_block_allocator.py
- test_naive_block.py
- test_prefix_caching_block.py
- __init__.py
- test_block_manager.py
- test_chunked_prefill_scheduler.py
- test_scheduler.py
- utils.py
- __init__.py
- test_basic_distributed_correctness.py
- test_chunked_prefill_distributed.py
- test_comm_ops.py
- test_custom_all_reduce.py
- test_multimodal_broadcast.py
- test_parallel_state.py
- test_pipeline_parallel.py
- test_pynccl.py
- test_same_node.py
- test_shm_broadcast.py
- test_utils.py
- __init__.py
- test_multi_step.py
- test_stop_checker.py
- __init__.py
- test_computed_prefix_blocks.py
- test_detokenization.py
- test_multiproc_workers.py
- test_skip_tokenizer_init.py
- test_stop_reason.py
- test_stop_strings.py
- __init__.py
- test_encode.py
- test_generate.py
- test_generate_multiple_loras.py
- __init__.py
- test_chat.py
- test_completion.py
- test_embedding.py
- test_guided_processors.py
- test_models.py
- test_oot_registration.py
- test_run_batch.py
- test_serving_chat.py
- test_vision.py
- __init__.py
- __init__.py
- allclose_default.py
- conftest.py
- test_activation.py
- test_attention.py
- test_attention_selector.py
- test_blocksparse_attention.py
- test_cache.py
- test_cutlass.py
- test_flash_attn.py
- test_int8_quant.py
- test_layernorm.py
- test_marlin_gemm.py
- test_moe.py
- test_pos_encoding.py
- test_prefix_prefill.py
- test_rand.py
- test_sampler.py
- utils.py
- __init__.py
- long_context_test_data.py
- __init__.py
- conftest.py
- test_baichuan.py
- test_chatglm3.py
- test_gemma.py
- test_layer_variation.py
- test_layers.py
- test_llama.py
- test_long_context.py
- test_lora.py
- test_lora_checkpoints.py
- test_lora_manager.py
- test_mixtral.py
- test_phi.py
- test_punica.py
- test_quant_model.py
- test_tokenizer_group.py
- test_utils.py
- test_worker.py
- utils.py
- __init__.py
- test_metrics.py
- __init__.py
- weight_utils.py
- __init__.py
- test_aqlm.py
- test_big_models.py
- test_compressed_tensors.py
- test_embedding.py
- test_fp8.py
- test_gptq_marlin.py
- test_gptq_marlin_24.py
- test_jamba.py
- test_llava.py
- test_llava_next.py
- test_marlin.py
- test_mistral.py
- test_models.py
- test_oot_registration.py
- test_phi3v.py
- test_registry.py
- utils.py
- __init__.py
- test_mapper.py
- test_utils.py
- __init__.py
- test_disable_sliding_window.py
- test_prefix_caching.py
- example.txt
- summary.txt
- __init__.py
- test_bitsandbytes.py
- test_compressed_tensors.py
- test_configs.py
- test_fp8.py
- test_lm_head.py
- utils.py
- __init__.py
- test_beam_search.py
- test_ignore_eos.py
- test_logits_processor.py
- test_logprobs.py
- test_ranks.py
- test_rejection_sampler.py
- test_sampler.py
- test_seeded_generate.py
- test_typical_acceptance_sampler.py
- __init__.py
- conftest.py
- test_compatibility.py
- test_integration.py
- test_integration_dist_tp2.py
- test_integration_dist_tp4.py
- test_logprobs.py
- test_mlp_correctness.py
- test_multistep_correctness.py
- test_ngram_correctness.py
- __init__.py
- test_batch_expansion.py
- test_dynamic_spec_decode.py
- test_metrics.py
- test_multi_step_worker.py
- test_ngram_worker.py
- test_spec_decode_worker.py
- test_utils.py
- utils.py
- __init__.py
- test_tensorizer.py
- __init__.py
- test_cached_tokenizer.py
- test_detokenize.py
- test_get_eos.py
- test_tokenizer.py
- test_tokenizer_group.py
- __init__.py
- test_tracing.py
- __init__.py
- test_model_input.py
- test_model_runner.py
- test_swap.py
- __init__.py
- conftest.py
- test_cache_block_hashing.py
- test_config.py
- test_inputs.py
- test_logger.py
- test_logits_processor.py
- test_regression.py
- test_sampling_params.py
- test_sequence.py
- test_sharded_state_loader.py
- test_utils.py
- utils.py
- __init__.py
- abstract.py
- blocksparse_attn.py
- flash_attn.py
- flash_attn_string.py
- flashinfer.py
- ipex_attn.py
- openvino.py
- pallas.py
- rocm_flash_attn.py
- torch_sdpa.py
- xformers.py
- __init__.py
- blocksparse_attention_kernel.py
- interface.py
- utils.py
- __init__.py
- ipex_attn.py
- paged_attn.py
- prefix_prefill.py
- triton_flash_attention.py
- __init__.py
- layer.py
- selector.py
- __init__.py
- block_table.py
- common.py
- cpu_gpu_block_allocator.py
- interfaces.py
- naive_block.py
- prefix_caching_block.py
- utils.py
- __init__.py
- block_manager_v1.py
- block_manager_v2.py
- embedding_model_block_manager.py
- evictor_v1.py
- evictor_v2.py
- interfaces.py
- policy.py
- scheduler.py
- __init__.py
- cuda_wrapper.py
- custom_all_reduce.py
- custom_all_reduce_utils.py
- pynccl.py
- pynccl_wrapper.py
- shm_broadcast.py
- __init__.py
- communication_op.py
- parallel_state.py
- utils.py
- __init__.py
- interfaces.py
- multi_step.py
- single_step.py
- stop_checker.py
- util.py
- __init__.py
- arg_utils.py
- async_llm_engine.py
- async_timeout.py
- llm_engine.py
- metrics.py
- __init__.py
- api_server.py
- cli_args.py
- protocol.py
- run_batch.py
- serving_chat.py
- serving_completion.py
- serving_embedding.py
- serving_engine.py
- __init__.py
- api_server.py
- llm.py
- __init__.py
- cpu_executor.py
- distributed_gpu_executor.py
- executor_base.py
- gpu_executor.py
- multiproc_gpu_executor.py
- multiproc_worker_utils.py
- neuron_executor.py
- openvino_executor.py
- ray_gpu_executor.py
- ray_utils.py
- ray_xpu_executor.py
- tpu_executor.py
- xpu_executor.py
- __init__.py
- data.py
- registry.py
- __init__.py
- formatter.py
- __init__.py
- fully_sharded_layers.py
- layers.py
- lora.py
- models.py
- punica.py
- request.py
- utils.py
- worker_manager.py
- __init__.py
- lm_format_enforcer_decoding.py
- outlines_decoding.py
- outlines_logits_processors.py
- README
- __init__.py
- fused_moe.py
- layer.py
- __init__.py
- rand.py
- sample.py
- __init__.py
- compressed_tensors_scheme.py
- compressed_tensors_unquantized.py
- compressed_tensors_w4a16_24.py
- compressed_tensors_w8a8.py
- compressed_tensors_wNa16.py
- __init__.py
- compressed_tensors.py
- utils.py
- __init__.py
- format_24.py
- marlin_24_perms.py
- marlin_perms.py
- marlin_utils.py
- quant_utils.py
- __init__.py
- aqlm.py
- awq.py
- base_config.py
- bitsandbytes.py
- deepspeedfp.py
- fp8.py
- gptq.py
- gptq_marlin.py
- gptq_marlin_24.py
- marlin.py
- schema.py
- squeezellm.py
- __init__.py
- activation.py
- layernorm.py
- linear.py
- logits_processor.py
- pooler.py
- rejection_sampler.py
- rotary_embedding.py
- sampler.py
- spec_decode_base_sampler.py
- typical_acceptance_sampler.py
- vocab_parallel_embedding.py
- __init__.py
- loader.py
- neuron.py
- openvino.py
- tensorizer.py
- utils.py
- weight_utils.py
- __init__.py
- arctic.py
- baichuan.py
- bloom.py
- chatglm.py
- clip.py
- commandr.py
- dbrx.py
- decilm.py
- deepseek.py
- deepseek_v2.py
- falcon.py
- gemma.py
- gemma2.py
- gpt2.py
- gpt_bigcode.py
- gpt_j.py
- gpt_neox.py
- interfaces.py
- internlm2.py
- jais.py
- jamba.py
- llama.py
- llama_embedding.py
- llava.py
- llava_next.py
- minicpm.py
- mixtral.py
- mixtral_quant.py
- mlp_speculator.py
- mpt.py
- olmo.py
- opt.py
- orion.py
- phi.py
- phi3_small.py
- phi3v.py
- qwen.py
- qwen2.py
- qwen2_moe.py
- stablelm.py
- starcoder2.py
- utils.py
- xverse.py
- __init__.py
- custom_op.py
- pooling_metadata.py
- sampling_metadata.py
- utils.py
- __init__.py
- base.py
- image.py
- registry.py
- utils.py
- __init__.py
- cuda.py
- interface.py
- rocm.py
- __init__.py
- batch_expansion.py
- draft_model_runner.py
- interfaces.py
- metrics.py
- mlp_speculator_worker.py
- multi_step_worker.py
- ngram_worker.py
- proposer_worker_base.py
- smaller_tp_proposer_worker.py
- spec_decode_worker.py
- top1_proposer.py
- util.py
- __init__.py
- arctic.py
- chatglm.py
- dbrx.py
- falcon.py
- jais.py
- mlp_speculator.py
- mpt.py
- __init__.py
- base_tokenizer_group.py
- ray_tokenizer_group.py
- tokenizer_group.py
- __init__.py
- baichuan.py
- __init__.py
- config.py
- detokenizer.py
- image_processor.py
- tokenizer.py
- __init__.py
- usage_lib.py
- __init__.py
- cache_engine.py
- cpu_model_runner.py
- cpu_worker.py
- embedding_model_runner.py
- model_runner.py
- model_runner_base.py
- neuron_model_runner.py
- neuron_worker.py
- openvino_model_runner.py
- openvino_worker.py
- tpu_model_runner.py
- tpu_worker.py
- worker.py
- worker_base.py
- xpu_model_runner.py
- xpu_worker.py
- __init__.py
- _custom_ops.py
- _ipex_ops.py
- block.py
- config.py
- envs.py
- logger.py
- outputs.py
- pooling_params.py
- py.typed
- sampling_params.py
- sequence.py
- tracing.py
- utils.py
- version.py
- .clang-format
- .dockerignore
- .gitignore
- .readthedocs.yaml
- .yapfignore
- CMakeLists.txt
- collect_env.py
- CONTRIBUTING.md
- Dockerfile
- Dockerfile.cpu
- Dockerfile.neuron
- Dockerfile.openvino
- Dockerfile.ppc64le
- Dockerfile.rocm
- Dockerfile.tpu
- Dockerfile.xpu
- format.sh
- LICENSE
- MANIFEST.in
- pyproject.toml
- README.md
- requirements-build.txt
- requirements-common.txt
- requirements-cpu.txt
- requirements-cuda.txt
- requirements-dev.txt
- requirements-lint.txt
- requirements-mamba.txt
- requirements-neuron.txt
- requirements-openvino.txt
- requirements-rocm.txt
- requirements-test.txt
- requirements-tpu.txt
- requirements-xpu.txt
- setup.py
- .gitignore
- chat_with_pdf.py
- gradio_demo.py
- LICENSE
- README.md
- string_for_llama.py
- string_for_qwen2.py
// repository documentation
Was this content helpful?
(0 ratings)
