diff --git a/patch.py b/patch.py new file mode 100644 index 0000000..e8109ca --- /dev/null +++ b/patch.py @@ -0,0 +1,19 @@ +path = '/usr/local/lib/python3.12/dist-packages/transformers/tokenization_utils_base.py' +with open(path, 'r') as f: + content = f.read() + +old = (' self.SPECIAL_TOKENS_ATTRIBUTES = self.SPECIAL_TOKENS_ATTRIBUTES' + ' + list(special_tokens.keys())') +new = (' # PATCH: some models have extra_special_tokens as list instead of dict\n' + ' if isinstance(special_tokens, list):\n' + ' special_tokens = {t: t for t in special_tokens}\n' + ' self.SPECIAL_TOKENS_ATTRIBUTES = self.SPECIAL_TOKENS_ATTRIBUTES' + ' + list(special_tokens.keys())') + +if old in content: + content = content.replace(old, new) + with open(path, 'w') as f: + f.write(content) + print('Patch applied successfully') +else: + print('WARNING: pattern not found') \ No newline at end of file diff --git a/patch_triton.py b/patch_triton.py new file mode 100644 index 0000000..8fb3579 --- /dev/null +++ b/patch_triton.py @@ -0,0 +1,24 @@ +path = '/usr/local/lib/python3.12/dist-packages/vllm/v1/attention/backends/triton_attn.py' +with open(path, 'r') as f: + content = f.read() + +old = ''' def validate_head_size(cls, head_size: int) -> None: + # Triton Attention supports any head size above 32 + if head_size < 32: + raise ValueError( + f"Head size {head_size} is not supported by TritonAttention." + f"Head sizes need to be larger or equal 32 for this backend. " + "Set VLLM_ATTENTION_BACKEND=FLEX_ATTENTION to use " + "FlexAttention backend which supports all head sizes.")''' + +new = ''' def validate_head_size(cls, head_size: int) -> None: + # PATCH: allow all head sizes (Triton compiles at runtime) + return''' + +if old in content: + content = content.replace(old, new) + with open(path, 'w') as f: + f.write(content) + print('patch_triton: validate_head_size bypassed successfully') +else: + print('patch_triton: WARNING - pattern not found, patch skipped')