Text Generation
Transformers
Safetensors
fixed-width-addition
arithmetic
interpretability
arxiv:2405.14813
custom_code
Instructions to use melephant/1-layer-addition-v2 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use melephant/1-layer-addition-v2 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="melephant/1-layer-addition-v2", trust_remote_code=True)# Load model directly from transformers import AutoModelForCausalLM model = AutoModelForCausalLM.from_pretrained("melephant/1-layer-addition-v2", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use melephant/1-layer-addition-v2 with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "melephant/1-layer-addition-v2" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "melephant/1-layer-addition-v2", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker
docker model run hf.co/melephant/1-layer-addition-v2
- SGLang
How to use melephant/1-layer-addition-v2 with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "melephant/1-layer-addition-v2" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "melephant/1-layer-addition-v2", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "melephant/1-layer-addition-v2" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "melephant/1-layer-addition-v2", "prompt": "Once upon a time,", "max_tokens": 512, "temperature": 0.5 }' - Docker Model Runner
How to use melephant/1-layer-addition-v2 with Docker Model Runner:
docker model run hf.co/melephant/1-layer-addition-v2
| from __future__ import annotations | |
| import math | |
| import torch | |
| from torch import nn | |
| class RotaryEmbedding(nn.Module): | |
| def __init__(self, d_head: int, max_seq_len: int, theta: float = 10_000.0) -> None: | |
| super().__init__() | |
| if d_head < 2 or d_head % 2 != 0: | |
| raise ValueError("RoPE requires a positive, even attention head dimension.") | |
| if max_seq_len < 1: | |
| raise ValueError("max_seq_len must be positive.") | |
| theta = float(theta) | |
| if not math.isfinite(theta) or theta <= 0: | |
| raise ValueError("RoPE theta must be positive.") | |
| self.d_head = d_head | |
| self.max_seq_len = max_seq_len | |
| self.theta = theta | |
| self.rope_dim = d_head // 2 | |
| self.register_buffer("_cos", torch.empty(0), persistent=False) | |
| self.register_buffer("_sin", torch.empty(0), persistent=False) | |
| def forward( | |
| self, | |
| query: torch.Tensor, | |
| key: torch.Tensor, | |
| ) -> tuple[torch.Tensor, torch.Tensor]: | |
| if query.shape != key.shape: | |
| raise ValueError("RoPE query and key tensors must have the same shape.") | |
| return self.rotate(query), self.rotate(key) | |
| def rotate(self, x: torch.Tensor) -> torch.Tensor: | |
| if x.ndim != 4 or x.shape[-1] != self.d_head: | |
| raise ValueError(f"RoPE expects shape [batch, heads, sequence, {self.d_head}].") | |
| seq_len = x.shape[-2] | |
| if seq_len > self.max_seq_len: | |
| raise ValueError(f"Sequence length {seq_len} exceeds RoPE limit {self.max_seq_len}.") | |
| cos, sin = self._cos_sin(x.device) | |
| cos = cos[:, :, :seq_len].to(dtype=x.dtype) | |
| sin = sin[:, :, :seq_len].to(dtype=x.dtype) | |
| first_half = x[..., : self.rope_dim] | |
| second_half = x[..., self.rope_dim :] | |
| return torch.cat( | |
| ( | |
| cos * second_half + sin * first_half, | |
| -sin * second_half + cos * first_half, | |
| ), | |
| dim=-1, | |
| ) | |
| def _cos_sin(self, device: torch.device) -> tuple[torch.Tensor, torch.Tensor]: | |
| if self._cos.numel() == 0 or self._cos.device != device: | |
| inverse_frequencies = self.theta ** ( | |
| -torch.arange(self.rope_dim, dtype=torch.float32, device=device) / self.rope_dim | |
| ) | |
| frequencies = torch.outer( | |
| torch.arange(self.max_seq_len, dtype=torch.float32, device=device), | |
| inverse_frequencies, | |
| ) | |
| self._cos = frequencies.cos()[None, None, :, :] | |
| self._sin = frequencies.sin()[None, None, :, :] | |
| return self._cos, self._sin | |