Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 4 additions & 4 deletions .pre-commit-config.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -38,7 +38,7 @@ repos:
- id: detect-private-key

- repo: https://github.com/codespell-project/codespell
rev: v2.4.2
rev: v2.4.3
hooks:
- id: codespell
additional_dependencies: [tomli]
Expand Down Expand Up @@ -71,19 +71,19 @@ repos:
args: ["--print-width=140"]

- repo: https://github.com/astral-sh/ruff-pre-commit
rev: v0.15.9
rev: v0.16.10
hooks:
- id: ruff
args: ["--fix"]
- id: ruff-format
- id: ruff

- repo: https://github.com/tox-dev/pyproject-fmt
rev: v2.21.0
rev: v2.30.1
hooks:
- id: pyproject-fmt
additional_dependencies: [tox]
- repo: https://github.com/abravalheri/validate-pyproject
rev: v0.25
rev: "0.26"
hooks:
- id: validate-pyproject
4 changes: 2 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -282,9 +282,9 @@ Test the server in a separate terminal and integrate the model API into your AI
```python
# 3) Use the server (in a separate Python session)
import requests, json

response = requests.post(
"http://127.0.0.1:8000/predict",
json={"prompt": "Fix typos in the following sentence: Example input"}
"http://127.0.0.1:8000/predict", json={"prompt": "Fix typos in the following sentence: Example input"}
)
print(response.json()["output"])
```
Expand Down
922 changes: 561 additions & 361 deletions extensions/thunder/README.md

Large diffs are not rendered by default.

25 changes: 13 additions & 12 deletions pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,7 @@ classifiers = [
"Programming Language :: Python :: 3.12",
"Programming Language :: Python :: 3.13",
"Programming Language :: Python :: 3.14",
"Programming Language :: Python :: 3.15",
]
dependencies = [
# download models:
Expand Down Expand Up @@ -89,15 +90,15 @@ urls.homepage = "https://github.com/lightning-AI/litgpt"
scripts.litgpt = "litgpt.__main__:main"

[tool.setuptools]
package-data.litgpt = [
"LICENSE.md",
"README.md",
]
packages.find.include = [
"litgpt",
"litgpt.*",
]
packages.find.exclude = []
package-data.litgpt = [
"LICENSE.md",
"README.md",
]

[tool.ruff]
target-version = "py310"
Expand All @@ -114,12 +115,6 @@ lint.select = [
"UP", # see: https://docs.astral.sh/ruff/rules/#pyupgrade-up
"W", # see: https://pypi.org/project/pycodestyle
]
# extend-select = [
# "C4", # see: https://pypi.org/project/flake8-comprehensions
# "PT", # see: https://pypi.org/project/flake8-pytest-style
# "RET", # see: https://pypi.org/project/flake8-return
# "SIM", # see: https://pypi.org/project/flake8-simplify
# ]
lint.ignore = [
"E501", # Line too long
"E731", # Do not assign a lambda expression, use a def
Expand All @@ -128,14 +123,20 @@ lint.ignore = [
]
# Use Google-style docstrings.
lint.pydocstyle.convention = "google"
# extend-select = [
# "C4", # see: https://pypi.org/project/flake8-comprehensions
# "PT", # see: https://pypi.org/project/flake8-pytest-style
# "RET", # see: https://pypi.org/project/flake8-return
# "SIM", # see: https://pypi.org/project/flake8-simplify
# ]

[tool.codespell]
# skip = '*.py'
quiet-level = 3
ignore-words-list = """
tral, \
Rockerfeller
"""
# skip = "*.py"
quiet-level = 3

[tool.pytest]
ini_options.addopts = [
Expand Down
6 changes: 2 additions & 4 deletions tutorials/convert_lit_models.md
Original file line number Diff line number Diff line change
Expand Up @@ -27,9 +27,7 @@ from transformers import AutoModel


state_dict = torch.load("output_dir/model.pth")
model = AutoModel.from_pretrained(
"output_dir/", local_files_only=True, state_dict=state_dict
)
model = AutoModel.from_pretrained("output_dir/", local_files_only=True, state_dict=state_dict)
```

Alternatively, you can also load the model without copying the `config.json` file as follows:
Expand Down Expand Up @@ -107,7 +105,7 @@ litgpt convert_from_litgpt $finetuned_dir/final/ out/hf-tinyllama/converted
import torch
from transformers import AutoModel

state_dict = torch.load('out/hf-tinyllama/converted/model.pth')
state_dict = torch.load("out/hf-tinyllama/converted/model.pth")
model = AutoModel.from_pretrained("TinyLlama/TinyLlama-1.1B-intermediate-step-1431k-3T", state_dict=state_dict)
```

Expand Down
14 changes: 4 additions & 10 deletions tutorials/deploy.md
Original file line number Diff line number Diff line change
Expand Up @@ -35,8 +35,7 @@ You can now send requests to the inference server you started in step 2. For exa
import requests, json

response = requests.post(
"http://127.0.0.1:8000/predict",
json={"prompt": "Fix typos in the following sentence: Example input"}
"http://127.0.0.1:8000/predict", json={"prompt": "Fix typos in the following sentence: Example input"}
)

print(response.json()["output"])
Expand All @@ -63,9 +62,7 @@ Then, use the following updated code to query the inference server:
import requests, json

response = requests.post(
"http://127.0.0.1:8000/predict",
json={"prompt": "Fix typos in the following sentence: Example input"},
stream=True
"http://127.0.0.1:8000/predict", json={"prompt": "Fix typos in the following sentence: Example input"}, stream=True
)

# stream the response
Expand Down Expand Up @@ -121,14 +118,11 @@ from openai import OpenAI
# Configure the client to use your local LitGPT server
client = OpenAI(
base_url="http://127.0.0.1:8000/v1",
api_key="not-needed" # LitGPT doesn't require authentication by default
api_key="not-needed", # LitGPT doesn't require authentication by default
)

response = client.chat.completions.create(
model="SmolLM2-135M-Instruct",
messages=[
{"role": "user", "content": "Hello! How are you?"}
]
model="SmolLM2-135M-Instruct", messages=[{"role": "user", "content": "Hello! How are you?"}]
)

print(response.choices[0].message.content)
Expand Down
66 changes: 34 additions & 32 deletions tutorials/developer-docs/adding-models.md
Original file line number Diff line number Diff line change
Expand Up @@ -41,23 +41,25 @@ For example, suppose an entry for Llama 3 8B already exists and you want to add
Copy the Llama 3 8B entry:

```python
# https://huggingface.co/meta-llama/Meta-Llama-3-8B/blob/main/config.json
dict(
name="Llama-3-8B{}",
hf_config=dict(org="meta-llama", name="Meta-Llama-3-8B{}"),
vocab_size=128256,
padding_multiple=64,
n_layer=32,
n_head=32,
n_query_groups=8,
rotary_percentage=1.0,
parallel_residual=False,
bias=False,
norm_class_name="RMSNorm",
mlp_class_name="LLaMAMLP",
intermediate_size=14336,
rope_base=500000,
),
# https://huggingface.co/meta-llama/Meta-Llama-3-8B/blob/main/config.json
(
dict(
name="Llama-3-8B{}",
hf_config=dict(org="meta-llama", name="Meta-Llama-3-8B{}"),
vocab_size=128256,
padding_multiple=64,
n_layer=32,
n_head=32,
n_query_groups=8,
rotary_percentage=1.0,
parallel_residual=False,
bias=False,
norm_class_name="RMSNorm",
mlp_class_name="LLaMAMLP",
intermediate_size=14336,
rope_base=500000,
),
)
```

Then create the entry for the 70B model. Here, make sure you update the values according to the `config.json` file available on the HF hub:
Expand Down Expand Up @@ -130,21 +132,21 @@ If you are adding a new model class, find out its prompt style. First, check [li

```python
class Llama3(PromptStyle):
def apply(self, prompt: str, **kwargs: str) -> str:
# https://github.com/meta-llama/llama3/blob/359887376f0aaf30e433f23e25df858d8c2a9833/llama/tokenizer.py#L202-L229
return (
"<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\n"
"You are a helpful assistant.<|eot_id|>\n" # The system prompt is optional
"<|start_header_id|>user<|end_header_id|>\n\n"
f"{prompt}<|eot_id|>\n"
"<|start_header_id|>assistant<|end_header_id|>\n\n"
)

def stop_tokens(self, tokenizer: "Tokenizer") -> Tuple[List[int], ...]:
return (
[tokenizer.eos_id],
[tokenizer.token_to_id("<|eot_id|>")],
)
def apply(self, prompt: str, **kwargs: str) -> str:
# https://github.com/meta-llama/llama3/blob/359887376f0aaf30e433f23e25df858d8c2a9833/llama/tokenizer.py#L202-L229
return (
"<|begin_of_text|><|start_header_id|>system<|end_header_id|>\n\n"
"You are a helpful assistant.<|eot_id|>\n" # The system prompt is optional
"<|start_header_id|>user<|end_header_id|>\n\n"
f"{prompt}<|eot_id|>\n"
"<|start_header_id|>assistant<|end_header_id|>\n\n"
)

def stop_tokens(self, tokenizer: "Tokenizer") -> Tuple[List[int], ...]:
return (
[tokenizer.eos_id],
[tokenizer.token_to_id("<|eot_id|>")],
)
```

If your model requires a different prompt template, create a new `PromptStyle` class.
Expand Down
10 changes: 2 additions & 8 deletions tutorials/developer-docs/python-api.md
Original file line number Diff line number Diff line change
Expand Up @@ -74,12 +74,7 @@ dataset = llm.prepare_dataset(


```python
llm.instruction_finetune(
config=None,
dataset=dataset,
max_iter=10,
method="full | lora | adapter | adapter_v2"
)
llm.instruction_finetune(config=None, dataset=dataset, max_iter=10, method="full | lora | adapter | adapter_v2")
```

```python
Expand All @@ -100,8 +95,7 @@ Then in another Python session:
import requests, json

response = requests.post(
"http://127.0.0.1:8000/predict",
json={"prompt": "Fix typos in the following sentence: Example input"}
"http://127.0.0.1:8000/predict", json={"prompt": "Fix typos in the following sentence: Example input"}
)

print(response.json()["output"])
Expand Down
12 changes: 4 additions & 8 deletions tutorials/evaluation.md
Original file line number Diff line number Diff line change
Expand Up @@ -112,15 +112,11 @@ Suppose you have a test dataset with the following structure:

```python
test_data = [
{
"instruction": "Name the author of 'Pride and Prejudice'.",
"input": "",
"output": "Jane Austen."
},
{"instruction": "Name the author of 'Pride and Prejudice'.", "input": "", "output": "Jane Austen."},
{
"instruction": "Pick out the adjective from the following list.",
"input": "run, tall, quickly",
"output": "The correct adjective from the list is 'tall.'"
"output": "The correct adjective from the list is 'tall.'",
},
]
```
Expand Down Expand Up @@ -173,7 +169,7 @@ Next, we use a second LLM to calculate the response quality on a scale from 0 to


```python
del llm # delete previous `llm` to free up GPU memory
del llm # delete previous `llm` to free up GPU memory
scorer = LLM.load("meta-llama/Meta-Llama-3-8B-Instruct", access_token="...")
```

Expand Down Expand Up @@ -207,7 +203,7 @@ def generate_model_scores(data_dict, model, response_field="response", target_fi
scores = generate_model_scores(test_data, model=scorer)
print(f"\n{llm}")
print(f"Number of scores: {len(scores)} of {len(test_data)}")
print(f"Average score: {sum(scores)/len(scores):.2f}\n")
print(f"Average score: {sum(scores) / len(scores):.2f}\n")
```

This will print out the average score on all test set entries:
Expand Down
25 changes: 8 additions & 17 deletions tutorials/python-api.md
Original file line number Diff line number Diff line change
Expand Up @@ -104,6 +104,7 @@ To start with random weights, for example, if you plan a pretraining script, ini

```python
from litgpt.api import LLM

llm = LLM.load("pythia-160m", init="random", tokenizer_dir="EleutherAI/pythia-160m")
```

Expand All @@ -121,15 +122,12 @@ The `generate_strategy="sequential"` setting loads different parts of the models
```python
from litgpt.api import LLM

llm = LLM.load(
"microsoft/phi-2",
distribute=None
)
llm = LLM.load("microsoft/phi-2", distribute=None)

llm.distribute(
generate_strategy="sequential",
devices=4, # Optional setting, otherwise uses all available GPUs
fixed_kv_cache_size=256 # Optionally use a small kv-cache to further reduce memory usage
fixed_kv_cache_size=256, # Optionally use a small kv-cache to further reduce memory usage
)
```

Expand Down Expand Up @@ -161,11 +159,7 @@ from litgpt.api import LLM


if __name__ == "__main__":

llm = LLM.load(
model="meta-llama/Meta-Llama-3.1-8B-Instruct",
distribute=None
)
llm = LLM.load(model="meta-llama/Meta-Llama-3.1-8B-Instruct", distribute=None)

llm.distribute(generate_strategy="tensor_parallel", devices=4)

Expand All @@ -183,10 +177,7 @@ Use the `.benchmark()` method to compare the computational performance of differ
from litgpt.api import LLM
from pprint import pprint

llm = LLM.load(
model="microsoft/phi-2",
distribute=None
)
llm = LLM.load(model="microsoft/phi-2", distribute=None)

llm.distribute(fixed_kv_cache_size=500)

Expand Down Expand Up @@ -355,7 +346,6 @@ lit_model.llm.generate("hello world")
The continued pretraining or finetuning from a downloaded model checkpoint is similar to the example above, except that we can skip the initial steps of instantiating a model with random weights.

```python

lit_model = LitLLM(checkpoint_dir="EleutherAI/pythia-160m")
data = Alpaca2k()

Expand All @@ -380,16 +370,16 @@ lit_model.llm.generate("hello world")
Suppose you trained a model and decide to follow up with a few additional training rounds. This can be achieved as follows by loading an existing Trainer checkpoint:

```python

import os


def find_latest_checkpoint(directory):
latest_checkpoint = None
latest_time = 0

for root, _, files in os.walk(directory):
for file in files:
if file.endswith('.ckpt'):
if file.endswith(".ckpt"):
file_path = os.path.join(root, file)
file_time = os.path.getmtime(file_path)
if file_time > latest_time:
Expand All @@ -398,6 +388,7 @@ def find_latest_checkpoint(directory):

return latest_checkpoint


lit_model = LitLLM(checkpoint_dir="EleutherAI/pythia-160m", trainer_ckpt_path=find_latest_checkpoint("lightning_logs"))

data.connect(lit_model.llm.tokenizer, batch_size=batch_size, max_seq_length=512)
Expand Down
Loading