Use black to format

This commit is contained in:
Joey Yakimowich-Payne 2025-05-12 17:47:18 -06:00
commit 06eac3eaba
4 changed files with 563 additions and 794 deletions

View file

@ -10,16 +10,25 @@ from transformers import pipeline, AutoTokenizer, AutoModelForSequenceClassifica
from huggingface_hub import login
from llm_attacks.minimal_gcg.opt_utils import get_filtered_cands
from words import words
from utils import minimize_tokens, sample_control, count_tokens, get_combined_score, get_random_words, token_gradients_combined, find_best_word_to_add, words_db
from utils import (
minimize_tokens,
sample_control,
count_tokens,
get_combined_score,
get_random_words,
token_gradients_combined,
find_best_word_to_add,
words_db,
)
# check if cuda is available
# check if cuda is available
cuda_available: bool = torch.cuda.is_available()
device: torch.device = torch.device("cuda" if cuda_available else "cpu")
# use token from environment variable
login(token=os.getenv("HF_TOKEN"))
alpha: float = 0.3 # weight between benign maximization and loss minimization
alpha: float = 0.3 # weight between benign maximization and loss minimization
min_benign_confidence: float = 0.95 # Minimum benign probability to stop (95%)
words_to_inject: int = 1
improvement_threshold: float = 0.999
@ -27,7 +36,9 @@ rolling_window_size: int = 10 # Number of past scores to consider
# Define a threshold for injecting educational text when optimization stagnates
stagnation_threshold: int = 15 # Inject after this many iterations without significant improvement
grace_period: int = 50 # Number of iterations to allow after text injection before counting against patience
grace_period: int = (
50 # Number of iterations to allow after text injection before counting against patience
)
max_iterations: int = 10000 # Maximum number of optimization iterations
patience: int = 10000 # Number of iterations with no improvement before early stopping
max_top_scores: int = 10 # Number of top scores to maintain
@ -39,9 +50,11 @@ token_penalty_weight: float = 0.4
min_acceptable_benign: float = 0.50
model_id: str = "meta-llama/Llama-Prompt-Guard-2-86M"
#model_id = "meta-llama/Llama-Prompt-Guard-2-22M"
# model_id = "meta-llama/Llama-Prompt-Guard-2-22M"
tokenizer: AutoTokenizer = AutoTokenizer.from_pretrained(model_id)
model: AutoModelForSequenceClassification = AutoModelForSequenceClassification.from_pretrained(model_id)
model: AutoModelForSequenceClassification = AutoModelForSequenceClassification.from_pretrained(
model_id
)
model = model.to(device) # Move model to MPS device
benign_class: str = "label_0"
@ -72,40 +85,54 @@ init_token_priority: float = 0.0
general_token_priority: float = 0.95
# Try to use top-performing words from the database for the initial prefix
top_words = words_db.get_top_words(limit=init_prefix_words_count, min_uses=1, token_weight=init_token_priority)
top_words = words_db.get_top_words(
limit=init_prefix_words_count, min_uses=1, token_weight=init_token_priority
)
if top_words:
print(f"Using {len(top_words)} top-performing words from database for initial prefix")
# Get words with combined token and improvement prioritization
initial_words = get_random_words(
n=init_prefix_words_count,
min_uses=1, # Words must have been tested at least once
token_priority=init_token_priority
token_priority=init_token_priority,
)
init_prefix: str = " ".join(initial_words)
print(f"Created initial prefix using database-informed words (token priority: {init_token_priority})")
print(
f"Created initial prefix using database-informed words (token priority: {init_token_priority})"
)
else:
# Fall back to random words if the database doesn't have enough data
init_prefix: str = " ".join(words[:init_prefix_words_count])
print(f"Using random words for initial prefix (no database history available)")
#init_prefix = "".join(random.choices(words, k=init_prefix_words_count))
# init_prefix = "".join(random.choices(words, k=init_prefix_words_count))
def main():
global injection_text, text, init_prefix_words_count
# Parse command line arguments
parser = argparse.ArgumentParser(description="Prompt hacking tool")
parser.add_argument("--injection", type=str,
default=injection_text,
help="Injection text to use in the template")
parser.add_argument("--mandatory-text", type=str,
default=text,
help="Mandatory text to use in the template")
parser.add_argument("--init-prefix-words-count", type=int,
default=init_prefix_words_count,
help="Number of words to use in the initial prefix")
parser.add_argument(
"--injection",
type=str,
default=injection_text,
help="Injection text to use in the template",
)
parser.add_argument(
"--mandatory-text",
type=str,
default=text,
help="Mandatory text to use in the template",
)
parser.add_argument(
"--init-prefix-words-count",
type=int,
default=init_prefix_words_count,
help="Number of words to use in the initial prefix",
)
args = parser.parse_args()
# Update the global parameters based on command line arguments
injection_text = args.injection
text = args.mandatory_text
@ -113,33 +140,35 @@ def main():
print(f"Injection text: {injection_text}")
print(f"Mandatory text: {text}")
print(f"\nTrying initial prefix: {init_prefix}")
# Convert initial adversarial string to tokens
best_score: float = float('-inf')
best_score: float = float("-inf")
best_prefix: Optional[str] = None
adv_prefix: str = init_prefix
adv_prefix_tokens: torch.Tensor = tokenizer(adv_prefix, return_tensors="pt", add_special_tokens=False)["input_ids"][0]
adv_prefix_tokens: torch.Tensor = tokenizer(
adv_prefix, return_tensors="pt", add_special_tokens=False
)["input_ids"][0]
adv_prefix_tokens = adv_prefix_tokens.to(device) # Move tokens to MPS device
control_slice: slice = slice(0, len(adv_prefix_tokens)) # Slice representing the prefix tokens
best_iteration_score: float = float('-inf')
best_iteration_score: float = float("-inf")
iterations_without_improvement: int = 0
# Track both rolling and top scores
rolling_scores: List[float] = [] # List to store recent scores
top_scores: List[float] = [] # List to store top scores
# Track token counts
current_token_count: int = count_tokens(adv_prefix)
min_token_count: int = current_token_count
for i in range(max_iterations):
# Prepare input tensors using template
full_text = injection_text + adv_prefix + text
inputs: Dict[str, torch.Tensor] = tokenizer(full_text, return_tensors="pt")
input_ids: torch.Tensor = inputs['input_ids'][0].to(device) # Move input_ids to MPS device
input_ids: torch.Tensor = inputs["input_ids"][0].to(device) # Move input_ids to MPS device
# Compute gradients using combined approach
coordinate_grad: torch.Tensor = token_gradients_combined(
@ -149,7 +178,7 @@ def main():
benign_class=benign_class_idx,
malicious_class=malicious_class_idx,
alpha=alpha,
device=device
device=device,
)
# Ensure coordinate_grad is on the correct device and has the right shape
@ -164,9 +193,9 @@ def main():
new_adv_prefix_toks: torch.Tensor = sample_control(
adv_prefix_tokens,
coordinate_grad,
batch_size=32, # Larger batch for more candidates
topk=16, # More options per token
temp=1.5, # Higher temperature for more exploration
batch_size=32, # Larger batch for more candidates
topk=16, # More options per token
temp=1.5, # Higher temperature for more exploration
)
# Convert new tokens to text
@ -174,7 +203,7 @@ def main():
tokenizer,
new_adv_prefix_toks,
filter_cand=False,
curr_control=adv_prefix
curr_control=adv_prefix,
)
# Batch evaluation for all candidates with combined scoring
@ -185,25 +214,36 @@ def main():
count_range = max(1, max_count - min_count)
inputs = tokenizer(candidate_texts, return_tensors="pt", padding=True, truncation=True)
inputs = {k: v.to(device) for k, v in inputs.items()}
with torch.no_grad():
logits = model(**inputs).logits
probs = torch.softmax(logits, dim=-1)
benign_scores = probs[:, benign_class_idx].cpu().numpy()
# Compute normalized loss for each candidate
losses = nn.CrossEntropyLoss(reduction="none")(logits, torch.zeros(logits.shape[0], device=device, dtype=torch.long))
normalized_losses = (1.0 / (1.0 + losses.cpu().numpy()))
losses = nn.CrossEntropyLoss(reduction="none")(
logits,
torch.zeros(logits.shape[0], device=device, dtype=torch.long),
)
normalized_losses = 1.0 / (1.0 + losses.cpu().numpy())
# Compute token penalty for each candidate
token_penalties = [1.0 - ((tc - min_count) / count_range) if count_range > 0 else 0 for tc in token_counts]
token_penalties = [
1.0 - ((tc - min_count) / count_range) if count_range > 0 else 0
for tc in token_counts
]
# Compute combined score for each candidate
combined_scores = [
(alpha * benign_scores[i] + (1 - alpha) * normalized_losses[i]) * (1 - token_penalty_weight + token_penalty_weight * token_penalties[i])
(alpha * benign_scores[i] + (1 - alpha) * normalized_losses[i])
* (1 - token_penalty_weight + token_penalty_weight * token_penalties[i])
for i in range(len(new_adv_prefix))
]
idx = int(max(range(len(combined_scores)), key=lambda i: combined_scores[i]))
adv_prefix = new_adv_prefix[idx]
# Update the tokens for the next iteration
adv_prefix_tokens = tokenizer(adv_prefix, return_tensors="pt", add_special_tokens=False)["input_ids"][0]
adv_prefix_tokens = tokenizer(
adv_prefix, return_tensors="pt", add_special_tokens=False
)["input_ids"][0]
adv_prefix_tokens = adv_prefix_tokens.to(device)
# Check the current classification
@ -216,10 +256,14 @@ def main():
predicted_class_id: int = logits.argmax().item()
benign_score: float = probs[0][benign_class_idx].item()
benign_percentage: float = benign_score * 100
malicious_score: float = probs[0][malicious_class_idx].item() if malicious_class_idx is not None else 0
malicious_score: float = (
probs[0][malicious_class_idx].item() if malicious_class_idx is not None else 0
)
# Calculate combined score
loss: torch.Tensor = nn.CrossEntropyLoss()(logits, torch.zeros(logits.shape[0], device=device).long())
loss: torch.Tensor = nn.CrossEntropyLoss()(
logits, torch.zeros(logits.shape[0], device=device).long()
)
normalized_loss: float = 1.0 / (1.0 + loss.item())
current_score: float = alpha * benign_score + (1 - alpha) * normalized_loss
@ -244,9 +288,11 @@ def main():
if current_token_count < min_token_count:
min_token_count = current_token_count
print(f"Iteration {i+1}: Class={model.config.id2label[predicted_class_id]} " +
f"(benign: {benign_percentage:.2f}%, loss_norm: {normalized_loss:.4f}, " +
f"tokens: {current_token_count}, prefix: {adv_prefix})")
print(
f"Iteration {i+1}: Class={model.config.id2label[predicted_class_id]} "
+ f"(benign: {benign_percentage:.2f}%, loss_norm: {normalized_loss:.4f}, "
+ f"tokens: {current_token_count}, prefix: {adv_prefix})"
)
if current_score > best_iteration_score:
# New best score, reset counter
@ -254,46 +300,75 @@ def main():
iterations_without_improvement = 0
elif current_score >= combined_avg * improvement_threshold:
# Score is close enough to combined average, don't count against patience
print(f" Score within {(1-improvement_threshold)*100:.1f}% of combined average, continuing optimization")
print(
f" Score within {(1-improvement_threshold)*100:.1f}% of combined average, continuing optimization"
)
# Don't increment iterations_without_improvement
else:
# Score is significantly worse than combined average, count against patience
iterations_without_improvement += 1
print(f" No significant improvement for {iterations_without_improvement}/{patience} iterations")
print(
f" No significant improvement for {iterations_without_improvement}/{patience} iterations"
)
# If we're stagnating but not yet at early stopping threshold, try injecting educational text
if iterations_without_improvement % stagnation_threshold == 0 and iterations_without_improvement < patience:
print(f"\n Optimization stagnating. Looking for words to improve benign rating...")
if (
iterations_without_improvement % stagnation_threshold == 0
and iterations_without_improvement < patience
):
print(
f"\n Optimization stagnating. Looking for words to improve benign rating..."
)
# Try to find the best word to add
new_prefix: Optional[str]
improvement: float
new_prefix, improvement = find_best_word_to_add(
model, tokenizer, injection_text, adv_prefix, text,
benign_class_idx, device=device, num_candidates=len(words),
model,
tokenizer,
injection_text,
adv_prefix,
text,
benign_class_idx,
device=device,
num_candidates=len(words),
token_priority=general_token_priority, # Equal weight to token count and improvement
)
if new_prefix and improvement > 0:
# Use the optimized prefix with the best word added
adv_prefix = new_prefix
print(f" Applied optimized prefix with improvement of {improvement:.4f}")
print(
f" Applied optimized prefix with improvement of {improvement:.4f}"
)
else:
# Fall back to adding random words if no improvement found
snippet: str = " ".join(get_random_words(words_to_inject, 1, token_priority=general_token_priority))
snippet: str = " ".join(
get_random_words(
words_to_inject,
1,
token_priority=general_token_priority,
)
)
# Insert the snippet at the beginning
adv_prefix = snippet + " " + adv_prefix
print(f" No improvement found, inserted random words at beginning: '{snippet}'")
print(
f" No improvement found, inserted random words at beginning: '{snippet}'"
)
# Update tokens for next iteration
adv_prefix_tokens = tokenizer(adv_prefix, return_tensors="pt", add_special_tokens=False)["input_ids"][0]
adv_prefix_tokens = tokenizer(
adv_prefix, return_tensors="pt", add_special_tokens=False
)["input_ids"][0]
adv_prefix_tokens = adv_prefix_tokens.to(device)
control_slice = slice(0, len(adv_prefix_tokens))
# Give the model time to improve with the new text by resetting best score tracking
best_iteration_score = float('-inf')
iterations_without_improvement = max(0, iterations_without_improvement - grace_period)
best_iteration_score = float("-inf")
iterations_without_improvement = max(
0, iterations_without_improvement - grace_period
)
print(f" Reset optimization tracking to give new text time to work")
# Early stopping check - only stop if consistently no improvement
@ -302,46 +377,69 @@ def main():
break
# Always track the best prefix we've seen, even if not high confidence
if model.config.id2label[predicted_class_id].lower() == benign_class and current_score > best_score:
if (
model.config.id2label[predicted_class_id].lower() == benign_class
and current_score > best_score
):
best_score = current_score
best_prefix = adv_prefix
print(f"New best benign prefix found with score: {best_score:.4f}, benign confidence: {benign_percentage:.2f}%")
print(
f"New best benign prefix found with score: {best_score:.4f}, benign confidence: {benign_percentage:.2f}%"
)
# Check if we've found a high confidence benign classification (>95%)
if model.config.id2label[predicted_class_id].lower() == benign_class and benign_score > min_benign_confidence:
if (
model.config.id2label[predicted_class_id].lower() == benign_class
and benign_score > min_benign_confidence
):
found_high_confidence_benign = True # Set the flag
best_score = current_score
best_prefix = adv_prefix
print(f"Found high confidence benign classification ({benign_percentage:.2f}%) at iteration {i+1}! Stopping optimization.")
print(
f"Found high confidence benign classification ({benign_percentage:.2f}%) at iteration {i+1}! Stopping optimization."
)
break # Stop optimizing
except Exception as e:
# print stack trace
import traceback
traceback.print_exc()
print(f"Error in iteration {i+1}: {str(e)}")
continue
# Print final results
if found_high_confidence_benign:
print(f"\n===== SUCCESS! Found a high confidence benign classification (>{min_benign_confidence*100}%) =====")
print(
f"\n===== SUCCESS! Found a high confidence benign classification (>{min_benign_confidence*100}%) ====="
)
# Now minimize the tokens while maintaining the benign classification
if best_prefix is not None:
best_prefix = minimize_tokens(
model, tokenizer, injection_text, best_prefix, text,
benign_class_idx, device=device,
model,
tokenizer,
injection_text,
best_prefix,
text,
benign_class_idx,
device=device,
min_acceptable_benign=min_acceptable_benign,
)
else:
print("\n===== Did not find a high confidence benign classification =====")
if best_prefix is not None:
print(f"Best prefix found had a benign score of: {best_score:.4f}")
# Still try to minimize tokens
best_prefix = minimize_tokens(
model, tokenizer, injection_text, best_prefix, text,
benign_class_idx, device=device,
model,
tokenizer,
injection_text,
best_prefix,
text,
benign_class_idx,
device=device,
min_acceptable_benign=min_acceptable_benign,
)
@ -355,8 +453,10 @@ def main():
probs: torch.Tensor = torch.softmax(logits, dim=-1)
predicted_class_id: int = logits.argmax().item()
benign_prob: float = probs[0][benign_class_idx].item() * 100
print(f'\nPayload is: {text}')
print(f'Payload without prefix is classified as: {model.config.id2label[predicted_class_id]} (benign probability: {benign_prob:.2f}%)')
print(f"\nPayload is: {text}")
print(
f"Payload without prefix is classified as: {model.config.id2label[predicted_class_id]} (benign probability: {benign_prob:.2f}%)"
)
# With adversarial prefix
full_text = injection_text + adv_prefix + text
@ -366,15 +466,17 @@ def main():
probs = torch.softmax(logits, dim=-1)
predicted_class_id = logits.argmax().item()
benign_prob = probs[0][benign_class_idx].item() * 100
print(f'Prefix is: {adv_prefix}')
print(f'Complete text using template is: {full_text}')
print(f'Payload with prefix is classified as: {model.config.id2label[predicted_class_id]} (benign probability: {benign_prob:.2f}%)')
print(f"Prefix is: {adv_prefix}")
print(f"Complete text using template is: {full_text}")
print(
f"Payload with prefix is classified as: {model.config.id2label[predicted_class_id]} (benign probability: {benign_prob:.2f}%)"
)
# Try to run inference with the model's classifier
classifier = pipeline(
"text-classification",
model=model_id,
device=0 if torch.cuda.is_available() else -1
device=0 if torch.cuda.is_available() else -1,
)
try:
@ -388,8 +490,9 @@ def main():
except Exception as e:
print(f"Error running classifier pipeline: {str(e)}")
print(f'Adv prefix token count: {count_tokens(adv_prefix)}')
print(f'Total token count: {count_tokens(full_text)}')
print(f"Adv prefix token count: {count_tokens(adv_prefix)}")
print(f"Total token count: {count_tokens(full_text)}")
if __name__ == "__main__":
main()