Add files via upload

This commit is contained in:
redwyd 2025-05-15 14:10:22 +08:00
commit 1284bb346b
238 changed files with 13931 additions and 3 deletions

View file

@ -0,0 +1,3 @@
from .metric import Metric
from .metric_ASR import AttackSuccessRate
from .metric_perplexit import Perplexity

View file

@ -0,0 +1,46 @@
"""
Metric Class
========================
This module defines the Metric class, an abstract base class used for creating various
metrics that evaluate the results and data quality in the context of adversarial examples
and attacks. It provides a standardized interface for defining and implementing custom
metrics in the TextAttack framework.
"""
from abc import ABC, abstractmethod
from easyjailbreak.datasets import JailbreakDataset
__all__ = ["Metric"]
class Metric(ABC):
r"""
Abstract base class for defining metrics to evaluate adversarial example results
and data quality in the context of TextAttack.
This class serves as a blueprint for implementing various types of metrics, ensuring
consistent interfaces and functionalities across different metric implementations.
"""
@abstractmethod
def __init__(self, **kwargs):
r"""
Initializes the Metric instance.
This abstract method should be implemented in subclasses to set up any necessary
configurations for the specific metric.
:param **kwargs: Arbitrary keyword arguments specific to each metric implementation.
"""
raise NotImplementedError()
@abstractmethod
def calculate(self, dataset: JailbreakDataset):
r"""
Abstract method for computing metric values based on the provided results.
This method should be implemented in subclasses to calculate the metric based on
the attack results.
:param ~JailbreakDataset dataset: A list of instances with the results
of attacks on different instances in the dataset.
"""
raise NotImplementedError

View file

@ -0,0 +1,47 @@
"""
Metrics on AttackSuccessRate
================================
This module contains the implementation of the AttackSuccessRate metric,
which is designed to evaluate the effectiveness of jailbreak attacks in a dataset.
It calculates the number of successful and failed attacks, and computes the overall
attack success rate.
"""
from easyjailbreak.metrics.Metric.metric import Metric
from easyjailbreak.datasets import JailbreakDataset
__all__ = ['AttackSuccessRate']
class AttackSuccessRate(Metric):
"""
A metric to evaluate the success rate of jailbreak attacks.
It calculates the number of successful and failed attacks within a dataset,
and determines the overall attack success rate.
"""
def __init__(self):
self.failed_attacks = 0
self.successful_attacks = 0
self.all_metrics = {}
def calculate(self, dataset: JailbreakDataset):
"""
Calculate the attack success rate from the given dataset.
:param ~JailbreakDataset dataset: The dataset containing jailbreak attack results.
:return dict: A dictionary containing the number of successful attacks, failed attacks,
and the attack success rate.
"""
if len(dataset) == 0:
raise ValueError("The dataset is empty.")
for Instance in dataset:
if Instance.eval_results[-1] == 1:
self.successful_attacks += 1
else:
self.failed_attacks += 1
self.all_metrics["successful_attacks"] = self.successful_attacks
self.all_metrics["failed_attacks"] = self.failed_attacks
self.all_metrics["attack_success_rate"] = round(self.successful_attacks * 100.0 / len(dataset), 2)
return self.all_metrics

View file

@ -0,0 +1,88 @@
"""
Perplexity Metric:
-------------------------------------------------------
Class for calculating perplexity from Jailbreak_Dataset
"""
import torch
from easyjailbreak.metrics.Metric.metric import Metric
from easyjailbreak.datasets import JailbreakDataset
from easyjailbreak.models import WhiteBoxModelBase
class Perplexity(Metric):
def __init__(self, model:WhiteBoxModelBase, max_length=512, stride=512):
"""
Initializes the evaluator with a given language model and tokenizer.
:param model: The WhiteBoxModelBase to be used, which include model and tokenizer.
:param tokenizer: The tokenizer to be used with the language model.
:param max_length: The maximum length of tokens for the model. If None, it will be set from the model config.
:param stride: The stride to be used during tokenization. Default is 512.
# Example usage:
# from transformers import GPT2LMHeadModel, GPT2Tokenizer
# model = GPT2LMHeadModel.from_pretrained("gpt2")
# tokenizer = GPT2Tokenizer.from_pretrained("gpt2")
# evaluator = LanguageModelEvaluator(model, tokenizer)
"""
self.all_metrics = {}
self.prompts = [] # prompts to calculate ppl
# Initialize model and tokenizer
self.ppl_model = model.model
self.ppl_tokenizer = model.tokenizer
# Set the model to evaluation mode
self.ppl_model.eval()
# Set max_length from the model configuration if not provided
self.max_length = max_length
self.stride = stride
def calculate(self, dataset: JailbreakDataset):
"""Calculates average Perplexity on the final prompts generated by attacker using a
pre-trained small GPT-2 model.
Args:
dataset (``Jailbreak_Dataset`` objects):
list of instances with attack results
"""
self.dataset = dataset
for Instance in self.dataset:
self.prompts.append(Instance.jailbreak_prompt)
ppl = self.calc_ppl(self.prompts)
self.all_metrics["avg_prompt_perplexity"] = round(ppl, 2)
return self.all_metrics
def calc_ppl(self, texts):
with torch.no_grad():
text = " ".join(texts)
eval_loss = []
input_ids = torch.tensor(
self.ppl_tokenizer.encode(text, add_special_tokens=True)
).unsqueeze(0)
# Strided perplexity calculation from huggingface.co/transformers/perplexity.html
for i in range(0, input_ids.size(1), self.stride):
begin_loc = max(i + self.stride - self.max_length, 0)
end_loc = min(i + self.stride, input_ids.size(1))
trg_len = end_loc - i
input_ids_t = input_ids[:, begin_loc:end_loc].to(
self.ppl_model.device
)
target_ids = input_ids_t.clone()
target_ids[:, :-trg_len] = -100
outputs = self.ppl_model(input_ids_t, labels=target_ids)
log_likelihood = outputs[0] * trg_len
eval_loss.append(log_likelihood)
return torch.exp(torch.stack(eval_loss).sum() / end_loc).item()