Add files via upload
This commit is contained in:
parent
0978bb2f1d
commit
1284bb346b
238 changed files with 13931 additions and 3 deletions
3
easyjailbreak/metrics/Metric/__init__.py
Normal file
3
easyjailbreak/metrics/Metric/__init__.py
Normal file
|
|
@ -0,0 +1,3 @@
|
|||
from .metric import Metric
|
||||
from .metric_ASR import AttackSuccessRate
|
||||
from .metric_perplexit import Perplexity
|
||||
46
easyjailbreak/metrics/Metric/metric.py
Normal file
46
easyjailbreak/metrics/Metric/metric.py
Normal file
|
|
@ -0,0 +1,46 @@
|
|||
"""
|
||||
Metric Class
|
||||
========================
|
||||
This module defines the Metric class, an abstract base class used for creating various
|
||||
metrics that evaluate the results and data quality in the context of adversarial examples
|
||||
and attacks. It provides a standardized interface for defining and implementing custom
|
||||
metrics in the TextAttack framework.
|
||||
"""
|
||||
|
||||
from abc import ABC, abstractmethod
|
||||
from easyjailbreak.datasets import JailbreakDataset
|
||||
__all__ = ["Metric"]
|
||||
|
||||
class Metric(ABC):
|
||||
r"""
|
||||
Abstract base class for defining metrics to evaluate adversarial example results
|
||||
and data quality in the context of TextAttack.
|
||||
|
||||
This class serves as a blueprint for implementing various types of metrics, ensuring
|
||||
consistent interfaces and functionalities across different metric implementations.
|
||||
"""
|
||||
|
||||
@abstractmethod
|
||||
def __init__(self, **kwargs):
|
||||
r"""
|
||||
Initializes the Metric instance.
|
||||
|
||||
This abstract method should be implemented in subclasses to set up any necessary
|
||||
configurations for the specific metric.
|
||||
|
||||
:param **kwargs: Arbitrary keyword arguments specific to each metric implementation.
|
||||
"""
|
||||
raise NotImplementedError()
|
||||
|
||||
@abstractmethod
|
||||
def calculate(self, dataset: JailbreakDataset):
|
||||
r"""
|
||||
Abstract method for computing metric values based on the provided results.
|
||||
|
||||
This method should be implemented in subclasses to calculate the metric based on
|
||||
the attack results.
|
||||
|
||||
:param ~JailbreakDataset dataset: A list of instances with the results
|
||||
of attacks on different instances in the dataset.
|
||||
"""
|
||||
raise NotImplementedError
|
||||
47
easyjailbreak/metrics/Metric/metric_ASR.py
Normal file
47
easyjailbreak/metrics/Metric/metric_ASR.py
Normal file
|
|
@ -0,0 +1,47 @@
|
|||
"""
|
||||
Metrics on AttackSuccessRate
|
||||
================================
|
||||
This module contains the implementation of the AttackSuccessRate metric,
|
||||
which is designed to evaluate the effectiveness of jailbreak attacks in a dataset.
|
||||
It calculates the number of successful and failed attacks, and computes the overall
|
||||
attack success rate.
|
||||
"""
|
||||
from easyjailbreak.metrics.Metric.metric import Metric
|
||||
from easyjailbreak.datasets import JailbreakDataset
|
||||
|
||||
__all__ = ['AttackSuccessRate']
|
||||
|
||||
class AttackSuccessRate(Metric):
|
||||
"""
|
||||
A metric to evaluate the success rate of jailbreak attacks.
|
||||
It calculates the number of successful and failed attacks within a dataset,
|
||||
and determines the overall attack success rate.
|
||||
"""
|
||||
def __init__(self):
|
||||
self.failed_attacks = 0
|
||||
self.successful_attacks = 0
|
||||
self.all_metrics = {}
|
||||
|
||||
def calculate(self, dataset: JailbreakDataset):
|
||||
"""
|
||||
Calculate the attack success rate from the given dataset.
|
||||
|
||||
:param ~JailbreakDataset dataset: The dataset containing jailbreak attack results.
|
||||
|
||||
:return dict: A dictionary containing the number of successful attacks, failed attacks,
|
||||
and the attack success rate.
|
||||
"""
|
||||
if len(dataset) == 0:
|
||||
raise ValueError("The dataset is empty.")
|
||||
|
||||
for Instance in dataset:
|
||||
if Instance.eval_results[-1] == 1:
|
||||
self.successful_attacks += 1
|
||||
else:
|
||||
self.failed_attacks += 1
|
||||
|
||||
self.all_metrics["successful_attacks"] = self.successful_attacks
|
||||
self.all_metrics["failed_attacks"] = self.failed_attacks
|
||||
self.all_metrics["attack_success_rate"] = round(self.successful_attacks * 100.0 / len(dataset), 2)
|
||||
|
||||
return self.all_metrics
|
||||
88
easyjailbreak/metrics/Metric/metric_perplexit.py
Normal file
88
easyjailbreak/metrics/Metric/metric_perplexit.py
Normal file
|
|
@ -0,0 +1,88 @@
|
|||
"""
|
||||
|
||||
Perplexity Metric:
|
||||
-------------------------------------------------------
|
||||
Class for calculating perplexity from Jailbreak_Dataset
|
||||
|
||||
"""
|
||||
|
||||
import torch
|
||||
from easyjailbreak.metrics.Metric.metric import Metric
|
||||
from easyjailbreak.datasets import JailbreakDataset
|
||||
from easyjailbreak.models import WhiteBoxModelBase
|
||||
|
||||
|
||||
class Perplexity(Metric):
|
||||
def __init__(self, model:WhiteBoxModelBase, max_length=512, stride=512):
|
||||
"""
|
||||
Initializes the evaluator with a given language model and tokenizer.
|
||||
:param model: The WhiteBoxModelBase to be used, which include model and tokenizer.
|
||||
:param tokenizer: The tokenizer to be used with the language model.
|
||||
:param max_length: The maximum length of tokens for the model. If None, it will be set from the model config.
|
||||
:param stride: The stride to be used during tokenization. Default is 512.
|
||||
|
||||
# Example usage:
|
||||
# from transformers import GPT2LMHeadModel, GPT2Tokenizer
|
||||
# model = GPT2LMHeadModel.from_pretrained("gpt2")
|
||||
# tokenizer = GPT2Tokenizer.from_pretrained("gpt2")
|
||||
# evaluator = LanguageModelEvaluator(model, tokenizer)
|
||||
"""
|
||||
self.all_metrics = {}
|
||||
self.prompts = [] # prompts to calculate ppl
|
||||
|
||||
# Initialize model and tokenizer
|
||||
self.ppl_model = model.model
|
||||
self.ppl_tokenizer = model.tokenizer
|
||||
|
||||
# Set the model to evaluation mode
|
||||
self.ppl_model.eval()
|
||||
|
||||
# Set max_length from the model configuration if not provided
|
||||
self.max_length = max_length
|
||||
self.stride = stride
|
||||
|
||||
|
||||
|
||||
def calculate(self, dataset: JailbreakDataset):
|
||||
"""Calculates average Perplexity on the final prompts generated by attacker using a
|
||||
pre-trained small GPT-2 model.
|
||||
|
||||
Args:
|
||||
dataset (``Jailbreak_Dataset`` objects):
|
||||
list of instances with attack results
|
||||
"""
|
||||
self.dataset = dataset
|
||||
|
||||
for Instance in self.dataset:
|
||||
self.prompts.append(Instance.jailbreak_prompt)
|
||||
|
||||
ppl = self.calc_ppl(self.prompts)
|
||||
|
||||
self.all_metrics["avg_prompt_perplexity"] = round(ppl, 2)
|
||||
|
||||
return self.all_metrics
|
||||
|
||||
def calc_ppl(self, texts):
|
||||
with torch.no_grad():
|
||||
text = " ".join(texts)
|
||||
eval_loss = []
|
||||
input_ids = torch.tensor(
|
||||
self.ppl_tokenizer.encode(text, add_special_tokens=True)
|
||||
).unsqueeze(0)
|
||||
# Strided perplexity calculation from huggingface.co/transformers/perplexity.html
|
||||
for i in range(0, input_ids.size(1), self.stride):
|
||||
begin_loc = max(i + self.stride - self.max_length, 0)
|
||||
end_loc = min(i + self.stride, input_ids.size(1))
|
||||
trg_len = end_loc - i
|
||||
input_ids_t = input_ids[:, begin_loc:end_loc].to(
|
||||
self.ppl_model.device
|
||||
)
|
||||
target_ids = input_ids_t.clone()
|
||||
target_ids[:, :-trg_len] = -100
|
||||
|
||||
outputs = self.ppl_model(input_ids_t, labels=target_ids)
|
||||
log_likelihood = outputs[0] * trg_len
|
||||
|
||||
eval_loss.append(log_likelihood)
|
||||
|
||||
return torch.exp(torch.stack(eval_loss).sum() / end_loc).item()
|
||||
Loading…
Add table
Add a link
Reference in a new issue