Remove more bible
This commit is contained in:
parent
7812ec0505
commit
329f4e668e
2 changed files with 3 additions and 7 deletions
|
|
@ -19,8 +19,6 @@ device: torch.device = torch.device("cuda" if cuda_available else "cpu")
|
||||||
# use token from environment variable
|
# use token from environment variable
|
||||||
login(token=os.getenv("HF_TOKEN"))
|
login(token=os.getenv("HF_TOKEN"))
|
||||||
|
|
||||||
bible_words: List[str] = open("bible.txt").read().split()
|
|
||||||
|
|
||||||
alpha: float = 0.3 # weight between benign maximization and loss minimization
|
alpha: float = 0.3 # weight between benign maximization and loss minimization
|
||||||
min_benign_confidence: float = 0.95 # Minimum benign probability to stop (95%)
|
min_benign_confidence: float = 0.95 # Minimum benign probability to stop (95%)
|
||||||
words_to_inject: int = 1
|
words_to_inject: int = 1
|
||||||
|
|
|
||||||
8
utils.py
8
utils.py
|
|
@ -150,6 +150,8 @@ def find_best_word_to_add(
|
||||||
results = []
|
results = []
|
||||||
best_combined_score = 0
|
best_combined_score = 0
|
||||||
best_result_idx = -1
|
best_result_idx = -1
|
||||||
|
|
||||||
|
print(f"Running score analysis for {len(all_candidate_prefixes)} candidate prefixes")
|
||||||
|
|
||||||
for idx, candidate in enumerate(all_candidate_prefixes):
|
for idx, candidate in enumerate(all_candidate_prefixes):
|
||||||
benign_score = benign_scores[idx]
|
benign_score = benign_scores[idx]
|
||||||
|
|
@ -187,7 +189,7 @@ def find_best_word_to_add(
|
||||||
token_count, combined_score
|
token_count, combined_score
|
||||||
)
|
)
|
||||||
|
|
||||||
print(f"Word '{candidate['word']}' at {candidate['position']}: {benign_score:.4f} (Δ: {improvement:.4f}, tokens: {token_count}, combined: {combined_score:.4f})")
|
#print(f"Word '{candidate['word']}' at {candidate['position']}: {benign_score:.4f} (Δ: {improvement:.4f}, tokens: {token_count}, combined: {combined_score:.4f})")
|
||||||
|
|
||||||
# Only consider improvements (benign_score > baseline_score)
|
# Only consider improvements (benign_score > baseline_score)
|
||||||
if improvement > 0 and combined_score > best_combined_score:
|
if improvement > 0 and combined_score > best_combined_score:
|
||||||
|
|
@ -297,10 +299,8 @@ def analyze_token_contributions(
|
||||||
adv_prefix: str,
|
adv_prefix: str,
|
||||||
text: str,
|
text: str,
|
||||||
benign_class_idx: int,
|
benign_class_idx: int,
|
||||||
min_benign_confidence: float,
|
|
||||||
device: torch.device,
|
device: torch.device,
|
||||||
min_acceptable_benign: float = 0.6,
|
min_acceptable_benign: float = 0.6,
|
||||||
token_length_weight: float = 0.3, # Weight for prioritizing removal of short tokens
|
|
||||||
order_template: str = "{injection}{prefix}{text}" # Template for ordering components
|
order_template: str = "{injection}{prefix}{text}" # Template for ordering components
|
||||||
) -> str:
|
) -> str:
|
||||||
"""
|
"""
|
||||||
|
|
@ -412,7 +412,6 @@ def minimize_tokens(
|
||||||
benign_class_idx: int,
|
benign_class_idx: int,
|
||||||
min_benign_confidence: float,
|
min_benign_confidence: float,
|
||||||
device: torch.device,
|
device: torch.device,
|
||||||
target_tokens: int = 1,
|
|
||||||
min_acceptable_benign: float = 0.6,
|
min_acceptable_benign: float = 0.6,
|
||||||
token_length_weight: float = 0.3 # Weight for prioritizing removal of short tokens
|
token_length_weight: float = 0.3 # Weight for prioritizing removal of short tokens
|
||||||
) -> str:
|
) -> str:
|
||||||
|
|
@ -543,7 +542,6 @@ def get_combined_score(
|
||||||
text: str,
|
text: str,
|
||||||
candidates: List[str],
|
candidates: List[str],
|
||||||
benign_idx: int,
|
benign_idx: int,
|
||||||
malicious_idx: int,
|
|
||||||
device: torch.device,
|
device: torch.device,
|
||||||
alpha: float = 0.5,
|
alpha: float = 0.5,
|
||||||
token_penalty_weight: float = 0.1,
|
token_penalty_weight: float = 0.1,
|
||||||
|
|
|
||||||
Loading…
Add table
Add a link
Reference in a new issue