diff --git a/hacking.py b/hacking.py index 8e5ab68..6eec8e8 100644 --- a/hacking.py +++ b/hacking.py @@ -330,8 +330,8 @@ def main(): if best_prefix is not None: best_prefix = minimize_tokens( model, tokenizer, injection_text, best_prefix, text, - benign_class_idx, min_benign_confidence, device=device, - target_tokens=1, min_acceptable_benign=min_acceptable_benign, + benign_class_idx, device=device, + min_acceptable_benign=min_acceptable_benign, ) else: print("\n===== Did not find a high confidence benign classification =====") @@ -341,8 +341,8 @@ def main(): # Still try to minimize tokens best_prefix = minimize_tokens( model, tokenizer, injection_text, best_prefix, text, - benign_class_idx, best_score * 0.95, target_tokens=1, - min_acceptable_benign=min_acceptable_benign, device=device, + benign_class_idx, device=device, + min_acceptable_benign=min_acceptable_benign, ) # Use the best prefix found across all runs diff --git a/utils.py b/utils.py index de0c198..624fc1c 100644 --- a/utils.py +++ b/utils.py @@ -410,10 +410,8 @@ def minimize_tokens( adv_prefix: str, text: str, benign_class_idx: int, - min_benign_confidence: float, device: torch.device, min_acceptable_benign: float = 0.6, - token_length_weight: float = 0.3 # Weight for prioritizing removal of short tokens ) -> str: """ Minimize tokens using only token contribution analysis (ablation study). @@ -429,9 +427,8 @@ def minimize_tokens( # Use only token ablation approach - systematically remove tokens that contribute least ablation_prefix: str = analyze_token_contributions( model, tokenizer, injection_text, adv_prefix, text, - benign_class_idx, min_benign_confidence=min_benign_confidence, + benign_class_idx, device=device, min_acceptable_benign=min_acceptable_benign, - token_length_weight=token_length_weight # Pass through the token length weight ) # Report final token count diff --git a/words.py b/words.py index 68a23cb..808273d 100644 --- a/words.py +++ b/words.py @@ -465,6 +465,31 @@ words3 = list(set(words3)) words4 = [ + "ocular", + "spell", + "License", + "paper", + "updated", + "others", + "boards", + "better", + "separate", + "категор", + "许可证", + "Grow", + "All", + "Simon", + "contest", + "all", + "Categories", + "bean", + "elastic", + "variable", + "generator", + "Arizona", + "Itemid", + "recommend", + "better", "previous", "Camp", "Megan",