Files
AI-Red-Teaming-CSCD94/ai-evasion/skills assessment.ipynb
T
2026-07-26 23:12:07 -04:00

13 KiB

In [3]:
import os
import requests
import pickle
import numpy as np
from typing import List, Dict, Tuple

SEED = 1337
np.random.seed(SEED)

BASE_URL = "http://154.57.164.67:30687"

class WhiteBoxAttacker:
    def __init__(self, base_url: str):
        self.base_url = base_url
        self.model = None
        self.vectorizer = None
        self.feature_names = None
        self.classes = None

    def download_model(self):
        print("[*] Downloading model...")
        r = requests.get(f"{self.base_url}/model/download")
        r.raise_for_status()

        with open("/tmp/model.pkl", "wb") as f:
            f.write(r.content)

        with open("/tmp/model.pkl", "rb") as f:
            bundle = pickle.load(f)

        self.model = bundle["classifier"]
        self.vectorizer = bundle["vectorizer"]
        self.feature_names = bundle["feature_names"]
        self.classes = bundle["classes"]
        print(f"[+] Model loaded: {len(self.feature_names)} features")

    def calculate_word_scores(self, target_class: str) -> List[Tuple[str, float]]:
        target_idx = self.classes.index(target_class)
        other_idx = 1 - target_idx

        scores = []
        for i, feature in enumerate(self.feature_names):
            if " " in feature:
                continue
            target_prob = np.exp(self.model.feature_log_prob_[target_idx][i])
            other_prob = np.exp(self.model.feature_log_prob_[other_idx][i])
            score = target_prob / (other_prob + 1e-10)
            scores.append((feature, score))

        scores.sort(key=lambda x: x[1], reverse=True)
        return scores

    def attack_review(self, text: str, target: str, max_words: int) -> Tuple[str, int]:
        word_scores = self.calculate_word_scores(target)

        augmented = text
        for num_words in range(1, max_words + 1):
            words_to_add = [w for w, _ in word_scores[:num_words]]
            augmented = text + " " + " ".join(words_to_add)

            vec = self.vectorizer.transform([augmented])
            prediction = self.model.predict(vec)[0]

            if prediction == target:
                return augmented, num_words

        return augmented, max_words

    def solve_whitebox(self) -> Dict:
        print("\n[*] Starting white-box phase...")

        r = requests.get(f"{self.base_url}/challenge/whitebox")
        r.raise_for_status()
        challenge = r.json()

        reviews = challenge["reviews"]
        max_words = challenge["max_added_words"]

        self.download_model()

        solutions = []
        for review in reviews:
            print(f"  Attacking review {review['id']}...", end=" ")
            augmented, words_used = self.attack_review(
                review["text"], review["target_sentiment"], max_words
            )
            solutions.append({"id": review["id"], "augmented_text": augmented})
            print(f"Done ({words_used} words)")

        r = requests.post(
            f"{self.base_url}/submit/whitebox", json={"solutions": solutions}
        )
        r.raise_for_status()
        result = r.json()

        if "results" in result:
            successes = sum(1 for r in result["results"] if r.get("success", False))
            print(f"[+] White-box phase: {successes}/10 completed")

        return result

def main():
    wb_attacker = WhiteBoxAttacker(BASE_URL)
    wb_result = wb_attacker.solve_whitebox()

    if not wb_result.get("phase_complete", False):
        print("[-] Failed to complete white-box phase")
        return

    print("[+] White-box phase completed successfully!")

if __name__ == "__main__":
    main()
[*] Starting white-box phase...
[*] Downloading model...
[+] Model loaded: 89972 features
  Attacking review wb_0... Done (6 words)
  Attacking review wb_1... Done (3 words)
  Attacking review wb_2... Done (3 words)
  Attacking review wb_3... Done (4 words)
  Attacking review wb_4... Done (6 words)
  Attacking review wb_5... Done (11 words)
  Attacking review wb_6... Done (5 words)
  Attacking review wb_7... Done (4 words)
  Attacking review wb_8... Done (24 words)
  Attacking review wb_9... Done (7 words)
[+] White-box phase: 10/10 completed
[+] White-box phase completed successfully!
In [4]:
import os
import requests
import numpy as np
from typing import List, Dict, Tuple

SEED = 1337
np.random.seed(SEED)

BASE_URL = "http://154.57.164.67:30687"

class BlackBoxAttacker:
    def __init__(self, base_url: str):
        self.base_url = base_url
        self.word_impacts = {}
        self.query_count = 0

    def predict(self, text: str) -> Dict:
        self.query_count += 1
        r = requests.post(f"{self.base_url}/predict", json={"text": text})
        r.raise_for_status()
        return r.json()

    def estimate_word_impacts(
        self, base_text: str, vocabulary: List[str]
    ) -> List[Tuple[str, float]]:
        base_pred = self.predict(base_text)
        base_pos_prob = base_pred["positive_probability"]

        impacts = []
        for word in vocabulary:
            augmented = base_text + " " + word
            pred = self.predict(augmented)
            impact = pred["positive_probability"] - base_pos_prob
            impacts.append((word, impact))

        impacts.sort(key=lambda x: x[1], reverse=True)
        return impacts

    def attack_review(self, text: str, max_words: int) -> Tuple[str, int]:
        positive_vocabulary = [
            "excellent", "amazing", "wonderful", "fantastic", "brilliant",
            "outstanding", "superb", "magnificent", "perfect", "exceptional",
            "masterpiece", "genius", "beautiful", "stunning", "remarkable",
            "awesome", "incredible", "phenomenal", "spectacular", "marvelous",
            "great", "good", "love", "loved", "best", "favorite", "enjoyed",
            "recommend", "highly", "definitely", "must", "liked", "appreciate",
            "admire", "adore", "enjoy", "compelling", "engaging", "captivating",
            "mesmerizing", "powerful", "touching", "moving", "inspiring",
            "uplifting", "heartwarming", "clever", "witty", "funny", "hilarious",
            "entertaining"
        ]

        impacts = self.estimate_word_impacts(text, positive_vocabulary[:50])

        augmented = text
        for i, (word, impact) in enumerate(impacts, 1):
            if i > max_words:
                break

            augmented = augmented + " " + word
            pred = self.predict(augmented)

            if pred["label"] == "positive":
                return augmented, i

        if impacts:
            top_words = [w for w, _ in impacts[:10]]
            words_to_add = []
            while len(words_to_add) < max_words:
                words_to_add.extend(top_words)
            augmented = text + " " + " ".join(words_to_add[:max_words])

        return augmented, max_words

    def solve_blackbox(self) -> Dict:
        print("\n[*] Starting black-box phase...")

        r = requests.get(f"{self.base_url}/challenge/blackbox")
        r.raise_for_status()
        challenge = r.json()

        reviews = challenge["reviews"]
        max_words = challenge["max_added_words"]

        solutions = []
        for review in reviews:
            print(f"  Attacking review {review['id']}...", end=" ")
            augmented, words_used = self.attack_review(review["text"], max_words)
            solutions.append({"id": review["id"], "augmented_text": augmented})
            print(f"Done ({words_used} words, {self.query_count} queries total)")

        r = requests.post(
            f"{self.base_url}/submit/blackbox", json={"solutions": solutions}
        )
        r.raise_for_status()
        result = r.json()

        if "results" in result:
            successes = sum(1 for r in result["results"] if r.get("success", False))
            print(f"[+] Black-box phase: {successes}/10 completed")

        return result

def main():
    bb_attacker = BlackBoxAttacker(BASE_URL)
    bb_result = bb_attacker.solve_blackbox()

    if bb_result.get("flag"):
        print("\n" + "=" * 60)
        print("[+] SUCCESS! All challenges completed!")
        print(f"[+] Flag: {bb_result['flag']}")
        print("=" * 60)
    else:
        print("[-] Failed to complete black-box phase")

if __name__ == "__main__":
    main()
[*] Starting black-box phase...
  Attacking review bb_0... Done (22 words, 73 queries total)
  Attacking review bb_1... Done (40 words, 164 queries total)
  Attacking review bb_2... Done (20 words, 235 queries total)
  Attacking review bb_3... Done (17 words, 303 queries total)
  Attacking review bb_4... Done (4 words, 358 queries total)
  Attacking review bb_5... Done (38 words, 447 queries total)
  Attacking review bb_6... Done (40 words, 538 queries total)
  Attacking review bb_7... Done (4 words, 593 queries total)
  Attacking review bb_8... Done (13 words, 657 queries total)
  Attacking review bb_9... Done (6 words, 714 queries total)
[+] Black-box phase: 10/10 completed

============================================================
[+] SUCCESS! All challenges completed!
[+] Flag: HTB{f34tur3_0bfu5c4t10n_m45t3r3d}
============================================================
In [ ]: