From 7261ca27102f7291a75a7e42b5bc4dc9ab0705c8 Mon Sep 17 00:00:00 2001 From: shashank Date: Mon, 28 Sep 2026 01:44:46 +0530 Subject: [PATCH 1/2] FEAT Let techniques disable judge feedback and use it for GOAT GOAT's attacker never sees judge output (paper section 3.3), but the goat technique inherited RedTeamingAttack's default of appending the scorer rationale to every attacker turn, and a technique had no way to override the scenario's scoring config. - AttackTechniqueFactory gains use_score_as_feedback, applied to a copy of the scenario's scoring config in create(). - The goat technique turns feedback off; judging and early stopping are kept. - AttackIdentifier records use_score_as_feedback when feedback is off, so scenario resume cannot reuse results across feedback modes. Enabled feedback is omitted, keeping existing hashes stable. Adds the column and migration. Closes #2889 --- doc/code/scenarios/0_attack_techniques.ipynb | 3 + doc/code/scenarios/0_attack_techniques.py | 3 + .../datasets/executors/red_teaming/goat.yaml | 16 ++- .../red_teaming/goat_follow_up_prompt.yaml | 4 +- pyrit/executor/attack/core/attack_strategy.py | 7 ++ ...e9_add_use_score_as_feedback_to_attack_.py | 35 ++++++ pyrit/memory/memory_models.py | 1 + pyrit/models/identifiers/attack_identifier.py | 2 + .../scenario/core/attack_technique_factory.py | 64 ++++++++++- pyrit/setup/initializers/techniques/extra.py | 2 + .../attack/core/test_attack_strategy.py | 52 ++++++++- .../test_interface_identifiers.py | 19 +++- tests/unit/memory/test_migration.py | 21 ++++ .../core/test_attack_technique_factory.py | 104 ++++++++++++++++++ .../unit/setup/test_technique_initializer.py | 30 +++++ 15 files changed, 344 insertions(+), 19 deletions(-) create mode 100644 pyrit/memory/alembic/versions/34a18645c7e9_add_use_score_as_feedback_to_attack_.py diff --git a/doc/code/scenarios/0_attack_techniques.ipynb b/doc/code/scenarios/0_attack_techniques.ipynb index 92e0512f93..748d5608ba 100644 --- a/doc/code/scenarios/0_attack_techniques.ipynb +++ b/doc/code/scenarios/0_attack_techniques.ipynb @@ -34,6 +34,9 @@ "- a **`AttackTechniqueSeedGroup`** (`seed_technique`) of general-technique seeds, which can carry a\n", " **system prompt**, a **prepended_conversation**, a **simulated_conversation**\n", " (`SeedSimulatedConversation`), and a **next_message**;\n", + "- a **score-feedback override** (`use_score_as_feedback`): the scenario still supplies the scorer,\n", + " but the technique can decide whether its attacker sees the scorer's rationale each turn (the\n", + " `goat` technique turns this off to match its paper);\n", "- the selection metadata that lets a scenario pick it: its `name` and `technique_tags`.\n", "\n", "The objective is *not* part of the technique — it stays separate and is supplied by the dataset at\n", diff --git a/doc/code/scenarios/0_attack_techniques.py b/doc/code/scenarios/0_attack_techniques.py index 7e96150f83..52434970df 100644 --- a/doc/code/scenarios/0_attack_techniques.py +++ b/doc/code/scenarios/0_attack_techniques.py @@ -38,6 +38,9 @@ # - a **`AttackTechniqueSeedGroup`** (`seed_technique`) of general-technique seeds, which can carry a # **system prompt**, a **prepended_conversation**, a **simulated_conversation** # (`SeedSimulatedConversation`), and a **next_message**; +# - a **score-feedback override** (`use_score_as_feedback`): the scenario still supplies the scorer, +# but the technique can decide whether its attacker sees the scorer's rationale each turn (the +# `goat` technique turns this off to match its paper); # - the selection metadata that lets a scenario pick it: its `name` and `technique_tags`. # # The objective is *not* part of the technique — it stays separate and is supplied by the dataset at diff --git a/pyrit/datasets/executors/red_teaming/goat.yaml b/pyrit/datasets/executors/red_teaming/goat.yaml index 64c723de44..5eee1fb50e 100644 --- a/pyrit/datasets/executors/red_teaming/goat.yaml +++ b/pyrit/datasets/executors/red_teaming/goat.yaml @@ -15,21 +15,19 @@ description: | objective is met. This technique keeps RedTeamingAttack's per-turn judging and early stopping rather than reproducing GOAT's fixed-turn, judge-free loop, since that is existing, already-tested PyRIT behavior and changing it would mean a new executor. - 2. Judge feedback. RedTeamingAttack's AttackScoringConfig.use_score_as_feedback defaults to - True, so the objective judge's rationale is appended to the feedback the attacker sees each - turn. GOAT's own Chain-of-Attack-Thought loop (paper section 3.3) never sees judge output -- - only the raw target response and its own prior reasoning. This technique keeps the - RedTeamingAttack default (judge feedback included) rather than adding a new per-technique - scoring-config override to the factory, which is a separate, more invasive change than a - prompt/schema swap; disabling it is a reasonable follow-up if a closer match is wanted. - 3. Per-turn follow-up prompt. GOAT's follow-up prompt (paper Figure A.3) re-supplies the + 2. Per-turn follow-up prompt. GOAT's follow-up prompt (paper Figure A.3) re-supplies the attacker's own previous prompt (P) as an explicit field, alongside the goal and the target's latest response. RedTeamingAttack's per-turn adversarial template only receives feedback_text - (the target's latest response, optionally with judge feedback) and objective -- the + (built from the target's latest response, without judge feedback) and objective -- the attacker's own previous prompt is not currently exposed to per-turn templates. This technique's follow-up prompt (goat_follow_up_prompt.yaml) omits that field rather than approximate it, since the attacker's full prior reasoning already stays in the adversarial chat's own conversation history regardless. + + Judge feedback matches the paper: GOAT's Chain-of-Attack-Thought loop (paper section 3.3) never + sees judge output -- only the raw target response and its own prior reasoning. The technique + sets use_score_as_feedback=False, so the objective judge still scores every turn (keeping early + stopping) but its rationale is not appended to what the attacker sees. groups: - AI Red Team source: AI Red Team diff --git a/pyrit/datasets/executors/red_teaming/goat_follow_up_prompt.yaml b/pyrit/datasets/executors/red_teaming/goat_follow_up_prompt.yaml index f5cf391120..2a2d14d604 100644 --- a/pyrit/datasets/executors/red_teaming/goat_follow_up_prompt.yaml +++ b/pyrit/datasets/executors/red_teaming/goat_follow_up_prompt.yaml @@ -5,8 +5,8 @@ description: | generate follow-up adversarial reply given a target LLM response and prior conversation prompt"), with one deliberate omission: Figure A.3 also re-supplies the attacker's own previous prompt (P) as an explicit field. RedTeamingAttack's per-turn adversarial template only receives - feedback_text (built from the Defender's latest response, optionally with judge feedback) and - objective -- the attacker's own previous prompt is not currently exposed to per-turn templates, + feedback_text (built from the Defender's latest response; the technique disables judge feedback) + and objective -- the attacker's own previous prompt is not currently exposed to per-turn templates, so that field is left out here rather than approximated. It stays available to the attacker regardless, since the adversarial chat's own conversation history already includes everything it said before. diff --git a/pyrit/executor/attack/core/attack_strategy.py b/pyrit/executor/attack/core/attack_strategy.py index 45db4fd60a..e14b24a3cc 100644 --- a/pyrit/executor/attack/core/attack_strategy.py +++ b/pyrit/executor/attack/core/attack_strategy.py @@ -689,6 +689,12 @@ def _create_identifier( scoring_config.objective_scorer.get_identifier() ) + # Disabled feedback changes what the adversarial chat sees, so it must change the eval hash. + # Enabled feedback is the default and is omitted to keep existing hashes stable. + use_score_as_feedback: bool | None = None + if scoring_config is not None and not scoring_config.use_score_as_feedback: + use_score_as_feedback = False + # Add adversarial chat target and its effective prompts if present. The adversarial # target becomes a child (filtered to model params by the eval rule), while the # effective system/seed prompts land on the attack-strategy node so they are included @@ -738,6 +744,7 @@ def _create_identifier( adversarial_system_prompt=adversarial_system_prompt, adversarial_seed_prompt=adversarial_seed_prompt, adversarial_prompt_template=adversarial_prompt_template, + use_score_as_feedback=use_score_as_feedback, ) @staticmethod diff --git a/pyrit/memory/alembic/versions/34a18645c7e9_add_use_score_as_feedback_to_attack_.py b/pyrit/memory/alembic/versions/34a18645c7e9_add_use_score_as_feedback_to_attack_.py new file mode 100644 index 0000000000..72f720c315 --- /dev/null +++ b/pyrit/memory/alembic/versions/34a18645c7e9_add_use_score_as_feedback_to_attack_.py @@ -0,0 +1,35 @@ +# Copyright (c) Microsoft Corporation. +# Licensed under the MIT license. + +""" +add use_score_as_feedback to attack identifiers. + +Revision ID: 34a18645c7e9 +Revises: aca1eba410d9 +Create Date: 2026-09-28 00:27:55.099128 +""" + +from collections.abc import Sequence + +import sqlalchemy as sa +from alembic import op + +# revision identifiers, used by Alembic. +revision: str = "34a18645c7e9" +down_revision: str | None = "aca1eba410d9" +branch_labels: str | Sequence[str] | None = None +depends_on: str | Sequence[str] | None = None + + +def upgrade() -> None: + """Apply this schema upgrade.""" + # ### commands auto generated by Alembic - please adjust! ### + op.add_column("AttackIdentifiers", sa.Column("use_score_as_feedback", sa.Boolean(), nullable=True)) + # ### end Alembic commands ### + + +def downgrade() -> None: + """Revert this schema upgrade.""" + # ### commands auto generated by Alembic - please adjust! ### + op.drop_column("AttackIdentifiers", "use_score_as_feedback") + # ### end Alembic commands ### diff --git a/pyrit/memory/memory_models.py b/pyrit/memory/memory_models.py index 55964896bb..0c0248c939 100644 --- a/pyrit/memory/memory_models.py +++ b/pyrit/memory/memory_models.py @@ -841,6 +841,7 @@ class AttackIdentifierEntry(ComponentIdentifierEntry[AttackIdentifier]): adversarial_system_prompt: Mapped[str | None] = mapped_column(Unicode, nullable=True) adversarial_seed_prompt: Mapped[str | None] = mapped_column(Unicode, nullable=True) adversarial_prompt_template: Mapped[str | None] = mapped_column(Unicode, nullable=True) + use_score_as_feedback: Mapped[bool | None] = mapped_column(Boolean, nullable=True) objective_target_hash: Mapped[str | None] = mapped_column( String(64), ForeignKey(f"{TargetIdentifierEntry.__tablename__}.hash"), nullable=True ) diff --git a/pyrit/models/identifiers/attack_identifier.py b/pyrit/models/identifiers/attack_identifier.py index 360862ec05..ee39245acf 100644 --- a/pyrit/models/identifiers/attack_identifier.py +++ b/pyrit/models/identifiers/attack_identifier.py @@ -41,6 +41,8 @@ class AttackIdentifier(ComponentIdentifier): adversarial_seed_prompt: Annotated[str | None, Evaluate.Include()] = None #: Effective per-turn adversarial prompt template text, if the strategy uses one. adversarial_prompt_template: Annotated[str | None, Evaluate.Include()] = None + #: ``False`` when the adversarial chat does not see scorer rationales; omitted when enabled (the default). + use_score_as_feedback: Annotated[bool | None, Evaluate.Include()] = None #: The objective target the attack drives. objective_target: Annotated[TargetIdentifier | None, Evaluate.Include(only_params=frozenset({"temperature"}))] = ( None diff --git a/pyrit/scenario/core/attack_technique_factory.py b/pyrit/scenario/core/attack_technique_factory.py index 9ac1af5b0e..085dd9216a 100644 --- a/pyrit/scenario/core/attack_technique_factory.py +++ b/pyrit/scenario/core/attack_technique_factory.py @@ -96,6 +96,7 @@ def __init__( uses_adversarial: bool | None = None, supports_additional_request_converters: bool = False, scorer_override_policy: ScorerOverridePolicy = ScorerOverridePolicy.WARN, + use_score_as_feedback: bool | None = None, ) -> None: """ Initialize the factory with a technique-specific configuration. @@ -145,14 +146,23 @@ class constructor signature and seed-technique shape. scorer_override_policy: What to do when a scenario's scorer is incompatible with the attack's ``attack_scoring_config`` type annotation. Defaults to WARN. + use_score_as_feedback: Optional technique-level override for + ``AttackScoringConfig.use_score_as_feedback``. When set, ``create()`` + applies it to a copy of the scenario's scoring config, keeping the + scenario's scorers. Use ``False`` for techniques whose attacker must + not see scorer rationales while every turn is still scored. ``None`` + (the default) leaves the scenario's config unchanged. Only applies when + the scenario's config is forwarded to the attack (see + ``scorer_override_policy``). Raises: TypeError: If any kwarg name is not a valid constructor parameter, or if the attack class constructor uses ``**kwargs``. ValueError: If ``objective_target`` or ``attack_adversarial_config`` is included in ``attack_kwargs``, - or if ``uses_adversarial=False`` while an adversarial chat or - prompt is wired. + if ``uses_adversarial=False`` while an adversarial chat or + prompt is wired, or if ``use_score_as_feedback`` is set for an + attack that does not accept ``attack_scoring_config``. """ self._name = name self._attack_class = attack_class @@ -172,12 +182,14 @@ class constructor signature and seed-technique shape. self._seed_technique = seed_technique self._supports_additional_request_converters = supports_additional_request_converters self._scorer_override_policy = scorer_override_policy + self._use_score_as_feedback = use_score_as_feedback self._uses_adversarial = uses_adversarial if uses_adversarial is not None else self._derive_uses_adversarial() self._validate_kwargs() self._validate_converter_composition() self._validate_adversarial_flags() + self._validate_score_feedback_override() @classmethod def with_simulated_conversation( @@ -393,6 +405,20 @@ def _validate_converter_composition(self) -> None: f"but {self._attack_class.__name__} does not accept 'attack_converter_config'." ) + def _validate_score_feedback_override(self) -> None: + """ + Validate that a feedback override can reach the attack's scoring config. + + Raises: + ValueError: If ``use_score_as_feedback`` is set but the attack class + does not accept ``attack_scoring_config``. + """ + if self._use_score_as_feedback is not None and "attack_scoring_config" not in self._get_accepted_params(): + raise ValueError( + f"Factory '{self._name}': use_score_as_feedback requires {self._attack_class.__name__} " + f"to accept 'attack_scoring_config'." + ) + def _validate_kwargs(self) -> None: """ Validate that all kwargs are valid parameters for the attack class constructor. @@ -725,7 +751,9 @@ class constructor accepts ``attack_converter_config``. attack_scoring_config=attack_scoring_config, accepted_params=accepted_params, ): - kwargs["attack_scoring_config"] = attack_scoring_config + kwargs["attack_scoring_config"] = self._apply_score_feedback_override( + attack_scoring_config=attack_scoring_config + ) if "attack_adversarial_config" in accepted_params and ( create_time_target is not None or adversarial_system_prompt is not None @@ -751,6 +779,29 @@ class constructor accepts ``attack_converter_config``. attack = self._attack_class(**kwargs) return AttackTechnique(attack=attack, seed_technique=self._seed_technique) + def _apply_score_feedback_override(self, *, attack_scoring_config: AttackScoringConfig) -> AttackScoringConfig: + """ + Apply this technique's ``use_score_as_feedback`` override to the scenario's scoring config. + + A shallow copy keeps the scenario's scorers and config subtype (e.g. TAP's) without + re-running its constructor, and leaves the caller's config unchanged. + + Args: + attack_scoring_config: The scoring config supplied by the caller. + + Returns: + AttackScoringConfig: The caller's config when no override applies, otherwise + a copy with the override applied. + """ + if ( + self._use_score_as_feedback is None + or attack_scoring_config.use_score_as_feedback == self._use_score_as_feedback + ): + return attack_scoring_config + overridden = copy.copy(attack_scoring_config) + overridden.use_score_as_feedback = self._use_score_as_feedback + return overridden + def _compose_converter_config( self, *, @@ -1036,8 +1087,9 @@ def _build_identifier(self) -> ComponentIdentifier: Build the behavioral identity for this factory. Includes the factory name, attack class, kwargs, adversarial chat, the - adversarial system-prompt prefix, and the adversarial-flag booleans so - factories with different configurations produce different hashes. When a + adversarial system-prompt prefix, the score-feedback override, and the + adversarial-flag booleans so factories with different configurations + produce different hashes. When a seed technique is present, its seeds are added as ``children["technique_seeds"]``. Returns: @@ -1063,6 +1115,8 @@ def _build_identifier(self) -> ComponentIdentifier: params["adversarial_prompt_template"] = self._serialize_value(self._adversarial_prompt_template) if self._adversarial_system_prompt_prefix is not None: params["adversarial_system_prompt_prefix"] = self._adversarial_system_prompt_prefix + if self._use_score_as_feedback is not None: + params["use_score_as_feedback"] = self._use_score_as_feedback children: dict[str, Any] = {} if self._seed_technique is not None: diff --git a/pyrit/setup/initializers/techniques/extra.py b/pyrit/setup/initializers/techniques/extra.py index e80773c30a..07cb6757b3 100644 --- a/pyrit/setup/initializers/techniques/extra.py +++ b/pyrit/setup/initializers/techniques/extra.py @@ -98,6 +98,8 @@ def get_technique_factories() -> list[AttackTechniqueFactory]: adversarial_prompt_template=SeedPrompt.from_yaml_file( EXECUTOR_RED_TEAM_PATH / "goat_follow_up_prompt.yaml" ), + # GOAT's attacker never sees judge output (paper section 3.3); every turn is still scored. + use_score_as_feedback=False, ), AttackTechniqueFactory( name="split_payload", diff --git a/tests/unit/executor/attack/core/test_attack_strategy.py b/tests/unit/executor/attack/core/test_attack_strategy.py index 2be788b70d..7180c973ec 100644 --- a/tests/unit/executor/attack/core/test_attack_strategy.py +++ b/tests/unit/executor/attack/core/test_attack_strategy.py @@ -9,7 +9,7 @@ import pytest from pyrit.exceptions.retry_collector import RetryCollector, get_retry_collector -from pyrit.executor.attack.core.attack_config import AttackAdversarialConfig +from pyrit.executor.attack.core.attack_config import AttackAdversarialConfig, AttackScoringConfig from pyrit.executor.attack.core.attack_parameters import AttackParameters from pyrit.executor.attack.core.attack_strategy import ( AttackContext, @@ -1319,9 +1319,10 @@ def _adv_target(*, model_name: str = "gpt-adv", extra_params: dict | None = None class _IdentityTestStrategy(AttackStrategy): """Minimal concrete strategy that exposes a settable adversarial config for identity tests.""" - def __init__(self, *, objective_target, adversarial_config=None): + def __init__(self, *, objective_target, adversarial_config=None, scoring_config=None): super().__init__(context_type=AttackContext, objective_target=objective_target) self._test_adversarial_config = adversarial_config + self._test_scoring_config = scoring_config def _validate_context(self, *, context): pass @@ -1345,6 +1346,9 @@ async def _teardown_async(self, *, context): def get_attack_adversarial_config(self): return self._test_adversarial_config + def get_attack_scoring_config(self): + return self._test_scoring_config + def _eval_hash(attack_identifier: ComponentIdentifier) -> str: composite = AtomicAttackIdentifier.build(attack_identifier=attack_identifier) @@ -1504,3 +1508,47 @@ def test_adversarial_presence_changes_hash_vs_none(self, mock_objective_target): adversarial_config=AttackAdversarialConfig(target=_adv_target(), system_prompt=None, first_message=None), ) assert plain.get_identifier().hash != adversarial.get_identifier().hash + + +@pytest.mark.usefixtures("patch_central_database") +class TestCreateIdentifierScoreFeedback: + """Tests for ``use_score_as_feedback`` in the attack identifier (component + eval hash).""" + + def test_disabled_feedback_stored_in_params(self, mock_objective_target): + strategy = _IdentityTestStrategy( + objective_target=mock_objective_target, + scoring_config=AttackScoringConfig(use_score_as_feedback=False), + ) + assert strategy.get_identifier().params["use_score_as_feedback"] is False + + @pytest.mark.parametrize( + "scoring_config", [None, AttackScoringConfig(), AttackScoringConfig(use_score_as_feedback=True)] + ) + def test_default_feedback_omitted_from_params(self, mock_objective_target, scoring_config): + """Enabled feedback is the default, so it is omitted to keep existing hashes stable.""" + strategy = _IdentityTestStrategy(objective_target=mock_objective_target, scoring_config=scoring_config) + assert "use_score_as_feedback" not in strategy.get_identifier().params + + def test_enabled_feedback_hash_matches_attack_without_scoring_config(self, mock_objective_target): + """Attacks created before this field existed must keep their component and eval hashes.""" + without_config = _IdentityTestStrategy(objective_target=mock_objective_target) + with_default_config = _IdentityTestStrategy( + objective_target=mock_objective_target, scoring_config=AttackScoringConfig() + ) + assert without_config.get_identifier().hash == with_default_config.get_identifier().hash + assert _eval_hash(without_config.get_identifier()) == _eval_hash(with_default_config.get_identifier()) + + def test_different_feedback_changes_full_and_eval_hash(self, mock_objective_target): + """Regression test: scenario resume matches completed objectives by eval hash, so attacks + that differ only in whether the adversarial chat sees the scorer rationale must not collide.""" + with_feedback = _IdentityTestStrategy( + objective_target=mock_objective_target, + scoring_config=AttackScoringConfig(use_score_as_feedback=True), + ) + without_feedback = _IdentityTestStrategy( + objective_target=mock_objective_target, + scoring_config=AttackScoringConfig(use_score_as_feedback=False), + ) + id1, id2 = with_feedback.get_identifier(), without_feedback.get_identifier() + assert id1.hash != id2.hash + assert _eval_hash(id1) != _eval_hash(id2) diff --git a/tests/unit/memory/memory_interface/test_interface_identifiers.py b/tests/unit/memory/memory_interface/test_interface_identifiers.py index 8f638655e7..34b33f688d 100644 --- a/tests/unit/memory/memory_interface/test_interface_identifiers.py +++ b/tests/unit/memory/memory_interface/test_interface_identifiers.py @@ -7,7 +7,7 @@ import pytest from pyrit.memory import MemoryInterface -from pyrit.memory.memory_models import TargetIdentifierEntry +from pyrit.memory.memory_models import AttackIdentifierEntry, TargetIdentifierEntry from pyrit.models import ( AtomicAttackIdentifier, AttackIdentifier, @@ -166,6 +166,23 @@ def test_get_target_identifiers_by_hash_and_promoted_field(sqlite_instance: Memo assert sqlite_instance.get_target_identifiers(supported_auth_modes=["API_KEY", "identity"]) == [] +def test_attack_identifier_disabled_score_feedback_round_trips(sqlite_instance: MemoryInterface) -> None: + attack = AttackIdentifier(class_name="TestAttack", class_module="tests.unit.memory", use_score_as_feedback=False) + + with closing(sqlite_instance.get_session()) as session: + sqlite_instance._persist_identifier(session=session, identifier=attack) + session.commit() + entry = session.get(AttackIdentifierEntry, attack.hash) + assert entry is not None + assert entry.use_score_as_feedback is False + + identifiers = sqlite_instance.get_attack_identifiers(identifier_hashes=[attack.hash]) + + assert identifiers == [attack] + assert identifiers[0].use_score_as_feedback is False + assert identifiers[0].hash == attack.hash + + def test_get_identifiers_reconstructs_each_typed_graph( sqlite_instance: MemoryInterface, identifier_graph: IdentifierGraph ) -> None: diff --git a/tests/unit/memory/test_migration.py b/tests/unit/memory/test_migration.py index de22018a6c..2d1d1b7c7f 100644 --- a/tests/unit/memory/test_migration.py +++ b/tests/unit/memory/test_migration.py @@ -196,6 +196,27 @@ def test_seed_conditions_and_follow_up_template_migrations_merge(starting_revisi engine.dispose() +def test_score_feedback_migration_adds_attack_identifier_column() -> None: + engine = create_engine("sqlite:///:memory:") + try: + with engine.begin() as connection: + config = _config_for(connection) + command.upgrade(config, "aca1eba410d9") + assert "use_score_as_feedback" not in { + column["name"] for column in inspect(connection).get_columns("AttackIdentifiers") + } + + run_schema_migrations(engine=engine) + check_schema_migrations(engine=engine) + + with engine.connect() as connection: + assert "use_score_as_feedback" in { + column["name"] for column in inspect(connection).get_columns("AttackIdentifiers") + } + finally: + engine.dispose() + + def test_scenario_progress_migration_adds_composite_index(): """The migration head contains the parent/timestamp/id keyset index.""" with tempfile.TemporaryDirectory() as temp_dir: diff --git a/tests/unit/scenario/core/test_attack_technique_factory.py b/tests/unit/scenario/core/test_attack_technique_factory.py index 0e5f6f8cbf..a0cd27c2c8 100644 --- a/tests/unit/scenario/core/test_attack_technique_factory.py +++ b/tests/unit/scenario/core/test_attack_technique_factory.py @@ -15,12 +15,14 @@ AttackConverterConfig, AttackScoringConfig, ) +from pyrit.executor.attack.multi_turn.tree_of_attacks import TAPAttackScoringConfig from pyrit.executor.attack.single_turn.prompt_sending import PromptSendingAttack from pyrit.models import AttackTechniqueSeedGroup, ComponentIdentifier, Identifiable, SeedPrompt from pyrit.prompt_normalizer import ConverterConfiguration from pyrit.prompt_target import PromptTarget from pyrit.scenario.core.attack_technique import AttackTechnique from pyrit.scenario.core.attack_technique_factory import AttackTechniqueFactory, ScorerOverridePolicy +from pyrit.score import FloatScaleThresholdScorer, Scorer, TrueFalseScorer def _make_seed_technique() -> AttackTechniqueSeedGroup: @@ -1390,3 +1392,105 @@ def test_final_user_message_disables_next_message_prompt(self): prompts = list(factory.seed_technique.prompts) assert prompts[0].value == "yes." assert prompts[0].sequence == sim.sequence_range.stop + + +class TestScoreFeedbackOverride: + """Tests for the technique-level ``use_score_as_feedback`` override.""" + + class _AdversarialAttack: + def __init__(self, *, objective_target=None, attack_scoring_config=None, attack_adversarial_config=None): + self.attack_scoring_config = attack_scoring_config + + def get_identifier(self): + return ComponentIdentifier(class_name="_AdversarialAttack", class_module="test") + + def test_unset_passes_scenario_config_through(self): + factory = AttackTechniqueFactory(name="test", attack_class=_StubAttack) + scoring = AttackScoringConfig() + + technique = factory.create(objective_target=MagicMock(spec=PromptTarget), attack_scoring_config=scoring) + + assert technique.attack.attack_scoring_config is scoring + + def test_override_applied_to_copy_keeping_scenario_scorers(self): + factory = AttackTechniqueFactory(name="test", attack_class=_StubAttack, use_score_as_feedback=False) + objective_scorer = MagicMock(spec=TrueFalseScorer) + auxiliary_scorers = [MagicMock(spec=Scorer)] + scoring = AttackScoringConfig(objective_scorer=objective_scorer, auxiliary_scorers=auxiliary_scorers) + + technique = factory.create(objective_target=MagicMock(spec=PromptTarget), attack_scoring_config=scoring) + + applied = technique.attack.attack_scoring_config + assert applied is not scoring + assert applied.use_score_as_feedback is False + assert applied.objective_scorer is objective_scorer + assert applied.auxiliary_scorers == auxiliary_scorers + # The scenario's config is shared across techniques and must not be changed. + assert scoring.use_score_as_feedback is True + + def test_matching_scenario_value_passes_config_through(self): + factory = AttackTechniqueFactory(name="test", attack_class=_StubAttack, use_score_as_feedback=False) + scoring = AttackScoringConfig(use_score_as_feedback=False) + + technique = factory.create(objective_target=MagicMock(spec=PromptTarget), attack_scoring_config=scoring) + + assert technique.attack.attack_scoring_config is scoring + + def test_override_keeps_scoring_config_subtype(self): + """TAP's config has its own constructor; copying must keep its type and threshold.""" + + class _TapStubAttack: + def __init__(self, *, objective_target, attack_scoring_config: TAPAttackScoringConfig | None = None): + self.attack_scoring_config = attack_scoring_config + + def get_identifier(self): + return ComponentIdentifier(class_name="_TapStubAttack", class_module="test") + + objective_scorer = MagicMock(spec=FloatScaleThresholdScorer) + objective_scorer.threshold = 0.7 + factory = AttackTechniqueFactory(name="test", attack_class=_TapStubAttack, use_score_as_feedback=False) + scoring = TAPAttackScoringConfig(objective_scorer=objective_scorer) + + technique = factory.create(objective_target=MagicMock(spec=PromptTarget), attack_scoring_config=scoring) + + applied = technique.attack.attack_scoring_config + assert type(applied) is TAPAttackScoringConfig + assert applied.use_score_as_feedback is False + assert applied.objective_scorer is objective_scorer + assert applied.threshold == 0.7 + assert scoring.use_score_as_feedback is True + + def test_override_requires_attack_scoring_config_param(self): + class _NoScoringAttack: + def __init__(self, *, objective_target): + pass + + def get_identifier(self): + return ComponentIdentifier(class_name="_NoScoringAttack", class_module="test") + + with pytest.raises(ValueError, match="use_score_as_feedback requires _NoScoringAttack"): + AttackTechniqueFactory(name="test", attack_class=_NoScoringAttack, use_score_as_feedback=False) + + def test_identifier_includes_override_only_when_set(self): + unset = AttackTechniqueFactory(name="test", attack_class=_StubAttack) + disabled = AttackTechniqueFactory(name="test", attack_class=_StubAttack, use_score_as_feedback=False) + enabled = AttackTechniqueFactory(name="test", attack_class=_StubAttack, use_score_as_feedback=True) + + assert "use_score_as_feedback" not in unset.get_identifier().params + assert disabled.get_identifier().params["use_score_as_feedback"] is False + assert len({unset.get_identifier().hash, disabled.get_identifier().hash, enabled.get_identifier().hash}) == 3 + + def test_prefixed_copy_keeps_override(self): + factory = AttackTechniqueFactory( + name="test", attack_class=self._AdversarialAttack, use_score_as_feedback=False, uses_adversarial=True + ) + + prefixed = factory.with_adversarial_system_prompt_prefix("Static guidance") + technique = prefixed.create( + objective_target=MagicMock(spec=PromptTarget), + attack_scoring_config=AttackScoringConfig(), + adversarial_chat=MagicMock(spec=PromptTarget), + ) + + assert technique.attack.attack_scoring_config.use_score_as_feedback is False + assert prefixed.get_identifier().params["use_score_as_feedback"] is False diff --git a/tests/unit/setup/test_technique_initializer.py b/tests/unit/setup/test_technique_initializer.py index 834b75ab53..a3e1467faf 100644 --- a/tests/unit/setup/test_technique_initializer.py +++ b/tests/unit/setup/test_technique_initializer.py @@ -7,10 +7,12 @@ from unittest.mock import MagicMock, patch import pytest +from unit.mocks import MockPromptTarget, get_mock_scorer_identifier from pyrit.common.path import EXECUTOR_RED_TEAM_PATH, EXECUTOR_SEED_PROMPT_PATH from pyrit.converter import CharNoiseConverter, CharSwapConverter, RandomCapitalLettersConverter from pyrit.executor.attack import ( + AttackScoringConfig, CrescendoAttack, PAIRAttack, PromptSendingAttack, @@ -21,6 +23,7 @@ from pyrit.prompt_target import PromptTarget from pyrit.registry import TargetRegistry from pyrit.registry.components.attack_technique_registry import AttackTechniqueRegistry +from pyrit.score import TrueFalseScorer from pyrit.score.true_false.self_ask_true_false_scorer import TrueFalseQuestionPaths from pyrit.setup.initializers import TechniqueInitializer from pyrit.setup.initializers.techniques import ( @@ -746,6 +749,33 @@ def test_effective_adversarial_config_uses_goat_prompts(self): assert isinstance(config.adversarial_prompt_template, SeedPrompt) assert "feedback_text" in (config.adversarial_prompt_template.parameters or []) + @pytest.mark.usefixtures("patch_central_database") + def test_created_attack_scores_every_turn_without_judge_feedback(self): + """Paper section 3.3: GOAT's attacker never sees judge output. The created attack must + disable score feedback while still scoring every turn (early stopping), leave the + scenario's shared config unchanged, and carry the setting into its eval identity.""" + objective_scorer = MagicMock(spec=TrueFalseScorer) + objective_scorer.get_identifier.return_value = get_mock_scorer_identifier() + scenario_scoring = AttackScoringConfig(objective_scorer=objective_scorer) + + attack = ( + self._goat_factory() + .create( + objective_target=MockPromptTarget(), + attack_scoring_config=scenario_scoring, + adversarial_chat=MockPromptTarget(), + ) + .attack + ) + + assert isinstance(attack, RedTeamingAttack) + applied = attack.get_attack_scoring_config() + assert applied.use_score_as_feedback is False + assert applied.objective_scorer is objective_scorer + assert attack._score_last_turn_only is False + assert scenario_scoring.use_score_as_feedback is True + assert attack.get_identifier().params["use_score_as_feedback"] is False + async def test_registered_when_extra_selected(self, mock_adversarial_target): init = TechniqueInitializer() init.params = {"tags": ["extra"]} From 698a20ae59d8d6ad7f246d68237b228c18dc0fd1 Mon Sep 17 00:00:00 2001 From: shashank Date: Mon, 28 Sep 2026 12:24:13 +0530 Subject: [PATCH 2/2] FIX Apply the feedback override when the scenario config is skipped When the scenario's scoring config is not forwarded (for example a plain AttackScoringConfig for TAP under WARN or SKIP), the override was never applied and the attack built its default config with feedback on. Apply the override to the attack_scoring_config baked into attack_kwargs instead, and raise when there is none rather than silently running with the default. --- .../scenario/core/attack_technique_factory.py | 33 +++++++--- .../core/test_attack_technique_factory.py | 63 ++++++++++++++++++- 2 files changed, 86 insertions(+), 10 deletions(-) diff --git a/pyrit/scenario/core/attack_technique_factory.py b/pyrit/scenario/core/attack_technique_factory.py index 085dd9216a..c0ceaef1cb 100644 --- a/pyrit/scenario/core/attack_technique_factory.py +++ b/pyrit/scenario/core/attack_technique_factory.py @@ -151,9 +151,10 @@ class constructor signature and seed-technique shape. applies it to a copy of the scenario's scoring config, keeping the scenario's scorers. Use ``False`` for techniques whose attacker must not see scorer rationales while every turn is still scored. ``None`` - (the default) leaves the scenario's config unchanged. Only applies when - the scenario's config is forwarded to the attack (see - ``scorer_override_policy``). + (the default) leaves the scenario's config unchanged. When the + scenario's config is not forwarded (see ``scorer_override_policy``), + the override is applied to the ``attack_scoring_config`` in + ``attack_kwargs`` instead, and ``create()`` raises if there is none. Raises: TypeError: If any kwarg name is not a valid constructor parameter, @@ -721,8 +722,9 @@ class constructor accepts ``attack_converter_config``. Raises: ValueError: If a create-time adversarial chat is supplied while the - factory already baked one, or if ``scorer_override_policy`` is RAISE - and the scenario scorer is incompatible with the attack's type annotation. + factory already baked one, if ``scorer_override_policy`` is RAISE + and the scenario scorer is incompatible with the attack's type annotation, + or if ``use_score_as_feedback`` is set but no scoring config reaches the attack. """ create_time_target: PromptTarget | None = adversarial_chat @@ -754,6 +756,18 @@ class constructor accepts ``attack_converter_config``. kwargs["attack_scoring_config"] = self._apply_score_feedback_override( attack_scoring_config=attack_scoring_config ) + elif self._use_score_as_feedback is not None: + # The scenario's config was skipped, so the override must reach the config the attack + # will actually use. Without a baked config the attack builds its own default, which + # cannot honor the override, so reject instead of silently running with the default. + baked_config = kwargs.get("attack_scoring_config") + if baked_config is None: + raise ValueError( + f"Factory '{self._name}': use_score_as_feedback={self._use_score_as_feedback} cannot be " + f"applied because the {type(attack_scoring_config).__name__} was not forwarded to " + f"{self._attack_class.__name__} and no attack_scoring_config is set in attack_kwargs." + ) + kwargs["attack_scoring_config"] = self._apply_score_feedback_override(attack_scoring_config=baked_config) if "attack_adversarial_config" in accepted_params and ( create_time_target is not None or adversarial_system_prompt is not None @@ -781,13 +795,14 @@ class constructor accepts ``attack_converter_config``. def _apply_score_feedback_override(self, *, attack_scoring_config: AttackScoringConfig) -> AttackScoringConfig: """ - Apply this technique's ``use_score_as_feedback`` override to the scenario's scoring config. + Apply this technique's ``use_score_as_feedback`` override to a scoring config. - A shallow copy keeps the scenario's scorers and config subtype (e.g. TAP's) without - re-running its constructor, and leaves the caller's config unchanged. + A shallow copy keeps the config's scorers and subtype (e.g. TAP's) without + re-running its constructor, and leaves the original config unchanged. Args: - attack_scoring_config: The scoring config supplied by the caller. + attack_scoring_config: The scenario's config, or the baked config when the + scenario's config is not forwarded. Returns: AttackScoringConfig: The caller's config when no override applies, otherwise diff --git a/tests/unit/scenario/core/test_attack_technique_factory.py b/tests/unit/scenario/core/test_attack_technique_factory.py index a0cd27c2c8..f0968ac676 100644 --- a/tests/unit/scenario/core/test_attack_technique_factory.py +++ b/tests/unit/scenario/core/test_attack_technique_factory.py @@ -15,7 +15,7 @@ AttackConverterConfig, AttackScoringConfig, ) -from pyrit.executor.attack.multi_turn.tree_of_attacks import TAPAttackScoringConfig +from pyrit.executor.attack.multi_turn.tree_of_attacks import TAPAttackScoringConfig, TreeOfAttacksWithPruningAttack from pyrit.executor.attack.single_turn.prompt_sending import PromptSendingAttack from pyrit.models import AttackTechniqueSeedGroup, ComponentIdentifier, Identifiable, SeedPrompt from pyrit.prompt_normalizer import ConverterConfiguration @@ -1460,6 +1460,67 @@ def get_identifier(self): assert applied.threshold == 0.7 assert scoring.use_score_as_feedback is True + @pytest.mark.parametrize("policy", [ScorerOverridePolicy.WARN, ScorerOverridePolicy.SKIP]) + def test_skipped_scenario_config_without_baked_config_raises(self, policy): + """A skipped scenario config leaves the attack to build its own default, which would + silently run with feedback on, so create() must reject the technique instead.""" + + class _TapStubAttack: + def __init__(self, *, objective_target, attack_scoring_config: TAPAttackScoringConfig | None = None): + self.attack_scoring_config = attack_scoring_config + + def get_identifier(self): + return ComponentIdentifier(class_name="_TapStubAttack", class_module="test") + + factory = AttackTechniqueFactory( + name="test", attack_class=_TapStubAttack, use_score_as_feedback=False, scorer_override_policy=policy + ) + + with pytest.raises(ValueError, match="use_score_as_feedback=False cannot be applied"): + factory.create(objective_target=MagicMock(spec=PromptTarget), attack_scoring_config=AttackScoringConfig()) + + def test_skipped_scenario_config_applies_override_to_baked_config(self): + class _TapStubAttack: + def __init__(self, *, objective_target, attack_scoring_config: TAPAttackScoringConfig | None = None): + self.attack_scoring_config = attack_scoring_config + + def get_identifier(self): + return ComponentIdentifier(class_name="_TapStubAttack", class_module="test") + + objective_scorer = MagicMock(spec=FloatScaleThresholdScorer) + baked = TAPAttackScoringConfig(objective_scorer=objective_scorer) + factory = AttackTechniqueFactory( + name="test", + attack_class=_TapStubAttack, + attack_kwargs={"attack_scoring_config": baked}, + use_score_as_feedback=False, + scorer_override_policy=ScorerOverridePolicy.SKIP, + ) + + technique = factory.create( + objective_target=MagicMock(spec=PromptTarget), attack_scoring_config=AttackScoringConfig() + ) + + applied = technique.attack.attack_scoring_config + assert type(applied) is TAPAttackScoringConfig + assert applied.use_score_as_feedback is False + assert applied.objective_scorer is objective_scorer + assert baked.use_score_as_feedback is True + + def test_tap_with_plain_scenario_config_rejects_instead_of_enabling_feedback(self): + """Regression test: TAP requires TAPAttackScoringConfig, so a plain scenario config is + skipped and TAP would otherwise build its default config with feedback on.""" + factory = AttackTechniqueFactory( + name="tap_no_feedback", attack_class=TreeOfAttacksWithPruningAttack, use_score_as_feedback=False + ) + + with pytest.raises(ValueError, match="not forwarded to TreeOfAttacksWithPruningAttack"): + factory.create( + objective_target=MagicMock(spec=PromptTarget), + attack_scoring_config=AttackScoringConfig(objective_scorer=MagicMock(spec=TrueFalseScorer)), + adversarial_chat=MagicMock(spec=PromptTarget), + ) + def test_override_requires_attack_scoring_config_param(self): class _NoScoringAttack: def __init__(self, *, objective_target):