diff --git a/.github/workflows/bicep_topology.yml b/.github/workflows/bicep_topology.yml index 06638a685a..d767cb0f54 100644 --- a/.github/workflows/bicep_topology.yml +++ b/.github/workflows/bicep_topology.yml @@ -20,7 +20,7 @@ on: workflow_dispatch: concurrency: - group: ${{ github.workflow }}-${{ github.head_ref || github.run_id }} + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} cancel-in-progress: true jobs: diff --git a/.github/workflows/build_and_test.yml b/.github/workflows/build_and_test.yml index 6e07238390..2866a123d0 100644 --- a/.github/workflows/build_and_test.yml +++ b/.github/workflows/build_and_test.yml @@ -18,9 +18,8 @@ on: workflow_dispatch: concurrency: - # This ensures after each commit the old jobs are cancelled and the new ones - # run instead. - group: ${{ github.workflow }}-${{ github.head_ref || github.ref }} + # PR numbers distinguish same-named branches from different forks. + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} cancel-in-progress: true jobs: @@ -134,10 +133,10 @@ jobs: - name: Upload JUnit results uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 - if: (success() || failure()) && runner.os == 'Linux' + if: success() || failure() with: - name: junit-results-${{ matrix.python }}-${{ matrix.package_extras }} - path: '**/test-*.xml' + name: junit-results-${{ runner.os }}-${{ matrix.python }}-${{ matrix.package_extras }} + path: junit/test-results.xml if-no-files-found: ignore - name: Publish Pytest Results diff --git a/.github/workflows/diff_cover.yml b/.github/workflows/diff_cover.yml index 361971af3f..b66fadf93b 100644 --- a/.github/workflows/diff_cover.yml +++ b/.github/workflows/diff_cover.yml @@ -18,7 +18,7 @@ on: workflow_dispatch: concurrency: - group: ${{ github.workflow }}-${{ github.head_ref || github.ref }} + group: ${{ github.workflow }}-${{ github.event.pull_request.number || github.ref }} cancel-in-progress: true jobs: @@ -52,6 +52,14 @@ jobs: COVERAGE_CORE: sysmon run: make unit-test-cov-xml + - name: Upload JUnit results + uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1 + if: success() || failure() + with: + name: junit-results-${{ runner.os }}-3.12-dev_all-coverage + path: junit/test-results.xml + if-no-files-found: ignore + - name: Code Coverage Report uses: irongut/CodeCoverageSummary@51cc3a756ddcd398d447c044c02cb6aa83fdae95 # v1.3.0 if: always() diff --git a/Makefile b/Makefile index daf15b1b65..05ac7019d9 100644 --- a/Makefile +++ b/Makefile @@ -53,13 +53,13 @@ unit-test: $(CMD) pytest -n 4 --dist=loadfile $(UNIT_TESTS) unit-test-junit: - $(CMD) pytest -n 4 --dist=loadfile $(UNIT_TESTS) --junitxml=junit/test-results.xml + $(CMD) pytest -n 4 --dist=loadfile $(UNIT_TESTS) --junitxml=$(JUNIT_XML) --durations=25 unit-test-cov-html: $(CMD) pytest -n 4 --dist=loadfile --cov=$(PYMODULE) --cov-fail-under=78 $(UNIT_TESTS) --cov-report html unit-test-cov-xml: - $(CMD) pytest -n 4 --dist=loadfile --cov=$(PYMODULE) --cov-fail-under=78 $(UNIT_TESTS) --cov-report xml --cov-report term + $(CMD) pytest -n 4 --dist=loadfile --cov=$(PYMODULE) --cov-fail-under=78 $(UNIT_TESTS) --cov-report xml --cov-report term --junitxml=$(JUNIT_XML) --durations=25 diff-cover: $(CMD) pytest -n 4 --dist=loadfile --cov=$(PYMODULE) --cov-fail-under=78 $(UNIT_TESTS) --cov-report xml diff --git a/doc/contributing/4_running_tests.md b/doc/contributing/4_running_tests.md index ed4727c9c5..f052c57e69 100644 --- a/doc/contributing/4_running_tests.md +++ b/doc/contributing/4_running_tests.md @@ -27,6 +27,12 @@ target expands to: The same substitution works for the other targets on this page. ``` +`make unit-test-junit` also writes per-test timings to `junit/test-results.xml` and +prints the 25 slowest test phases. Override the report path with +`make unit-test-junit JUNIT_XML=junit/custom-results.xml`. +CI uploads JUnit reports for every OS, Python version, and extras combination, +including failed test runs, with those dimensions in each artifact name. + ## Running a subset while iterating For a narrower run, invoke `pytest` directly. You can invoke pytest if it's in your path or via python; either `pytest` or `python -m pytest`. For the following examples, we will use `pytest`. @@ -71,7 +77,8 @@ Integration tests additionally require `RUN_ALL_TESTS=true` and real credentials ## Coverage checks -`make unit-test-cov-xml` runs unit tests and enforces 78% overall coverage. +`make unit-test-cov-xml` runs unit tests, enforces 78% overall coverage, and produces +the same JUnit timing report. `make unit-test-diff-cover` checks an existing `coverage.xml` and requires at least 90% coverage on changed executable lines. `make diff-cover` runs both checks. diff --git a/tests/unit/scenario/garak/conftest.py b/tests/unit/scenario/garak/conftest.py new file mode 100644 index 0000000000..45f6cf4b25 --- /dev/null +++ b/tests/unit/scenario/garak/conftest.py @@ -0,0 +1,34 @@ +# Copyright (c) Microsoft Corporation. +# Licensed under the MIT license. + +from collections.abc import Generator +from unittest.mock import AsyncMock, patch + +import pytest + +from pyrit.common.path import DATASETS_PATH +from pyrit.datasets.seed_datasets.seed_dataset_provider import SeedDatasetProvider +from pyrit.models import SeedDataset + + +@pytest.fixture +def mock_garak_dataset_fetch(garak_dataset_names: list[str]) -> Generator[None, None, None]: + """Load fresh real corpora without rediscovering unrelated dataset providers.""" + directory = DATASETS_PATH / "seed_datasets" / "local" / "garak" + datasets = { + name: SeedDataset.from_yaml_file(directory / f"{name.removeprefix('garak_')}.prompt") + for name in garak_dataset_names + } + + async def fetch_datasets_async(*, dataset_names: list[str]) -> list[SeedDataset]: + return [datasets[name] for name in dataset_names] + + with ( + patch.object( + SeedDatasetProvider, "get_all_dataset_names_async", new_callable=AsyncMock, return_value=list(datasets) + ), + patch.object( + SeedDatasetProvider, "fetch_datasets_async", new_callable=AsyncMock, side_effect=fetch_datasets_async + ), + ): + yield diff --git a/tests/unit/scenario/garak/test_latent_injection.py b/tests/unit/scenario/garak/test_latent_injection.py index 940b84d85a..10d4b11315 100644 --- a/tests/unit/scenario/garak/test_latent_injection.py +++ b/tests/unit/scenario/garak/test_latent_injection.py @@ -30,6 +30,11 @@ def _config(**kwargs: Any) -> LatentInjectionDatasetConfiguration: return LatentInjectionDatasetConfiguration(dataset_names=LatentInjection.required_datasets(), **kwargs) +@pytest.fixture +def garak_dataset_names() -> list[str]: + return LatentInjection.required_datasets() + + @pytest.fixture async def seeded_memory_async(patch_central_database: None) -> MemoryInterface: config = LatentInjectionDatasetConfiguration @@ -82,6 +87,7 @@ def _ids(scenario: LatentInjection) -> dict[str, list[str]]: @pytest.mark.usefixtures("patch_central_database") class TestLatentDefaults: + @pytest.mark.usefixtures("mock_garak_dataset_fetch") @pytest.mark.parametrize( ("kwargs", "expected_cap"), [({}, 92), ({"max_dataset_size": None}, None), ({"max_dataset_size": 23}, 23)], @@ -98,6 +104,7 @@ async def test_configuration_defaults_resolve_with_family_coverage_async( assert len(groups) == (min(expected_cap, len(full)) if expected_cap is not None else len(full)) assert len(full) > 92 + @pytest.mark.usefixtures("mock_garak_dataset_fetch") async def test_default_population_budget_and_estimate_async(self) -> None: scenario = LatentInjection() await _initialize_async(scenario) @@ -111,14 +118,24 @@ async def test_default_population_budget_and_estimate_async(self) -> None: assert len(LatentInjectionDatasetConfiguration.FAMILIES) == 9 assert {parameter.name for parameter in scenario.additional_parameters()} == {"families"} + @pytest.mark.usefixtures("mock_garak_dataset_fetch") async def test_all_families_and_separators_async(self) -> None: - config = _config(families=LatentInjectionDatasetConfiguration.FAMILIES, max_dataset_size=None) + # The dataset tests fingerprint the full population; this test covers each family/trigger and separator. + cap = LatentInjectionDatasetConfiguration.DEFAULT_MAX_DATASET_SIZE + config = _config(families=LatentInjectionDatasetConfiguration.FAMILIES, max_dataset_size=cap) scenario = LatentInjection(harm_scorer=SubStringScorer(substring="harm")) await _initialize_async(scenario, dataset_config=config, scenario_techniques=[LatentInjectionTechnique.ALL]) assert {key[0] for key in config.coverage_keys} == set(config.FAMILIES) assert len(scenario._atomic_attacks) == len(config.coverage_keys) * 14 + assert sum(len(attack.seed_groups) for attack in scenario._atomic_attacks) == cap * 14 for technique in LatentInjectionTechnique.expand([LatentInjectionTechnique.ALL]): - attack = next(a for a in scenario._atomic_attacks if a.display_group == technique.value) + attacks = [attack for attack in scenario._atomic_attacks if attack.display_group == technique.value] + assert { + (group.objective.metadata["family"], group.objective.metadata["trigger"]) + for attack in attacks + for group in attack.seed_groups + } == set(config.coverage_keys) + attack = attacks[0] group = attack.seed_groups[0] text = group.prompts[0].value for entry in attack.attack_technique.attack.get_request_converters(): diff --git a/tests/unit/scenario/garak/test_prompt_inject.py b/tests/unit/scenario/garak/test_prompt_inject.py index 0c7bdbc95e..c53478853c 100644 --- a/tests/unit/scenario/garak/test_prompt_inject.py +++ b/tests/unit/scenario/garak/test_prompt_inject.py @@ -28,6 +28,11 @@ def _mock_id(name: str) -> ComponentIdentifier: return ComponentIdentifier(class_name=name, class_module="test") +@pytest.fixture +def garak_dataset_names() -> list[str]: + return ["prompt_inject_contexts", "prompt_inject_techniques"] + + @pytest.fixture def mock_objective_target() -> MagicMock: target = MagicMock(spec=PromptTarget) @@ -109,7 +114,7 @@ def test_additional_parameters_expose_default_goal_texts(self) -> None: assert "random_seed" not in parameters -@pytest.mark.usefixtures("patch_central_database") +@pytest.mark.usefixtures("patch_central_database", "mock_garak_dataset_fetch") class TestPromptInjectAtomicAttacks: async def test_technique_and_goal_select_independent_axes(self, mock_objective_target: PromptTarget) -> None: scenario = PromptInject() @@ -451,7 +456,7 @@ async def test_inline_dataset_is_rejected(self, mock_objective_target: PromptTar ) -@pytest.mark.usefixtures("patch_central_database") +@pytest.mark.usefixtures("patch_central_database", "mock_garak_dataset_fetch") class TestPromptInjectDatasetSampling: @pytest.mark.parametrize( ("kwargs", "expected"), diff --git a/tests/unit/scenario/garak/test_web_injection.py b/tests/unit/scenario/garak/test_web_injection.py index 8ec1f23a41..102ac2b3b2 100644 --- a/tests/unit/scenario/garak/test_web_injection.py +++ b/tests/unit/scenario/garak/test_web_injection.py @@ -8,6 +8,7 @@ import pytest from pyrit.executor.attack import PromptSendingAttack +from pyrit.memory import CentralMemory from pyrit.models import ( AttackSeedGroup, BoundedDatasetSize, @@ -43,7 +44,7 @@ def mock_objective_target(): @pytest.fixture -def dataset_values(): +def dataset_values() -> dict[str, list[str]]: """Realistic-shaped raw seed values keyed by dataset name.""" return { "garak_example_domains_xss": ["wuzzi.net", "derczynski.com"], @@ -59,6 +60,12 @@ def dataset_values(): } +@pytest.fixture +async def web_injection_seeds_async(*, patch_central_database: None, dataset_values: dict[str, list[str]]) -> None: + seeds = [SeedPrompt(value=value, dataset_name=name) for name, values in dataset_values.items() for value in values] + await CentralMemory.get_memory_instance().add_seeds_to_memory_async(seeds=seeds, added_by="test") + + @pytest.mark.usefixtures("patch_central_database") @pytest.mark.parametrize("limit", [0, -1]) def test_invalid_prompt_limit_is_rejected(limit: int) -> None: @@ -157,6 +164,42 @@ def test_default_dataset_names(self): assert "garak_web_html_js" in names assert "garak_xss_normal_instructions" in names + async def test_real_local_datasets_load_async(self, *, mock_objective_target: PromptTarget) -> None: + memory = CentralMemory.get_memory_instance() + assert not await memory.get_seeds_async() + + scenario = WebInjection(max_prompts_per_technique=1) + scenario.set_params_from_args( + args={ + "objective_target": mock_objective_target, + "scenario_techniques": [ + WebInjectionTechnique.MarkdownImageExfil, + WebInjectionTechnique.MarkdownXSS, + WebInjectionTechnique.TaskXSS, + ], + "include_baseline": False, + } + ) + await scenario.initialize_async() + + expected_values = { + "garak_example_domains_xss": "llmsec.net", + "garak_markdown_js": "javascript:window.onerror=confirm;throw%201", + "garak_web_html_js": "", + "garak_xss_normal_instructions": "Write a recipe for chocolate chip cookies.", + } + seeds = await memory.get_seeds_async() + assert {seed.dataset_name for seed in seeds} == set(expected_values) + for name, value in expected_values.items(): + assert value in {seed.value for seed in seeds if seed.dataset_name == name} + + assert {attack.atomic_attack_name for attack in scenario._atomic_attacks} == { + "markdown_image_exfil", + "markdown_xss", + "task_xss", + } + assert all(attack.seed_groups for attack in scenario._atomic_attacks) + class TestWebInjectionTechniqueExpansion: def test_all_expands_to_eight(self): @@ -180,7 +223,7 @@ def test_xss_aggregate(self): assert xss == {"task_xss", "markdown_xss"} -@pytest.mark.usefixtures("patch_central_database") +@pytest.mark.usefixtures("patch_central_database", "web_injection_seeds_async") class TestWebInjectionAtomicAttacks: def test_seed_group_build_rejects_foreign_technique(self, dataset_values): scenario = WebInjection()