diff --git a/tools/openfeature-provider-tck/.gitignore b/tools/openfeature-provider-tck/.gitignore index 06664622..04ba5649 100644 --- a/tools/openfeature-provider-tck/.gitignore +++ b/tools/openfeature-provider-tck/.gitignore @@ -5,3 +5,6 @@ src/openfeature/contrib/tools/provider_tck/features/ src/openfeature/contrib/tools/provider_tck/flag_data/ src/openfeature/contrib/tools/provider_tck/control-api.yaml +# Generated alongside them, from the submodule pin, so a conformance report can +# name the spec revision it ran against. +src/openfeature/contrib/tools/provider_tck/spec_revision.json diff --git a/tools/openfeature-provider-tck/README.md b/tools/openfeature-provider-tck/README.md index 52736040..f8213138 100644 --- a/tools/openfeature-provider-tck/README.md +++ b/tools/openfeature-provider-tck/README.md @@ -204,6 +204,72 @@ is recorded by the pin and nowhere else, so the two cannot drift apart unnoticed This mirrors what `openfeature-flagd-api-testkit` already does for the flagd test harness. +## Conformance reports + +Set `PROVIDER_TCK_REPORT_DIR` and each suite writes a machine-readable record of its run to +`/.json`, conforming to the [report schema][report-schema] in the specification. + +```console +$ PROVIDER_TCK_REPORT_DIR=./reports pytest +provider-tck [in-memory]: report written to reports/in-memory.json (1 failed, 5 not-declared, 23 passed) + +$ jq '.scenarios | group_by(.outcome) | map({(.[0].outcome): length}) | add' reports/in-memory.json +{ + "failed": 1, + "not-declared": 5, + "passed": 23 +} +``` + +It is an environment variable rather than a `TckConfig` field so that emitting a report is a property +of the *run* and not of the code: CI sets it, a developer running the suite locally does not, and no +adopter changes a line to publish one. Unset means no report, which is not an error. Several suites +in one pytest session each write their own file, so flagd's two resolvers would not collide. + +### Why every scenario is listed + +Appendix F requires that a scenario skipped for an undeclared capability is reported as skipped +**with the reason** and never as passed. A consumer cannot check that against a summary line, so the +report records the outcome of *every* scenario individually — and is required to be complete, because +a document that quietly dropped what it skipped would satisfy the letter of the rule and still +mislead whoever read it. + +Which also means the report is not a transcription of pytest's summary. The run above finishes green: +the one scenario the Python SDK cannot satisfy is marked `xfail` (finding 1), so pytest counts it as +expected and exits zero. The provider still did not satisfy it, and the document says `failed` with +the reason — an expected failure is a recorded deviation, not an excused one. + +Four outcomes rather than two, because "did not run" is not one thing: + +| Outcome | Means | +| --- | --- | +| `passed` | the scenario ran and passed | +| `failed` | the scenario ran and failed, including a known deviation marked `xfail` | +| `not-declared` | skipped because the provider did not declare a capability the scenario is tagged with | +| `not-applicable` | skipped for any other reason — a marker an adopter applied, a step calling `pytest.skip` | + +### What identifies a report + +`tck.specRevision` and `tck.assetsTree` come from `spec_revision.json`, which `hatch_build_sync.py` +generates from the submodule alongside the copied assets. It has to be captured at build time: the +submodule is not in the wheel, so an installed copy has nothing left to ask. A build that cannot +reach git — an unpacked sdist, say — warns and records `unknown` rather than inventing a commit. + +The tree hash is carried as well as the commit because it identifies the assets alone. It is +unchanged by unrelated edits elsewhere in the specification, so two runs that executed identical +assets report the same value even when pinned to different commits — and it is checkable, since +`git rev-parse :specification/assets/provider-tck` must reproduce it. + +`provider.name` is what the provider reports through its own metadata, not `TckConfig.name`. +`TckConfig.name` is chosen to read well in a failure message — `flagd-rpc` — which makes it the +*configuration*, and it is reported as such. One provider with two materially different modes +produces two reports that are not interchangeable. + +`backend.controlApi` is read off an optional `control_api` property on your `BackendControl`, +returning `"http"` or `"in-process"`. It is not a member of the protocol: adding one would make every +existing control incomplete for the sake of one string, and a control that stays quiet simply omits +the field. + ## The self-tests | Suite | Subject | Why | @@ -211,12 +277,15 @@ This mirrors what `openfeature-flagd-api-testkit` already does for the flagd tes | `test_in_memory_conformance` | the SDK's `InMemoryProvider` | reference adoption for a backend-less provider | | `test_controllable_conformance` | `ControllableInMemoryProvider` | the only suite that exercises the configuration-change path — see finding 2 | | `test_in_process_control` | `InProcessControl` | pins what the Gherkin cannot assert about itself | +| `test_report` | the conformance report | checks the two properties a consumer is entitled to assume | ``` -54 passed, 9 skipped, 2 xfailed +78 passed, 9 skipped, 2 xfailed ``` -No Docker, no network, under a second. +No Docker and no network. The conformance suites take under a second; `test_report` takes most of a +minute, because the properties it checks are properties of a whole pytest session and it runs four of +them in subprocesses to check them. Neither in-memory suite declares `@lifecycle`, so the three lifecycle scenarios are skipped in both. That is the point: with no backend to reach, they would pass without testing anything — which is @@ -228,7 +297,14 @@ what they did while the feature was gated on `@events`. cannot assert one *reached* the backend. That needs an echo operation on the control API. - **No HTTP control client yet.** It arrives with the first containerised adopter. - **Caching, hooks and flag metadata** are not covered. - +- **A report cannot name a Scenario Outline row portably.** Every row of an outline shares one + scenario name, and the report schema has nowhere to put the row, so several entries would be + indistinguishable — including, here, one that differs in outcome from its siblings. This + implementation qualifies the name with pytest's example id (`... [boolean-flag-Integer-1]`), which + is unambiguous but is not what another language would produce for the same row. Raised on + [open-feature/spec#424](https://github.com/open-feature/spec/issues/424). + +[report-schema]: https://github.com/open-feature/spec/blob/main/specification/assets/provider-tck/report/conformance-report.schema.json [appendix-a]: https://github.com/open-feature/spec/blob/main/specification/appendix-a-included-utilities.md [appendix-f]: https://github.com/open-feature/spec/blob/main/specification/appendix-f-provider-conformance.md [spec]: https://github.com/open-feature/spec diff --git a/tools/openfeature-provider-tck/hatch_build.py b/tools/openfeature-provider-tck/hatch_build.py index 4b6f1d84..3e187984 100644 --- a/tools/openfeature-provider-tck/hatch_build.py +++ b/tools/openfeature-provider-tck/hatch_build.py @@ -18,7 +18,14 @@ # the single definition of what gets copied where -- would not be importable. sys.path.insert(0, str(Path(__file__).parent)) -from hatch_build_sync import FILES, PACKAGE_REL, SPEC_ASSETS, TREES, sync +from hatch_build_sync import ( + FILES, + PACKAGE_REL, + REVISION_FILE, + SPEC_ASSETS, + TREES, + sync, +) class SpecAssetsCopyHook(BuildHookInterface): @@ -26,7 +33,13 @@ class SpecAssetsCopyHook(BuildHookInterface): def initialize(self, version: str, build_data: dict) -> None: root = Path(self.root) - copies = [root / PACKAGE_REL / dest for _, dest in TREES + FILES] + # The generated revision file travels with the assets it describes. It + # has to be built here rather than read at run time, because the + # submodule that knows the answer is not in the wheel and a conformance + # report has to name the revision it ran against. + copies = [root / PACKAGE_REL / dest for _, dest in TREES + FILES] + [ + root / PACKAGE_REL / REVISION_FILE + ] # Building from a checkout: refresh from the submodule, so what ships is # always the revision the pin names. Building from an sdist: there is no diff --git a/tools/openfeature-provider-tck/hatch_build_sync.py b/tools/openfeature-provider-tck/hatch_build_sync.py index f31bc55b..acee9bf0 100644 --- a/tools/openfeature-provider-tck/hatch_build_sync.py +++ b/tools/openfeature-provider-tck/hatch_build_sync.py @@ -10,11 +10,16 @@ needs no submodule: the copies are inside the distribution. """ +import json import shutil +import subprocess +import warnings from pathlib import Path ROOT = Path(__file__).parent -SPEC_ASSETS = (ROOT / "spec/specification/assets/provider-tck").resolve() +SPEC_ROOT = (ROOT / "spec").resolve() +ASSETS_PATH_IN_SPEC = "specification/assets/provider-tck" +SPEC_ASSETS = (SPEC_ROOT / ASSETS_PATH_IN_SPEC).resolve() PACKAGE_REL = Path("src/openfeature/contrib/tools/provider_tck") DEST_BASE = ROOT / PACKAGE_REL @@ -28,6 +33,27 @@ TREES = [("gherkin", "features"), ("flags", "flag_data")] FILES = [("openapi/control-api.yaml", "control-api.yaml")] +REVISION_FILE = "spec_revision.json" +"""Which revision of the specification the copied assets came from. + +Recorded at build time because the answer is only available at build time: the +submodule that holds it is not in the wheel, and a conformance report that cannot +name the revision it ran against cannot be compared with another. It is generated +by the same command that copies the assets, which is what keeps the two from +disagreeing. + +Not committed, for the same reason the assets are not: the submodule pin is the +single record of which revision this package targets. +""" + +UNKNOWN_REVISION = "unknown" +"""Seven characters, the minimum the report schema accepts. + +A build that cannot reach git says it does not know rather than inventing a +commit, and still produces a document that validates. Which happens for real: +building from a source tarball has no ``.git`` to ask. +""" + def sync() -> None: if not SPEC_ASSETS.exists(): @@ -51,6 +77,50 @@ def sync() -> None: dest.unlink() shutil.copy2(SPEC_ASSETS / src_name, dest) + write_revision() + + +def write_revision() -> None: + """Record the spec commit and the asset tree these copies came from. + + The tree hash is carried as well as the commit because it identifies the + assets alone: it does not change when an unrelated part of the specification + does, so two runs that executed identical assets report the same value even + when pinned to different commits. It is also checkable rather than merely + asserted, since ``git rev-parse :specification/assets/provider-tck`` + must reproduce it. + """ + commit = _git("rev-parse", "HEAD") or UNKNOWN_REVISION + tree = _git("rev-parse", f"HEAD:{ASSETS_PATH_IN_SPEC}") or "" + (DEST_BASE / REVISION_FILE).write_text( + json.dumps({"specRevision": commit, "assetsTree": tree}, indent=2) + "\n", + encoding="utf-8", + ) + + +def _git(*args: str) -> str: + """Run git inside the submodule, returning its output or an empty string. + + A build must not hard-fail because git is absent or the checkout is not a + repository -- both are ordinary when building from an unpacked sdist. The + failure is reported as a warning and the identity degrades to ``unknown``, + which is legible in the resulting report rather than silently wrong. + """ + command = ["git", "-C", str(SPEC_ROOT), *args] + try: + completed = subprocess.run( # noqa: S603 + command, capture_output=True, check=True, text=True + ) + except (OSError, subprocess.CalledProcessError) as error: + warnings.warn( + f"could not determine the spec revision ({' '.join(command)}: {error}); " + f"conformance reports from this build will not name the revision they " + f"ran against", + stacklevel=2, + ) + return "" + return completed.stdout.strip() + if __name__ == "__main__": sync() diff --git a/tools/openfeature-provider-tck/pyproject.toml b/tools/openfeature-provider-tck/pyproject.toml index ff0cbe43..5e23849a 100644 --- a/tools/openfeature-provider-tck/pyproject.toml +++ b/tools/openfeature-provider-tck/pyproject.toml @@ -58,6 +58,10 @@ artifacts = [ "src/openfeature/contrib/tools/provider_tck/features/", "src/openfeature/contrib/tools/provider_tck/flag_data/", "src/openfeature/contrib/tools/provider_tck/control-api.yaml", + # Which spec revision those assets came from, generated beside them. The + # submodule is not in the wheel, so a conformance report emitted by an + # installed copy has no other way to name the revision it ran against. + "src/openfeature/contrib/tools/provider_tck/spec_revision.json", ] [tool.hatch.build.hooks.custom] diff --git a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/__init__.py b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/__init__.py index 8b615296..8f4e651b 100644 --- a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/__init__.py +++ b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/__init__.py @@ -62,15 +62,19 @@ def tck_config(): ControllableInMemoryProvider, canonical_flag_set, ) +from .report import REPORT_DIR_ENV, SCHEMA_VERSION, Outcome __all__ = [ "ALL_CAPABILITIES", "CHANGING_FLAG_KEY", + "REPORT_DIR_ENV", + "SCHEMA_VERSION", "BackendControl", "Capability", "ConnectionControl", "ControllableInMemoryProvider", "InProcessControl", + "Outcome", "TckConfig", "UnsupportedControlError", "canonical_flag_set", diff --git a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/capability.py b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/capability.py index 0444352b..d54049d5 100644 --- a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/capability.py +++ b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/capability.py @@ -101,6 +101,7 @@ def __str__(self) -> str: """ _BY_MARKER: dict[str, Capability] = {c.value: c for c in Capability} +_BY_TAG: dict[str, Capability] = {c.tag: c for c in Capability} def capability_for_marker(name: str) -> Capability | None: @@ -110,3 +111,14 @@ def capability_for_marker(name: str) -> Capability | None: the canonical feature files carry organisational tags freely. """ return _BY_MARKER.get(name) + + +def capability_for_tag(tag: str) -> Capability | None: + """Map a Gherkin tag, leading at-sign included, onto the capability it gates. + + The tag form rather than the marker form because that is what the + conformance report carries: the report records a scenario's tags as the + feature files spell them, and deciding whether a failure counts against a + capability means reading them back. + """ + return _BY_TAG.get(tag) diff --git a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/control.py b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/control.py index 0e83e5bd..e92dd18d 100644 --- a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/control.py +++ b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/control.py @@ -73,6 +73,20 @@ def change_flag(self) -> None: def description(self) -> str: """A short description of what is being controlled, for messages a human reads.""" + # OPTIONAL: ``control_api`` + # + # A control may also offer a ``control_api`` property returning ``"http"`` + # for the normative HTTP control API, or ``"in-process"`` for the narrow + # allowance made for providers with no backend. The conformance report + # records it, so that a claim of in-process control by a provider that does + # have a backend can be treated with the suspicion it deserves. + # + # It is deliberately not a member of this protocol. Adding one would make + # every existing control incomplete for the sake of one string, and there is + # nothing useful the TCK can do with a control that has not said: it cannot + # tell from the outside whether a control spoke HTTP or reached into the + # process, so the field is simply omitted. See ``report.control_api_of``. + @typing.runtime_checkable class ConnectionControl(typing.Protocol): diff --git a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/emitter.py b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/emitter.py new file mode 100644 index 00000000..808bbb1a --- /dev/null +++ b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/emitter.py @@ -0,0 +1,357 @@ +"""The pytest half of the conformance report: turning a run into the document. + +Kept apart from :mod:`report`, which knows what a report *is* and nothing about +pytest. Everything here is translation -- a pytest node into a scenario, a +:class:`pytest.TestReport` into an :class:`~.report.Outcome`, the end of a +session into a file on disk. + +The translation that matters is the one for skips. pytest reports a skip +honestly, unlike some runners, but "skipped" alone does not distinguish a +capability the provider never declared from a scenario the run had some other +reason not to execute, and the report format does. So the decision is made +against the scenario's own tags and the suite's declared capabilities rather than +against the wording of a skip message. +""" + +from __future__ import annotations + +import os +import typing +from pathlib import Path + +import pytest + +from .config import TckConfig +from .report import ( + REPORT_DIR_ENV, + Outcome, + PhaseOutcome, + ReportCollector, + ScenarioIdentity, + normalise_tags, + report_file_name, + write_report, +) + +__all__ = ["COLLECTOR_KEY", "ReportEmitter", "classify_phase", "scenario_identity"] + +COLLECTOR_KEY = pytest.StashKey[ReportCollector]() +"""Where the session's collector lives, so a fixture can reach it from a request.""" + +_EXAMPLE_PARAM = "_pytest_bdd_example" +"""The parameter pytest-bdd renders a Scenario Outline over. + +An implementation detail of pytest-bdd, named here rather than spelled inline so +that a version bump that renames it fails in one place. The alternative -- asking +the scenario template for its examples -- would have to work out which row *this* +node is, which is the question the callspec already answers. +""" + +_MAX_REASON = 500 +"""How much of a failure message the report carries. + +A reason is for a person reading a comparison page, not for debugging: whoever +ran the suite has the traceback. Whole tracebacks in a published document also +leak local paths. +""" + + +def scenario_identity(node: pytest.Item) -> ScenarioIdentity | None: + """Describe a pytest node as a Gherkin scenario, or return ``None``. + + ``__scenario__`` is what pytest-bdd hangs on the function it generates, so + its presence is also the test for "is this a TCK scenario at all" -- and it + is readable at collection, without running a single fixture, which is what + lets a scenario skipped before its first step still be accounted for. + """ + scenario = getattr(getattr(node, "function", None), "__scenario__", None) + if scenario is None: + return None + + feature = getattr(scenario, "feature", None) + tags: set[str] = set(getattr(scenario, "tags", None) or ()) + tags |= set(getattr(feature, "tags", None) or ()) + rule = getattr(scenario, "rule", None) + if rule is not None: + tags |= set(getattr(rule, "tags", None) or ()) + tags |= _examples_tags(node, scenario) + + return ScenarioIdentity( + feature=Path(str(getattr(feature, "filename", ""))).stem, + name=str(getattr(scenario, "name", "")), + example=_example_of(node), + tags=normalise_tags(tags), + ) + + +def _examples_tags(node: pytest.Item, scenario: object) -> set[str]: + """The tags of the Examples block *this row* came from. + + Gherkin allows an Examples block to carry its own tags, so two rows of one + Scenario Outline can differ in which capability gates them. Those tags are not + on the scenario, the feature or the rule, so a report built from those three + alone would show a row the capability gate skipped as carrying no capability + at all -- and it would then be classified ``not-applicable`` rather than + ``not-declared``, which is precisely the distinction Appendix F asks a report + to keep. It would also not count towards the capability rollup. + + Resolved by intersecting the tags the scenario's Examples blocks declare with + the markers pytest actually put on this node: pytest-bdd attaches an Examples + block's tags as marks on that block's parameter sets, so the intersection + names this row's blocks without having to work out which block a row came + from, and admits nothing that is not a Gherkin tag of this scenario. + + No canonical feature file uses per-Examples tags today, so this is latent -- + but it is latent in the direction of under-reporting a skip, which is the one + failure mode the format exists to rule out. + """ + declared: set[str] = set() + for examples in getattr(scenario, "examples", None) or (): + declared |= set(getattr(examples, "tags", None) or ()) + if not declared: + return set() + return declared & {marker.name for marker in node.iter_markers()} + + +def _example_of(node: pytest.Item) -> tuple[tuple[str, str], ...]: + """The Examples row this node came from, keyed by column header. + + Every row of a Scenario Outline shares one scenario name, so the row is what + tells eleven otherwise identical entries apart -- and in this suite one row + of the type-mismatch matrix genuinely differs in outcome from its ten + siblings. The row goes in its own field rather than into a mangled name + because the parameters *are* the identity and they come from the feature + file, whereas a name format would be a rule about this runner: pytest-bdd's + own id for the row above is ``boolean-flag-Integer-1``, which no other + language's runner has any reason to reproduce. + + pytest-bdd renders an outline by parametrizing the generated test over one + dict per row, keyed by the Examples column header, and pytest hangs it on the + node's callspec. A scenario that is not an outline is not parametrized and + has no callspec at all, which is why the empty tuple -- and therefore an + omitted field -- is the answer for one. + + Values are passed through as the parser produced them: Gherkin cells are + strings, and the report says what the table said rather than guessing that + ``1`` was meant as a number. + """ + params = getattr(getattr(node, "callspec", None), "params", None) + if not isinstance(params, dict): + return () + row = params.get(_EXAMPLE_PARAM) + if not isinstance(row, dict): + return () + # Column order, as the feature file wrote it, because dicts preserve + # insertion order and pytest-bdd builds this one from the header row. + return tuple((str(header), str(cell)) for header, cell in row.items()) + + +def _group_of(node: pytest.Item) -> str: + """Which module a scenario was generated into. + + pytest-bdd's ``scenarios()`` injects its tests into the module that called + it, and a module resolves one ``tck_config``, so the module is what says + which suite a scenario belongs to. Two modules sharing a ``tck_config`` from + a conftest are two groups pointing at one suite, which is exactly right. + """ + return node.nodeid.partition("::")[0] + + +class ReportEmitter: + """Collects outcomes for the session and writes one report per suite. + + A plugin object rather than module-level hook functions because + ``pytest_runtest_logreport`` is handed a report and nothing else: the state + it has to reach has to come from somewhere, and an instance is a less + surprising somewhere than a module global. + """ + + def __init__(self, config: pytest.Config) -> None: + self.collector = ReportCollector() + config.stash[COLLECTOR_KEY] = self.collector + + def pytest_collection_modifyitems(self, items: list[pytest.Item]) -> None: + """Enumerate every TCK scenario the session collected. + + At collection rather than as each runs, so that the document accounts for + scenarios that never got as far as running a fixture. + """ + for item in items: + identity = scenario_identity(item) + if identity is not None: + self.collector.collect(item.nodeid, _group_of(item), identity) + + def pytest_runtest_logreport(self, report: pytest.TestReport) -> None: + self.collector.observe(report.nodeid, _phase_outcome(report)) + + def pytest_sessionfinish(self, session: pytest.Session) -> None: + directory = os.environ.get(REPORT_DIR_ENV, "").strip() + if not directory: + return + self.write(session, Path(directory)) + + def write(self, session: pytest.Session, directory: Path) -> None: + """Write every suite's report, failing the session if one cannot be written. + + A run that asked for a report and silently did not get one is how a + publishing pipeline ends up serving a stale result forever, so both a + write failure and an incomplete document are loud and change the exit + status rather than being logged and forgotten. + """ + for problem in self.collector.resolve(classify_phase): + self._fail(session, f"provider-tck: {problem}") + + written: dict[str, str] = {} + for suite in self.collector.suites: + name = suite.config.name + file_name = report_file_name(name) + if written.get(file_name, name) != name: + self._fail( + session, + f"provider-tck: suites {written[file_name]!r} and {name!r} both " + f"write {file_name}; give them names that do not collide", + ) + continue + written[file_name] = name + + try: + path = write_report(directory, name, suite.build()) + except OSError as error: + self._fail( + session, + f"provider-tck [{name}]: could not write the conformance report " + f"to {directory}: {error}", + ) + continue + counts = ", ".join( + f"{count} {outcome}" + for outcome, count in sorted(suite.counts().items()) + ) + self._say( + session, f"provider-tck [{name}]: report written to {path} ({counts})" + ) + + def _say(self, session: pytest.Session, message: str) -> None: + reporter = session.config.pluginmanager.get_plugin("terminalreporter") + if reporter is not None: + reporter.write_line(message) + + def _fail(self, session: pytest.Session, message: str) -> None: + self._say(session, message) + session.exitstatus = pytest.ExitCode.INTERNAL_ERROR + + +def _phase_outcome(report: pytest.TestReport) -> PhaseOutcome: + """Reduce a pytest phase report to what the conformance report needs.""" + xfail_reason: str | None = getattr(report, "wasxfail", None) + message = _skip_reason(report) if report.skipped else _failure_reason(report) + return PhaseOutcome( + when=report.when or "", + outcome=report.outcome, + xfail_reason=xfail_reason, + message=message, + duration=report.duration, + ) + + +def classify_phase( + phase: PhaseOutcome, identity: ScenarioIdentity, config: TckConfig +) -> tuple[Outcome, str] | None: + """Map one phase onto an outcome, or onto nothing. + + Nothing is the answer for a setup or teardown that simply worked: it says + nothing about the scenario, and letting it speak would overwrite what the + call phase already established. + """ + if phase.outcome == "skipped" and phase.xfail_reason is not None: + # An expected failure is still a failure. The provider did not satisfy + # the scenario, and a report calling it anything else would hide exactly + # the deviation the marker was added to keep visible. + return Outcome.FAILED, _reason(f"expected failure: {phase.xfail_reason}") + if phase.outcome == "failed": + return Outcome.FAILED, phase.message or "failed" + if phase.outcome == "skipped": + return _skipped(phase, identity, config) + if phase.when == "call": + return Outcome.PASSED, "" + return None + + +def _skipped( + phase: PhaseOutcome, identity: ScenarioIdentity, config: TckConfig +) -> tuple[Outcome, str]: + """Tell a capability skip apart from every other kind. + + Decided from the scenario's tags and the suite's declared capabilities rather + than from the skip message, because the message is prose and the distinction + is not. Anything else that skipped a scenario -- a marker an adopter applied, + a step calling ``pytest.skip`` -- is reported as not applicable: it did not + run, and not because a capability was left undeclared. + """ + undeclared = [ + capability.tag + for capability in identity.capabilities() + if not config.declares(capability) + ] + if undeclared: + return Outcome.NOT_DECLARED, phase.message or ( + f"provider does not declare {' '.join(undeclared)}" + ) + return Outcome.NOT_APPLICABLE, phase.message or "skipped" + + +def _skip_reason(report: pytest.TestReport) -> str: + longrepr = report.longrepr + if isinstance(longrepr, tuple) and len(longrepr) == 3: + return _reason(str(longrepr[2]).removeprefix("Skipped: ")) + return _reason(str(longrepr)) if longrepr else "" + + +def _failure_reason(report: pytest.TestReport) -> str: + message = getattr(getattr(report.longrepr, "reprcrash", None), "message", "") + if not message: + message = str(report.longrepr) if report.longrepr else "" + return _reason(message) + + +def _reason(message: str) -> str: + collapsed = " ".join(message.split()) + if len(collapsed) <= _MAX_REASON: + return collapsed + return collapsed[: _MAX_REASON - 1].rstrip() + "…" + + +def observe_provider_name( + config: pytest.Config, tck_config: TckConfig, provider_name: str | None +) -> None: + """Record what the provider called itself, for the suite the run is in. + + The provider's own metadata name rather than the suite name, because the two + answer different questions: the suite name is chosen to read well in a + failure message, which makes it the configuration and it is reported as one. + """ + collector: ReportCollector | None = config.stash.get(COLLECTOR_KEY, None) + if collector is not None and provider_name: + collector.suite_for(tck_config).observe_provider_name(provider_name) + + +def bind_scenario(request: pytest.FixtureRequest) -> None: + """Tell the collector which suite this scenario's module is testing. + + Called from an autouse fixture that the capability gate depends on, so that a + scenario the gate stops has still contributed its suite. Only one scenario of + a module has to get this far, but the gate skips whole capabilities at a + time, and a module all of whose scenarios were skipped would otherwise have + no report to be written to. + """ + collector: ReportCollector | None = request.config.stash.get(COLLECTOR_KEY, None) + if collector is None or scenario_identity(request.node) is None: + # Checked before asking for the config so that a test which is not a TCK + # scenario instantiates nothing, which is the same bargain the capability + # gate makes. + return + try: + tck_config = typing.cast(TckConfig, request.getfixturevalue("tck_config")) + except pytest.FixtureLookupError: + return + collector.bind(request.node.nodeid, tck_config) diff --git a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/inprocess.py b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/inprocess.py index 1d69254c..11586e0f 100644 --- a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/inprocess.py +++ b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/inprocess.py @@ -63,6 +63,16 @@ def __init__(self) -> None: def description(self) -> str: return "in-process control of an in-memory provider" + @property + def control_api(self) -> str: + """Report how this backend was driven, for the conformance report. + + ``in-process`` is the narrow allowance for providers with no backend, + which is exactly what this control exists for. A provider that does have + a backend and reports this is claiming something it should not. + """ + return "in-process" + def new_provider(self) -> FeatureProvider: """Create the provider for the scenario about to run, at the baseline. diff --git a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/plugin.py b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/plugin.py index b8b1a73b..29aeb0d5 100644 --- a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/plugin.py +++ b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/plugin.py @@ -17,6 +17,7 @@ from .capability import Capability, capability_for_marker from .config import TckConfig +from .emitter import ReportEmitter, bind_scenario, observe_provider_name from .state import TckState # The step modules are registered as plugins in their own right, not merely @@ -32,22 +33,33 @@ def pytest_configure(config: pytest.Config) -> None: - """Register the capability tags as markers. + """Register the capability tags as markers, and the report emitter. pytest-bdd turns every Gherkin tag into a marker with ``getattr(pytest.mark, tag)`` without registering it, which raises ``PytestUnknownMarkWarning`` for each one -- noise at best, and a hard failure in a project configured with ``-W error``. + + The emitter is registered unconditionally even though it writes nothing + unless :data:`~.report.REPORT_DIR_ENV` is set. Accumulating the outcomes + costs a dictionary entry per scenario, and deciding at the end of the session + rather than at the start is one fewer way for a run to discover too late that + it was not recording. """ for capability in Capability: config.addinivalue_line( "markers", f"{capability.value}: OpenFeature provider TCK capability {capability.tag}", ) + config.pluginmanager.register( + ReportEmitter(config), "openfeature-provider-tck-report" + ) @pytest.fixture -def tck_state(tck_config: TckConfig) -> typing.Iterator[TckState]: +def tck_state( + request: pytest.FixtureRequest, tck_config: TckConfig +) -> typing.Iterator[TckState]: """Per-scenario state, carried between step definitions.""" # Resetting here rather than in an autouse fixture ties the reset to the # scenarios that actually use the TCK, and guarantees it happens after the @@ -56,11 +68,22 @@ def tck_state(tck_config: TckConfig) -> typing.Iterator[TckState]: tck_config.control.prepare_scenario() state = TckState(config=tck_config) yield state + # The provider is identified in the report by what it called itself, and the + # only thing that ever holds an instance is the scenario that made one. + observe_provider_name(request.config, tck_config, state.provider_name) state.teardown() @pytest.fixture(autouse=True) -def _tck_capability_gate(request: pytest.FixtureRequest) -> None: +def _tck_report_binding(request: pytest.FixtureRequest) -> None: + """Attribute this scenario to its suite before anything can skip it.""" + bind_scenario(request) + + +@pytest.fixture(autouse=True) +def _tck_capability_gate( + request: pytest.FixtureRequest, _tck_report_binding: None +) -> None: """Skip a scenario whose capability the provider did not declare. ``pytest.skip`` here reports the scenario as skipped **with the reason**, @@ -75,6 +98,11 @@ def _tck_capability_gate(request: pytest.FixtureRequest) -> None: Checking markers first also means the gate costs nothing, and instantiates nothing, for tests that are not TCK scenarios. + + ``_tck_report_binding`` is requested rather than left to autouse ordering so + that the scenario has reached its suite before this fixture can skip it. A + scenario skipped here is exactly the one the conformance report must account + for, and one that never reached a suite could not be reported at all. """ gated = [ capability diff --git a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/report.py b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/report.py new file mode 100644 index 00000000..ab3625b7 --- /dev/null +++ b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/report.py @@ -0,0 +1,573 @@ +"""The machine-readable conformance report: what a run of the suite claims. + +A run of the suite produces a pass or a fail on a terminal, which is enough for +the person who started it and useless to anyone else. The report is the same run +written down in a form something other than a human can read -- a comparison +page, an aggregator, a release gate -- against a schema owned by the +specification rather than by this package, so that four languages emit the same +document. + +The load-bearing part is the per-scenario list. Appendix F requires that a +scenario skipped for an undeclared capability is reported as skipped *with the +reason* and never as passed, and a summary line cannot be checked against that +rule by anything downstream. Recording every scenario's outcome individually +makes the rule checkable by the consumer instead of dependent on each runner's +summary being trustworthy -- and the outcomes are required to be complete, +because a report that silently omitted what it skipped would satisfy the letter +of the rule while still misleading its reader. + +See https://github.com/open-feature/spec/issues/424 for the format and +``specification/assets/provider-tck/report/`` for the schema. +""" + +from __future__ import annotations + +import importlib.metadata +import importlib.resources +import json +import re +import typing +from dataclasses import dataclass, field +from enum import Enum +from pathlib import Path + +from .capability import Capability, capability_for_tag +from .config import TckConfig + +__all__ = [ + "REPORT_DIR_ENV", + "SCHEMA_VERSION", + "Outcome", + "PhaseOutcome", + "ReportCollector", + "ScenarioIdentity", + "ScenarioRecord", + "SuiteReport", + "report_file_name", +] + +REPORT_DIR_ENV = "PROVIDER_TCK_REPORT_DIR" +"""Names the directory a conformance report is written to. + +An environment variable rather than a :class:`~.config.TckConfig` field, so that +emitting a report is a property of the *run* and not of the code: CI sets it, a +developer running the suite locally does not, and no adopter changes a line to +publish one. Each suite writes ``/.json``, so several suites in one +pytest session -- flagd's RPC and in-process resolvers, say -- each produce their +own file without colliding. + +Unset means no report, which is the default and is not an error. +""" + +SCHEMA_VERSION = "1" +"""The major version of the report schema this emitter produces.""" + +TCK_IMPLEMENTATION = "python-sdk-contrib/tools/openfeature-provider-tck" +"""Which TCK implementation produced the report, as the schema spells it.""" + +PROVIDER_LANGUAGE = "python" + +SDK_DISTRIBUTION = "openfeature-sdk" +TCK_DISTRIBUTION = "openfeature-provider-tck" + +UNKNOWN = "unknown" +"""Stands in for an identity that could not be read. + +Seven characters, which is the schema's minimum for ``tck.specRevision``, so a +build that could not reach git still emits a document that validates and says +plainly that it does not know rather than inventing a commit. +""" + +_PACKAGE = "openfeature.contrib.tools.provider_tck" + +_REVISION_FILE = "spec_revision.json" +"""Written at build time from the spec submodule; see ``hatch_build_sync.py``. + +Read from a data file rather than the submodule because the submodule is not in +the published wheel: an adopter installing this package has no ``spec/`` +directory to interrogate, and the revision the assets came from is exactly what +the report has to name. +""" + +_TAG_PATTERN = re.compile(r"^[a-z0-9-]+$") +"""What the schema accepts as a tag, minus the leading at-sign. + +Tags that do not match are dropped rather than emitted, because an invalid +document helps nobody; the canonical feature files carry none, so this only bites +a feature file that has been forked, which is itself worth noticing. +""" + +_UNSAFE_IN_FILENAME = re.compile(r"[^A-Za-z0-9._-]") + + +class Outcome(str, Enum): + """The result of one scenario, or of one capability. + + Four rather than two, because "did not run" is not one thing. A capability + the provider chose not to declare is a different statement from one the + language makes impossible -- ``@strict-numeric-typing`` cannot hold in a + language with no integer type -- and reporting both as not declared would + show a whole language as missing something none of its providers can have. + """ + + PASSED = "passed" + FAILED = "failed" + NOT_DECLARED = "not-declared" + NOT_APPLICABLE = "not-applicable" + + +@dataclass(frozen=True) +class ScenarioIdentity: + """What a scenario is, independent of how it turned out. + + Established at collection, from the pytest-bdd node alone, so that a scenario + skipped before a single step ran is identified exactly as fully as one that + passed. That is what lets the report account for every scenario rather than + only for the ones that got far enough to be interesting. + """ + + feature: str + name: str + tags: tuple[str, ...] + example: tuple[tuple[str, str], ...] = () + """The Examples row, as header/cell pairs, for a scenario from an outline. + + Pairs rather than a mapping so that this stays hashable and ordered: the + order is the feature file's column order, and the report carries it through + rather than imposing one of its own. + """ + + def capabilities(self) -> tuple[Capability, ...]: + """The capabilities this scenario's tags gate it behind.""" + gated = (capability_for_tag(tag) for tag in self.tags) + return tuple(capability for capability in gated if capability is not None) + + +@dataclass +class ScenarioRecord: + """One scenario's outcome, as the report will carry it.""" + + feature: str + """The feature file without its extension, e.g. ``errors``.""" + + name: str + tags: tuple[str, ...] + outcome: Outcome + example: tuple[tuple[str, str], ...] = () + """The Examples row this entry came from; empty for a scenario that is not + an outline, in which case the field is omitted rather than emitted empty.""" + + reason: str = "" + duration_ms: float = 0.0 + + def as_json(self) -> dict[str, typing.Any]: + document: dict[str, typing.Any] = { + "feature": self.feature, + "name": self.name, + "outcome": self.outcome.value, + } + if self.example: + document["example"] = dict(self.example) + if self.tags: + document["tags"] = list(self.tags) + if self.reason: + document["reason"] = self.reason + if self.duration_ms: + document["durationMs"] = round(self.duration_ms, 3) + return document + + +@dataclass +class SuiteReport: + """What one suite -- one :class:`~.config.TckConfig` -- accumulates as it runs. + + Records are keyed by pytest node id rather than appended to a list, which is + what makes "every scenario appears exactly once" a property of the structure + instead of a promise made by the code that fills it. A scenario reports + through several phases (setup, call, teardown) and each of them finds the + same entry. + """ + + config: TckConfig + provider_name: str | None = None + records: dict[str, ScenarioRecord] = field(default_factory=dict) + durations: dict[str, float] = field(default_factory=dict) + + def observe_provider_name(self, name: str) -> None: + """Remember what the provider called itself through its own metadata. + + Last one wins, and they should all agree: a suite tests one provider. + """ + if name: + self.provider_name = name + + def add_duration(self, node_id: str, seconds: float) -> None: + """Add one phase's time to a scenario's total. + + Kept apart from the record rather than added to it, because a scenario's + first phase can take time before anything has decided its outcome, and + time spent on a scenario that ended up skipped is still time. + """ + self.durations[node_id] = self.durations.get(node_id, 0.0) + seconds * 1000.0 + + def set_outcome( + self, + node_id: str, + identity: ScenarioIdentity, + outcome: Outcome, + reason: str = "", + ) -> None: + """Record, or revise, one scenario's outcome. + + A failure is never revised away. A scenario whose steps passed and whose + teardown then blew up is a failed scenario, and the phase that reports + last must not be the one that decides. + """ + record = self.records.get(node_id) + if record is None: + self.records[node_id] = ScenarioRecord( + feature=identity.feature, + name=identity.name, + tags=identity.tags, + outcome=outcome, + example=identity.example, + reason=reason, + ) + return + if record.outcome is Outcome.FAILED: + return + record.outcome = outcome + record.reason = reason or record.reason + + @property + def sorted_records(self) -> list[ScenarioRecord]: + for node_id, record in self.records.items(): + record.duration_ms = self.durations.get(node_id, 0.0) + # Sorted by the whole identity, example included, so that two rows of one + # outline come out in a stable order rather than in whichever order the + # dictionary happened to be filled. + return sorted( + self.records.values(), key=lambda r: (r.feature, r.name, r.example) + ) + + def counts(self) -> dict[str, int]: + """Outcome tallies, for a log line and for the tests that check them.""" + tally: dict[str, int] = {} + for record in self.records.values(): + tally[record.outcome.value] = tally.get(record.outcome.value, 0) + 1 + return tally + + def build(self) -> dict[str, typing.Any]: + """Assemble the report document.""" + records = self.sorted_records + spec_revision, assets_tree = spec_identity() + + tck: dict[str, typing.Any] = { + "implementation": TCK_IMPLEMENTATION, + "version": distribution_version(TCK_DISTRIBUTION), + "specRevision": spec_revision, + } + if assets_tree: + tck["assetsTree"] = assets_tree + + document: dict[str, typing.Any] = { + "schemaVersion": SCHEMA_VERSION, + "provider": { + # What the provider calls itself, not the suite name: the suite + # name is chosen to read well in a failure message -- "flagd-rpc" + # -- which makes it the configuration, and it is reported as one. + # A provider with two materially different modes therefore + # produces two reports that are not interchangeable. + "name": self.provider_name or self.config.name, + "language": PROVIDER_LANGUAGE, + "configuration": self.config.name, + }, + "sdk": { + "name": SDK_DISTRIBUTION, + "version": distribution_version(SDK_DISTRIBUTION), + }, + "tck": tck, + "capabilities": self._capabilities(records), + "scenarios": [record.as_json() for record in records], + } + + backend = self._backend() + if backend: + document["backend"] = backend + return document + + def _backend(self) -> dict[str, typing.Any]: + backend: dict[str, typing.Any] = {} + description = getattr(self.config.control, "description", "") + if isinstance(description, str) and description: + backend["description"] = description + control_api = control_api_of(self.config.control) + if control_api: + backend["controlApi"] = control_api + return backend + + def _capabilities( + self, records: list[ScenarioRecord] + ) -> dict[str, dict[str, typing.Any]]: + """Roll the per-scenario outcomes up to one verdict per capability. + + A capability is only reported as passed when everything gating on it + actually passed, and only reported as not declared when the provider did + not declare it -- in which case the reason says so, because "this + provider does not support configuration-change events" is exactly what + someone comparing providers came to find out. + + A capability the provider declared and *no scenario carries* is omitted + rather than reported. ``@targeting`` is reserved: it exists in the + vocabulary but nothing tests it, because asserting that an evaluation + context reached the backend needs an echo operation the control API does + not have. Reporting it as passed would be a green result for a claim + nothing examined -- the vacuous pass the capability vocabulary exists to + eliminate, arriving through the report rather than through the suite. + Omitting beats inventing a fifth outcome: the four the schema allows are + about what the provider did, and "the suite does not test this" is a fact + about the suite. + """ + # Counted rather than flagged, so that a failure can say how much of what + # failed, and so that "no scenario exercises this at all" is a case the + # rollup can see rather than one it silently reads as success. + exercised: dict[Capability, int] = {} + failed: dict[Capability, int] = {} + for record in records: + for tag in record.tags: + capability = capability_for_tag(tag) + if capability is None: + continue + exercised[capability] = exercised.get(capability, 0) + 1 + if record.outcome is Outcome.FAILED: + failed[capability] = failed.get(capability, 0) + 1 + + capabilities: dict[str, dict[str, typing.Any]] = {} + for capability in Capability: + if not self.config.declares(capability): + capabilities[capability.tag] = { + "state": Outcome.NOT_DECLARED.value, + "reason": ( + f"not declared by this provider's configuration; the " + f"{capability.tag} scenarios were skipped and did not " + f"contribute to this result" + ), + } + elif not exercised.get(capability): + continue + elif failed.get(capability): + capabilities[capability.tag] = { + "state": Outcome.FAILED.value, + "reason": ( + f"{failed[capability]} of {exercised[capability]} scenarios " + f"carrying {capability.tag} failed; the per-scenario results " + f"say which, and why" + ), + } + else: + capabilities[capability.tag] = {"state": Outcome.PASSED.value} + return capabilities + + +@dataclass(frozen=True) +class PhaseOutcome: + """One pytest phase report, reduced to what the conformance report needs. + + Reduced rather than kept, because a :class:`pytest.TestReport` holds a + formatted traceback and holding a session's worth of them to classify at the + end would be a memory leak with a nice name. + """ + + when: str + """``setup``, ``call`` or ``teardown``.""" + + outcome: str + """``passed``, ``failed`` or ``skipped``, as pytest decided.""" + + xfail_reason: str | None = None + """Set when pytest marked this an expected failure.""" + + message: str = "" + """The skip reason, or the failure's headline, already trimmed.""" + + duration: float = 0.0 + + +Classifier = typing.Callable[ + [PhaseOutcome, ScenarioIdentity, TckConfig], "tuple[Outcome, str] | None" +] + + +class ReportCollector: + """Session-wide accumulator: which scenario belongs to which suite, and how it went. + + One pytest session can run several suites -- the TCK's own tests run two, and + a provider with more than one resolver runs one per resolver -- so outcomes + are attributed to a suite rather than to the session, and each suite writes + its own file. + + Scenarios are enumerated at collection and resolved into records only at the + end of the session. The order matters. A scenario skipped by a marker never + runs a fixture, so a design that learned of a scenario when its fixtures ran + would leave it out of the document entirely -- and a report that silently + omits what it skipped satisfies "a skip is never reported as passed" while + still misleading the person reading it. + """ + + def __init__(self) -> None: + # Suites are keyed by the identity of their TckConfig, so two suites that + # happen to share a name stay distinct here; that collision is caught + # where it actually bites, when their file names turn out to be equal. + self._suites: dict[int, SuiteReport] = {} + self._suite_by_group: dict[str, SuiteReport] = {} + self._collected: dict[str, tuple[str, ScenarioIdentity]] = {} + self._phases: dict[str, list[PhaseOutcome]] = {} + + def collect(self, node_id: str, group: str, identity: ScenarioIdentity) -> None: + """Note that this scenario exists, and which group of tests it came from. + + The group is the module the scenario was generated into. pytest-bdd's + ``scenarios()`` injects its tests into the module that called it, and a + module resolves one ``tck_config``, so the module is what says which + suite a scenario belongs to -- and it says so without running anything. + """ + self._collected[node_id] = (group, identity) + + def observe(self, node_id: str, phase: PhaseOutcome) -> None: + """Record one phase's result for a scenario, if it is one of ours.""" + if node_id in self._collected: + self._phases.setdefault(node_id, []).append(phase) + + def bind(self, node_id: str, config: TckConfig) -> None: + """Learn which suite a group of scenarios is testing. + + Called from a fixture, because the ``TckConfig`` is a fixture value and + there is no way to know it without asking for it. Only one scenario of a + group has to get this far for the whole group to be attributed. + """ + entry = self._collected.get(node_id) + if entry is not None: + self._suite_by_group[entry[0]] = self.suite_for(config) + + def suite_for(self, config: TckConfig) -> SuiteReport: + return self._suites.setdefault(id(config), SuiteReport(config=config)) + + @property + def suites(self) -> list[SuiteReport]: + return list(self._suites.values()) + + def resolve(self, classify: Classifier) -> list[str]: + """Turn the collected phases into records, and report what could not be. + + Returns the problems, one string each, and they are meant to be shouted + about rather than logged: a scenario that ran but is missing from the + document is the one failure mode this format exists to rule out. + """ + problems: list[str] = [] + for node_id, (group, identity) in sorted(self._collected.items()): + suite = self._suite_by_group.get(group) + if suite is None: + problems.append( + f"{node_id}: no TckConfig was resolved for {group}, so its " + f"outcome belongs to no suite and is missing from every report" + ) + continue + phases = self._phases.get(node_id) + if not phases: + problems.append( + f"{node_id}: was collected but never ran, so the report for " + f"{suite.config.name!r} does not account for it" + ) + continue + for phase in phases: + classified = classify(phase, identity, suite.config) + if classified is not None: + outcome, reason = classified + suite.set_outcome(node_id, identity, outcome, reason) + suite.add_duration(node_id, phase.duration) + return problems + + +def control_api_of(control: object) -> str: + """Report how the backend was driven, if the control says. + + Read off an optional attribute rather than added to the + :class:`~.control.BackendControl` protocol, because a protocol member would + make every existing control incomplete for the sake of one string. A control + that does not offer it simply omits the field, which is the honest answer: + the TCK cannot infer from the outside whether a control spoke the normative + HTTP API or reached into the process. + """ + value = getattr(control, "control_api", None) + if isinstance(value, str) and value in {"http", "in-process"}: + return value + return "" + + +def normalise_tags(tags: typing.Iterable[str]) -> tuple[str, ...]: + """Turn Gherkin tags as pytest-bdd holds them into the form the schema wants. + + pytest-bdd strips the leading at-sign; the schema requires it back. + """ + return tuple(sorted(f"@{tag}" for tag in tags if _TAG_PATTERN.match(tag))) + + +def report_file_name(suite_name: str) -> str: + """Turn a suite name into a file name. + + Suite names are chosen to read well in a failure message rather than to be + path-safe, so anything not obviously safe becomes a hyphen. Without this a + suite named ``flagd/rpc`` would quietly write outside the directory it was + given. + """ + cleaned = _UNSAFE_IN_FILENAME.sub("-", suite_name).strip("-.") + return f"{cleaned or 'report'}.json" + + +def write_report( + directory: Path, suite_name: str, document: dict[str, typing.Any] +) -> Path: + """Write one report, returning where it went.""" + directory.mkdir(parents=True, exist_ok=True) + path = directory / report_file_name(suite_name) + path.write_text( + json.dumps(document, indent=2, ensure_ascii=False) + "\n", encoding="utf-8" + ) + return path + + +def distribution_version(distribution: str) -> str: + """Read an installed distribution's version. + + Read rather than declared, because a declared version is a second place to + be wrong: the report would go on claiming 0.8.2 after a dependency bump moved + the actual code underneath it. + """ + try: + return importlib.metadata.version(distribution) + except importlib.metadata.PackageNotFoundError: + return UNKNOWN + + +def spec_identity() -> tuple[str, str]: + """Return the spec commit and asset tree these feature files came from. + + Captured at build time rather than read here, because the submodule that + holds the answer is not in the wheel. A build that could not reach git says + so with :data:`UNKNOWN` instead of inventing a commit, and an installation + old enough to predate the generated file degrades the same way rather than + failing to emit a report at all. + """ + reference = importlib.resources.files(_PACKAGE) / _REVISION_FILE + try: + data = json.loads(reference.read_text(encoding="utf-8")) + except (OSError, ValueError): + return UNKNOWN, "" + if not isinstance(data, dict): + return UNKNOWN, "" + revision = data.get("specRevision") + tree = data.get("assetsTree") + return ( + revision if isinstance(revision, str) and len(revision) >= 7 else UNKNOWN, + tree if isinstance(tree, str) and re.fullmatch(r"[0-9a-f]{40}", tree) else "", + ) diff --git a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/state.py b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/state.py index 71ea4150..1b1faa86 100644 --- a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/state.py +++ b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/state.py @@ -93,6 +93,13 @@ class TckState: config: TckConfig client: OpenFeatureClient | None = None + provider_name: str | None = None + """What the provider called itself through its own metadata. + + Observed rather than configured, because it is what the conformance report + identifies the provider by: ``TckConfig.name`` is chosen to read well in a + failure message, which makes it the *configuration* rather than the provider. + """ flag_key: str | None = None flag_type: FlagType | None = None default_value: typing.Any = None diff --git a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/steps/provider_steps.py b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/steps/provider_steps.py index bca37fae..59520879 100644 --- a/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/steps/provider_steps.py +++ b/tools/openfeature-provider-tck/src/openfeature/contrib/tools/provider_tck/steps/provider_steps.py @@ -30,6 +30,7 @@ def a_stable_provider(tck_state: TckState) -> None: if provider is None: msg = "TckConfig.new_provider returned None" raise AssertionError(msg) + _observe_metadata_name(tck_state, provider) try: _set_provider_within(provider, config.domain, config.ready_timeout) @@ -79,6 +80,7 @@ def an_unavailable_provider(tck_state: TckState) -> None: if provider is None: msg = "TckConfig.new_unavailable_provider returned None" raise AssertionError(msg) + _observe_metadata_name(tck_state, provider) # A raising initialize is already converted to PROVIDER_ERROR by the SDK's # registry, so this is belt and braces: a provider that raises anyway must @@ -90,6 +92,21 @@ def an_unavailable_provider(tck_state: TckState) -> None: tck_state.client = api.get_client(config.domain) +def _observe_metadata_name(tck_state: TckState, provider: FeatureProvider) -> None: + """Note what the provider calls itself, for the conformance report. + + Before registration rather than after, so that a provider which fails to + initialise -- the ``@unavailable`` case, and any genuine failure -- is still + identified in the report by its own name. Metadata is a pure accessor by + contract, but a provider that raises from it must not take the scenario down + with it: the name is for a report, and no scenario asserts on it. + """ + with contextlib.suppress(Exception): + name = provider.get_metadata().name + if name: + tck_state.provider_name = name + + def _set_provider_within( provider: FeatureProvider, domain: str, timeout: float ) -> None: diff --git a/tools/openfeature-provider-tck/tests/test_report.py b/tools/openfeature-provider-tck/tests/test_report.py new file mode 100644 index 00000000..c27aaa68 --- /dev/null +++ b/tools/openfeature-provider-tck/tests/test_report.py @@ -0,0 +1,752 @@ +"""What the conformance report must never do. + +The report exists because a runner's summary cannot be checked by anything +downstream. So the tests that matter here are not about JSON shape; they are +about the two properties a consumer is entitled to assume, neither of which is +guaranteed by the code that happens to assemble the document: + +* a scenario skipped for an undeclared capability is never reported as passed, + and carries the reason it was skipped; +* every scenario the run collected appears exactly once, which is what makes the + first property checkable rather than merely asserted -- a document that quietly + dropped what it skipped would satisfy the letter of it and still mislead. + +Both are checked against a real pytest session in a subprocess, because both are +properties of how the suite runs rather than of how the document is assembled. +That session is also the only place all four outcomes occur together, and the +only place the document can be seen to disagree with the runner's summary -- +which it does, deliberately, for a known deviation. +""" + +from __future__ import annotations + +import collections +import dataclasses +import json +import os +import subprocess +import sys +import typing +from pathlib import Path + +import pytest + +from openfeature.contrib.tools.provider_tck import ( + Capability, + TckConfig, + features_path, +) +from openfeature.contrib.tools.provider_tck.emitter import classify_phase +from openfeature.contrib.tools.provider_tck.report import ( + REPORT_DIR_ENV, + Outcome, + PhaseOutcome, + ScenarioIdentity, + SuiteReport, + control_api_of, + normalise_tags, + report_file_name, + spec_identity, +) + +OUTCOMES = {outcome.value for outcome in Outcome} + +# The generated suite's name is deliberately not path-safe. +SUITE_NAME = "report/fixture" +SUITE_FILE = "report-fixture.json" + +UNKNOWN_KEY_SCENARIO = "An unknown flag key returns the code default" + +# The type-mismatch matrix: eleven Examples rows under one scenario name, one of +# which the Python SDK fails. It is the case the example field exists for. +MISMATCH_SCENARIO = "Requesting the wrong type returns the code default" + +# The row that fails, spelled as the feature file spells it -- strings, because +# Gherkin has no types and "1" is not 1. +DEVIATING_ROW = {"key": "boolean-flag", "requested": "Integer", "default": "1"} + +_SUITE_MODULE = '''\ +"""A one-fixture adoption, generated so the report can be checked end to end.""" + +import pytest +from pytest_bdd import scenarios + +from openfeature.contrib.tools.provider_tck import ( + Capability, + InProcessControl, + TckConfig, + features_path, +) + + +@pytest.fixture(scope="session") +def tck_config(): + control = InProcessControl() + return TckConfig( + name="{name}", + control=control, + new_provider=control.new_provider, + capabilities={capabilities}, + ) + + +scenarios(features_path()) +''' + +CAPABILITIES = ( + "{Capability.EVENTS, Capability.OBJECT, Capability.STRICT_NUMERIC_TYPING}" +) +"""What the main generated suite declares: enough to produce all four outcomes.""" + +# One scenario skipped outright and one known deviation marked xfail, so the run +# produces all four outcomes and finishes green while the document does not. +_CONFTEST_MODULE = """\ +import pytest + +SKIPPED = "test_an_unknown_flag_key_returns_the_code_default" +DEVIATION = "test_requesting_the_wrong_type_returns_the_code_default[boolean-flag-Integer-1]" + + +def pytest_collection_modifyitems(items): + for item in items: + if item.name == SKIPPED: + item.add_marker(pytest.mark.skip(reason="deliberately not run here")) + elif item.name == DEVIATION: + item.add_marker(pytest.mark.xfail(reason="python-sdk#619", strict=True)) +""" + + +# A Scenario Outline whose second Examples block carries a tag of its own, which +# no canonical feature file does yet. Written here so that the one case where two +# rows of an outline are gated differently is covered. +_TAGGED_FEATURE = """\ +Feature: Per-Examples tags + + Background: + Given a stable provider + + Scenario Outline: Requesting the wrong type returns the code default + Given a -flag with key "" and a default value "" + When the flag was evaluated with details + Then the resolved details value should be "" + And the reason should be "ERROR" + And the error-code should be "TYPE_MISMATCH" + And no exception should have been thrown + + Examples: ungated + | key | requested | default | + | string-flag | Boolean | false | + | string-flag | Integer | 1 | + + @object + Examples: gated behind a capability this suite does not declare + | key | requested | default | + | string-flag | Float | 0.1 | +""" + +_TAGGED_SUITE = '''\ +"""A suite over the feature file beside it, which tags one Examples block.""" + +import pathlib + +import pytest +from pytest_bdd import scenarios + +from openfeature.contrib.tools.provider_tck import ( + Capability, + InProcessControl, + TckConfig, +) + + +@pytest.fixture(scope="session") +def tck_config(): + control = InProcessControl() + return TckConfig( + name="per-examples", + control=control, + new_provider=control.new_provider, + capabilities={Capability.EVENTS}, + ) + + +scenarios(str(pathlib.Path(__file__).parent)) +''' + + +@dataclasses.dataclass(frozen=True) +class Run: + """One subprocess run of the generated suite.""" + + directory: Path + result: subprocess.CompletedProcess[str] + document: dict[str, typing.Any] + + @property + def scenarios(self) -> list[dict[str, typing.Any]]: + scenarios: list[dict[str, typing.Any]] = self.document["scenarios"] + return scenarios + + +# -- helpers ----------------------------------------------------------------- + + +class _StubControl: + """A control that says nothing about how it drove the backend.""" + + @property + def description(self) -> str: + return "a stub" + + def prepare_scenario(self) -> None: + return None + + def change_flag(self) -> None: + return None + + +class _HttpControl(_StubControl): + @property + def control_api(self) -> str: + return "http" + + +def _config(**overrides: typing.Any) -> TckConfig: + settings: dict[str, typing.Any] = { + "name": "stub", + "control": _StubControl(), + "new_provider": lambda: None, + "capabilities": {Capability.EVENTS}, + } + settings.update(overrides) + return TckConfig(**settings) + + +def _identity(*tags: str) -> ScenarioIdentity: + return ScenarioIdentity(feature="events", name="a scenario", tags=tags) + + +def _identity_of(scenario: dict[str, typing.Any]) -> tuple[typing.Any, ...]: + """What identifies one entry: feature, name and the Examples row together.""" + example = scenario.get("example") or {} + return (scenario["feature"], scenario["name"], tuple(sorted(example.items()))) + + +def _examples_from_the_feature_file(feature: str, outline: str) -> list[dict[str, str]]: + """Read an outline's Examples tables straight out of the Gherkin. + + Hand-read rather than taken from pytest-bdd's parser, because the parser is + what produced the values under test: asking it what it should have said would + check nothing. It is a small reader for a small shape -- the tables in these + files are plain pipe-delimited rows -- and it exists so that "the report says + what the table said" is checked against the table. + """ + source = Path(features_path()) / f"{feature}.feature" + lines = source.read_text(encoding="utf-8").splitlines() + rows: list[dict[str, str]] = [] + headers: list[str] = [] + inside = False + + for line in lines: + stripped = line.strip() + if stripped.startswith(("Scenario:", "Scenario Outline:")): + inside = stripped.split(":", 1)[1].strip() == outline + headers = [] + elif not inside: + continue + elif stripped.startswith("Examples"): + headers = [] + elif stripped.startswith("|"): + cells = [cell.strip() for cell in stripped.strip("|").split("|")] + if headers: + rows.append(dict(zip(headers, cells, strict=True))) + else: + headers = cells + + assert rows, f"no Examples rows found for {outline!r} in {feature}.feature" + return rows + + +def _phase(outcome: str, when: str = "call", **extra: typing.Any) -> PhaseOutcome: + return PhaseOutcome(when=when, outcome=outcome, **extra) + + +def _pytest( + *arguments: str, report_dir: Path | None = None +) -> subprocess.CompletedProcess[str]: + environment = dict(os.environ) + environment.pop(REPORT_DIR_ENV, None) + if report_dir is not None: + environment[REPORT_DIR_ENV] = str(report_dir) + return subprocess.run( # noqa: S603 + [sys.executable, "-m", "pytest", "-q", "-p", "no:cacheprovider", *arguments], + capture_output=True, + text=True, + env=environment, + check=False, + ) + + +def _write_suite( + directory: Path, + name: str = SUITE_NAME, + capabilities: str = CAPABILITIES, + deviations: bool = True, +) -> Path: + directory.mkdir(parents=True, exist_ok=True) + (directory / "test_suite.py").write_text( + _SUITE_MODULE.format(name=name, capabilities=capabilities), encoding="utf-8" + ) + if deviations: + (directory / "conftest.py").write_text(_CONFTEST_MODULE, encoding="utf-8") + return directory + + +def _run_suite( + tmp_path_factory: pytest.TempPathFactory, + file_name: str = SUITE_FILE, + **suite: typing.Any, +) -> Run: + """Run one generated suite in a subprocess and read the report it wrote.""" + directory = _write_suite(tmp_path_factory.mktemp("suite"), **suite) + reports = tmp_path_factory.mktemp("reports") + result = _pytest(str(directory), report_dir=reports) + + path = reports / file_name + assert path.exists(), ( + f"no report at {path}; pytest exited {result.returncode}\n" + f"{result.stdout}\n{result.stderr}" + ) + return Run( + directory=directory, + result=result, + document=json.loads(path.read_text(encoding="utf-8")), + ) + + +@pytest.fixture(scope="module") +def run(tmp_path_factory: pytest.TempPathFactory) -> Run: + """One real run of the generated suite, with a report asked for.""" + return _run_suite(tmp_path_factory) + + +@pytest.fixture(scope="module") +def narrow_run(tmp_path_factory: pytest.TempPathFactory) -> Run: + """A run of a suite that declares one capability the suite never tests. + + ``@targeting`` is reserved -- it is in the vocabulary and no scenario carries + it. ``@object`` is left undeclared so that a whole Scenario Outline is skipped + by the capability gate, which is the case that has to keep saying which row it + skipped. ``@strict-numeric-typing`` is declared and does have a scenario, so + the omission of ``@targeting`` is specific rather than a general failure to + report capabilities. + """ + return _run_suite( + tmp_path_factory, + file_name="narrow.json", + name="narrow", + capabilities="{Capability.STRICT_NUMERIC_TYPING, Capability.TARGETING}", + deviations=False, + ) + + +# -- the two properties that matter ------------------------------------------ + + +def test_a_capability_skip_is_never_reported_as_passed(run: Run) -> None: + """The rule Appendix F states, checked against the document, not the runner.""" + undeclared = { + tag + for tag, result in run.document["capabilities"].items() + if result["state"] == Outcome.NOT_DECLARED.value + } + assert undeclared, "the generated suite is meant to leave capabilities undeclared" + + gated = [s for s in run.scenarios if undeclared & set(s.get("tags", ()))] + assert gated, "the generated suite is meant to have scenarios behind those" + for scenario in gated: + assert scenario["outcome"] == Outcome.NOT_DECLARED.value, scenario + assert scenario.get("reason"), f"a skip must say why: {scenario}" + + +def test_every_collected_scenario_appears_exactly_once(run: Run) -> None: + """The property that makes the rule above checkable rather than promised. + + An entry is identified by feature, name **and example** together. Feature and + name alone are shared by every row of a Scenario Outline, so keying on them + would let eleven rows of the type-mismatch matrix collapse into one and this + test would not notice -- which is the ambiguity the example field exists to + remove. + + Counted against pytest's own collection rather than against a number written + down here, so that adding a scenario to the specification cannot leave this + passing while the report loses one. + """ + identities = [_identity_of(s) for s in run.scenarios] + assert len(identities) == len(set(identities)), "a scenario is reported twice" + + collected = _pytest("--collect-only", str(run.directory)) + assert len(identities) == sum( + 1 for line in collected.stdout.splitlines() if "::test_" in line + ) + + +def test_the_outcomes_account_for_every_scenario(run: Run) -> None: + counts = collections.Counter(s["outcome"] for s in run.scenarios) + assert set(counts) <= OUTCOMES, "an outcome outside the four the schema allows" + assert sum(counts.values()) == len(run.scenarios) + # All four occur, which is what makes the distinctions worth drawing. + assert set(counts) == OUTCOMES, counts + + +def test_the_document_does_not_repeat_the_runner_summary(run: Run) -> None: + """A known deviation is a failure in the report even when pytest finishes green. + + The suite marks the one scenario the Python SDK cannot satisfy as an expected + failure, so pytest exits zero. The provider still did not satisfy it, and a + document that agreed with the summary would hide exactly what the marker was + added to keep visible. + """ + assert run.result.returncode == 0, run.result.stdout + failed = [s for s in run.scenarios if s["outcome"] == Outcome.FAILED.value] + assert len(failed) == 1 + assert "python-sdk#619" in failed[0]["reason"] + + +def test_a_scenario_skipped_for_another_reason_is_not_a_missing_capability( + run: Run, +) -> None: + """A run that chose not to execute a scenario is a different fact from a gap. + + ``not-applicable`` rather than ``not-declared``, because nothing about the + provider's declared capabilities kept it from running -- and it appears at + all, even though a marker skip never runs a fixture. + """ + matching = [s for s in run.scenarios if s["name"] == UNKNOWN_KEY_SCENARIO] + assert len(matching) == 1 + assert matching[0]["outcome"] == Outcome.NOT_APPLICABLE.value + assert "deliberately not run here" in matching[0]["reason"] + + +# -- which row of an outline ------------------------------------------------- + + +def test_an_outline_row_is_named_by_its_example_not_by_its_name(run: Run) -> None: + """The eleven rows of the type-mismatch matrix are told apart, and only here. + + All eleven share one scenario name, which is the feature file's name and must + stay that way: it is what a report from Go or JavaScript carries for the same + row, and qualifying it with this runner's id for the row -- which an earlier + version of this emitter did -- makes the three disagree about a scenario they + all ran. + """ + rows = [s for s in run.scenarios if s["name"] == MISMATCH_SCENARIO] + expected = _examples_from_the_feature_file("errors", MISMATCH_SCENARIO) + assert len(rows) == len(expected) == 11 + + for row in rows: + assert row["name"] == MISMATCH_SCENARIO, "the name carries a runner's id" + + observed = [row["example"] for row in rows] + assert len(observed) == len({tuple(sorted(e.items())) for e in observed}) + assert sorted(map(sorted, (e.items() for e in observed))) == sorted( + map(sorted, (e.items() for e in expected)) + ) + + +def test_an_example_says_what_the_table_said(run: Run) -> None: + """Verbatim strings, because Gherkin has no types. + + A ``1`` in a table is the two-character cell the feature file contains, and a + report that emitted it as a number would be saying something the table did + not -- and would not validate, since the schema types the values as strings. + """ + rows = [s for s in run.scenarios if s["name"] == MISMATCH_SCENARIO] + for row in rows: + assert all(isinstance(value, str) for value in row["example"].values()), row + + failed = [row for row in rows if row["outcome"] == Outcome.FAILED.value] + assert len(failed) == 1 + assert failed[0]["example"] == DEVIATING_ROW + + +def test_a_scenario_that_is_not_an_outline_has_no_example(run: Run) -> None: + """Omitted rather than empty: there is no row, so there is nothing to say.""" + plain = [s for s in run.scenarios if s["name"] == UNKNOWN_KEY_SCENARIO] + assert len(plain) == 1 + assert "example" not in plain[0] + + +def test_a_capability_skipped_outline_row_still_carries_its_example( + narrow_run: Run, +) -> None: + """A skipped row is exactly as ambiguous as a failed one. + + Identity is established at collection, from the node alone, so it does not + depend on the scenario having run -- which is what lets a row the capability + gate stopped before its first step be told apart from its siblings just as + well as one that failed. + """ + outline = "Requesting a structured flag as a scalar returns the code default" + expected = _examples_from_the_feature_file("errors", outline) + rows = [s for s in narrow_run.scenarios if s["name"] == outline] + assert len(rows) == len(expected) + + for row in rows: + assert row["outcome"] == Outcome.NOT_DECLARED.value, row + assert row.get("example"), f"a skipped outline row must say which row: {row}" + assert sorted(map(sorted, (row["example"].items() for row in rows))) == sorted( + map(sorted, (e.items() for e in expected)) + ) + + +def test_a_row_gated_by_its_examples_block_is_a_capability_skip( + tmp_path: Path, +) -> None: + """Gherkin lets one Examples block of an outline carry its own tags. + + Two rows of one Scenario Outline can therefore differ in which capability + gates them. Those tags are on neither the scenario, the feature nor the rule, + and a report that read only those three would show the skipped row as + carrying no capability -- reporting a capability skip as ``not-applicable``, + which is exactly the distinction Appendix F asks a report to keep, and + leaving the capability out of the rollup. + + No canonical feature file does this yet, so the feature file is written here. + """ + directory = tmp_path / "suite" + directory.mkdir(parents=True) + (directory / "tagged.feature").write_text(_TAGGED_FEATURE, encoding="utf-8") + (directory / "test_tagged.py").write_text(_TAGGED_SUITE, encoding="utf-8") + + reports = tmp_path / "reports" + result = _pytest(str(directory), report_dir=reports) + path = reports / "per-examples.json" + assert path.exists(), f"pytest exited {result.returncode}\n{result.stdout}" + + document = json.loads(path.read_text(encoding="utf-8")) + by_row = {row["example"]["requested"]: row for row in document["scenarios"]} + # Every row is still reported: nothing about gating one row of an outline may + # drop its siblings from the document. + assert set(by_row) == {"Boolean", "Integer", "Float"}, document["scenarios"] + assert by_row["Boolean"]["outcome"] == Outcome.PASSED.value + assert by_row["Integer"]["outcome"] == Outcome.PASSED.value + + gated = by_row["Float"] + assert gated["outcome"] == Outcome.NOT_DECLARED.value, gated + assert gated["tags"] == [Capability.OBJECT.tag], gated + assert document["capabilities"][Capability.OBJECT.tag]["state"] == ( + Outcome.NOT_DECLARED.value + ) + + +# -- identity ---------------------------------------------------------------- + + +def test_the_provider_and_its_configuration_are_reported_separately(run: Run) -> None: + assert run.document["provider"]["name"] == "In-Memory Provider" + assert run.document["provider"]["configuration"] == SUITE_NAME + assert run.document["provider"]["language"] == "python" + + +def test_the_report_names_what_ran_it(run: Run) -> None: + assert run.document["schemaVersion"] == "1" + assert ( + run.document["tck"]["implementation"] + == "python-sdk-contrib/tools/openfeature-provider-tck" + ) + assert run.document["sdk"]["name"] == "openfeature-sdk" + assert run.document["sdk"]["version"] + assert len(run.document["tck"]["specRevision"]) >= 7 + assert run.document["backend"]["controlApi"] == "in-process" + + +def test_the_spec_revision_comes_from_the_build() -> None: + """Generated beside the assets, because the submodule is not in the wheel.""" + revision, tree = spec_identity() + assert len(revision) >= 7 + assert tree == "" or len(tree) == 40 + + +# -- opting in --------------------------------------------------------------- + + +def test_no_report_is_written_without_the_environment_variable( + tmp_path: Path, +) -> None: + """The default, and not an error: emitting is a property of the run.""" + directory = _write_suite(tmp_path / "suite") + result = _pytest(str(directory), report_dir=None) + assert result.returncode == 0, result.stdout + assert "report written" not in result.stdout + assert not list(tmp_path.rglob("*.json")) + + +def test_a_report_that_cannot_be_written_fails_the_run(tmp_path: Path) -> None: + """Loudly, because a pipeline that silently got no report serves a stale one. + + The destination is placed under a regular file, which no platform will let + ``mkdir`` turn into a directory. The run itself passes, so a non-zero exit + can only have come from the failure to write. + """ + blocker = tmp_path / "not-a-directory" + blocker.write_text("", encoding="utf-8") + directory = _write_suite(tmp_path / "suite") + result = _pytest(str(directory), report_dir=blocker / "reports") + assert "could not write the conformance report" in result.stdout + assert result.returncode != 0 + + +# -- assembling the document ------------------------------------------------- + + +def test_a_failure_is_not_revised_away_by_a_later_phase() -> None: + """A scenario whose steps passed and whose teardown blew up is a failure.""" + suite = SuiteReport(config=_config()) + identity = _identity() + suite.set_outcome("node", identity, Outcome.FAILED, "teardown exploded") + suite.set_outcome("node", identity, Outcome.PASSED) + assert suite.records["node"].outcome is Outcome.FAILED + assert suite.records["node"].reason == "teardown exploded" + + +def test_an_undeclared_capability_is_reported_with_a_reason() -> None: + suite = SuiteReport(config=_config()) + suite.set_outcome("node", _identity("@events"), Outcome.PASSED) + document = suite.build() + assert document["capabilities"]["@events"] == {"state": Outcome.PASSED.value} + stale = document["capabilities"]["@stale"] + assert stale["state"] == Outcome.NOT_DECLARED.value + assert "@stale" in stale["reason"] + + +def test_a_capability_whose_scenario_failed_is_not_reported_as_passed() -> None: + """And says how much failed, because the schema requires a reason. + + Reached by driving the builder directly: every self-test suite that runs end + to end passes, so nothing else gets near this branch -- and an entry without a + reason would be rejected by the schema at exactly the moment the report + matters most, when a provider is failing. + """ + suite = SuiteReport(config=_config(capabilities={Capability.EVENTS})) + suite.set_outcome("failed", _identity("@events"), Outcome.FAILED, "boom") + suite.set_outcome("passed", _identity("@events"), Outcome.PASSED) + + events = suite.build()["capabilities"]["@events"] + assert events["state"] == Outcome.FAILED.value + assert "1 of 2" in events["reason"], events + + +def test_every_capability_the_report_mentions_can_explain_itself(run: Run) -> None: + """The rule the schema enforces, checked here so a change fails in this package.""" + for tag, result in run.document["capabilities"].items(): + if result["state"] != Outcome.PASSED.value: + assert result.get("reason"), f"{tag} is {result['state']} with no reason" + + +def test_a_capability_no_scenario_exercises_is_not_reported_as_passed( + narrow_run: Run, +) -> None: + """The vacuous pass the capability vocabulary exists to eliminate. + + ``@targeting`` is declared by this suite and carried by no scenario, because + asserting that an evaluation context reached the backend needs an echo + operation the control API does not have. The suite asked no question, so it + has no answer: the tag is absent rather than green, and a consumer sees the + absence rather than a pass it cannot rely on. + """ + capabilities = narrow_run.document["capabilities"] + exercised = {tag for s in narrow_run.scenarios for tag in s.get("tags", ())} + + assert Capability.TARGETING.tag not in exercised, "the premise has changed" + assert Capability.TARGETING.tag not in capabilities, capabilities.get( + Capability.TARGETING.tag + ) + + # Specific rather than a general failure to report: the other declared + # capability is exercised, and is still reported. + numeric = Capability.STRICT_NUMERIC_TYPING.tag + assert numeric in exercised + assert capabilities[numeric]["state"] == Outcome.PASSED.value + # And an undeclared capability is still reported, with its reason, whether or + # not any scenario carries it: that is a fact about the provider. + assert capabilities[Capability.OBJECT.tag]["state"] == Outcome.NOT_DECLARED.value + + +def test_the_provider_name_falls_back_to_the_suite_name() -> None: + """A suite whose every scenario was skipped never saw a provider. + + Reporting the suite name is more useful than the empty string the schema + would reject. + """ + assert SuiteReport(config=_config()).build()["provider"]["name"] == "stub" + + +def test_the_control_api_is_omitted_when_the_control_does_not_say() -> None: + assert "controlApi" not in SuiteReport(config=_config()).build()["backend"] + http = SuiteReport(config=_config(control=_HttpControl())).build() + assert http["backend"]["controlApi"] == "http" + + +def test_control_api_ignores_a_value_the_schema_would_reject() -> None: + class Odd(_StubControl): + control_api = "carrier pigeon" + + assert control_api_of(Odd()) == "" + + +@pytest.mark.parametrize( + ("suite_name", "expected"), + [ + ("in-memory", "in-memory.json"), + ("flagd/rpc", "flagd-rpc.json"), + ("../escape", "escape.json"), + ("...", "report.json"), + ], +) +def test_a_suite_name_cannot_write_outside_its_directory( + suite_name: str, expected: str +) -> None: + """Suite names are chosen to read well in a failure message, not to be paths.""" + assert report_file_name(suite_name) == expected + + +def test_only_tags_the_schema_accepts_are_carried() -> None: + assert normalise_tags({"events", "Not A Tag", "stale"}) == ("@events", "@stale") + + +# -- classifying one phase --------------------------------------------------- + + +def test_an_expected_failure_is_still_a_failure() -> None: + """An xfail marker records a known deviation; it does not excuse one.""" + classified = classify_phase( + _phase("skipped", xfail_reason="the SDK coerces a bool to an int"), + _identity(), + _config(), + ) + assert classified is not None + outcome, reason = classified + assert outcome is Outcome.FAILED + assert "the SDK coerces a bool to an int" in reason + + +def test_a_phase_that_merely_worked_says_nothing() -> None: + assert ( + classify_phase(_phase("passed", when="setup"), _identity(), _config()) is None + ) + assert classify_phase(_phase("passed", when="call"), _identity(), _config()) == ( + Outcome.PASSED, + "", + ) + + +def test_a_gated_skip_and_an_ungated_skip_are_different_outcomes() -> None: + config = _config(capabilities={Capability.EVENTS}) + gated = classify_phase(_phase("skipped", when="setup"), _identity("@stale"), config) + assert gated == (Outcome.NOT_DECLARED, "provider does not declare @stale") + + ungated = classify_phase( + _phase("skipped", when="setup"), _identity("@events"), config + ) + assert ungated == (Outcome.NOT_APPLICABLE, "skipped")