From e8b0b114b0f76ee1b1acc5e905a66e8c3cdd8631 Mon Sep 17 00:00:00 2001 From: kevin Date: Mon, 14 Sep 2026 11:35:57 +0800 Subject: [PATCH 1/2] Fix verified network event contracts --- assets/dataset/webarena-verified.json | 40 ++++----- assets/dataset/webarna-verfied-hard.json | 7 +- .../evaluators/network_event_evaluator.py | 7 -- tests/api/test_navigation_xhr_evaluator.py | 82 +++++++++++++++++++ .../test_known_contract_corrections.py | 53 ++++++++++++ 5 files changed, 159 insertions(+), 30 deletions(-) create mode 100644 tests/api/test_navigation_xhr_evaluator.py create mode 100644 tests/dataset/test_known_contract_corrections.py diff --git a/assets/dataset/webarena-verified.json b/assets/dataset/webarena-verified.json index eb73812..50e8234 100644 --- a/assets/dataset/webarena-verified.json +++ b/assets/dataset/webarena-verified.json @@ -8593,7 +8593,7 @@ }, { "evaluator": "NetworkEventEvaluator", - "expected": {"url": "^__GITLAB__/a11yproject/a11yproject.com/-/issues/.*$"}, + "expected": {"url": "^__GITLAB__/a11yproject/a11yproject\\.com/-/issues/?$"}, "ignored_query_params_patterns": [".*"] }, { @@ -8608,7 +8608,7 @@ } } ], - "revision": 2 + "revision": 3 }, { "sites": ["gitlab"], @@ -12232,14 +12232,13 @@ "post_data": { "project[name]": "chatgpt_plugin", "project[namespace_id]": "2505", - "project[path]": "chatgpt_plugin", - "project[initialize_with_readme]": "0" + "project[path]": "chatgpt_plugin" }, "response_status": 302 } } ], - "revision": 2 + "revision": 3 }, { "sites": ["gitlab"], @@ -16695,7 +16694,7 @@ { "evaluator": "NetworkEventEvaluator", "expected": { - "url": ["__REDDIT__/submit", "__REDDIT__/submit/consoles"], + "url": ["__REDDIT__/submit", "__REDDIT__/submit/consoles", "__REDDIT__/submit/gaming"], "http_method": "POST", "post_data": { "submission[title]": "what is the recommended console to buy these days?", @@ -16705,7 +16704,7 @@ } } ], - "revision": 2 + "revision": 3 }, { "sites": ["reddit"], @@ -16843,7 +16842,7 @@ { "evaluator": "NetworkEventEvaluator", "expected": { - "url": ["__REDDIT__/submit", "__REDDIT__/submit/consoles"], + "url": ["__REDDIT__/submit", "__REDDIT__/submit/consoles", "__REDDIT__/submit/gaming"], "http_method": "POST", "post_data": { "submission[title]": "what is the recommended console to buy these days", @@ -16853,7 +16852,7 @@ } } ], - "revision": 2 + "revision": 3 }, { "sites": ["reddit"], @@ -16965,7 +16964,7 @@ { "evaluator": "NetworkEventEvaluator", "expected": { - "url": ["__REDDIT__/submit", "__REDDIT__/submit/deeplearning"], + "url": ["__REDDIT__/submit", "__REDDIT__/submit/deeplearning", "__REDDIT__/submit/technology"], "http_method": "POST", "post_data": { "submission[title]": "what is the SOTA web navigation agent repo", @@ -16975,7 +16974,7 @@ } } ], - "revision": 2 + "revision": 3 }, { "sites": ["reddit"], @@ -17473,7 +17472,7 @@ { "evaluator": "NetworkEventEvaluator", "expected": { - "url": ["__REDDIT__/submit", "__REDDIT__/submit/deeplearning"], + "url": ["__REDDIT__/submit", "__REDDIT__/submit/deeplearning", "__REDDIT__/submit/technology"], "http_method": "POST", "post_data": { "submission[title]": "the effectiveness of deep learning", @@ -17484,7 +17483,7 @@ } } ], - "revision": 2 + "revision": 3 }, { "sites": ["reddit"], @@ -19058,6 +19057,7 @@ }, { "evaluator": "NetworkEventEvaluator", + "last_event_only": false, "expected": { "url": "^__SHOPPING_ADMIN__/mui/index/render/.*$", "headers": {"referer": "__SHOPPING_ADMIN__/sales/order/"}, @@ -19072,7 +19072,7 @@ "ignored_query_params_patterns": ["^paging", "^sorting", "isAjax"] } ], - "revision": 2 + "revision": 3 }, { "sites": ["shopping_admin"], @@ -19859,15 +19859,15 @@ "response_status": 302, "post_data": { "name": "spring sale", - "website_ids": [1], - "customer_group_ids": [1], + "website_ids[0]": "1", + "customer_group_ids[0]": "1", "simple_action": "by_percent", "discount_amount": 20 } } } ], - "revision": 2 + "revision": 3 }, { "sites": ["shopping_admin"], @@ -19891,15 +19891,15 @@ "response_status": 302, "post_data": { "name": "fall discount", - "website_ids": [1], - "customer_group_ids": [1], + "website_ids[0]": "1", + "customer_group_ids[0]": "1", "simple_action": "cart_fixed", "discount_amount": 10 } } } ], - "revision": 2 + "revision": 3 }, { "sites": ["shopping_admin"], diff --git a/assets/dataset/webarna-verfied-hard.json b/assets/dataset/webarna-verfied-hard.json index 3aba551..e7ed1e0 100644 --- a/assets/dataset/webarna-verfied-hard.json +++ b/assets/dataset/webarna-verfied-hard.json @@ -10893,7 +10893,8 @@ "expected": { "url": [ "__REDDIT__/submit", - "__REDDIT__/submit/deeplearning" + "__REDDIT__/submit/deeplearning", + "__REDDIT__/submit/technology" ], "headers": null, "query_params": null, @@ -10922,7 +10923,7 @@ "instantiation_dict": { "topic": "the effectiveness of deep learning" }, - "revision": 2 + "revision": 3 }, { "sites": [ @@ -16524,4 +16525,4 @@ }, "revision": 2 } -] \ No newline at end of file +] diff --git a/src/webarena_verified/core/evaluation/evaluators/network_event_evaluator.py b/src/webarena_verified/core/evaluation/evaluators/network_event_evaluator.py index 288bbf4..a08ad79 100644 --- a/src/webarena_verified/core/evaluation/evaluators/network_event_evaluator.py +++ b/src/webarena_verified/core/evaluation/evaluators/network_event_evaluator.py @@ -575,13 +575,6 @@ def _filter_events_by_criteria( if not events: return () - # Handle navigate tasks (We only care about the last navigation event) - # For cases where we check navigation events via subsequent XHR/fetch requests, - # we use the normal filtering logic below. - if context.task.is_navigate_task and config.expected.http_method == "GET": - last_navigation_event = [e for e in events if e.is_navigation_event] - return (last_navigation_event[-1],) if last_navigation_event else () - matches = [] try: expected_url: URL = self._normalized_url(config.expected.url, context=context, config=config, strict=True) # type: ignore diff --git a/tests/api/test_navigation_xhr_evaluator.py b/tests/api/test_navigation_xhr_evaluator.py new file mode 100644 index 0000000..4369bc8 --- /dev/null +++ b/tests/api/test_navigation_xhr_evaluator.py @@ -0,0 +1,82 @@ +"""Regression coverage for navigation tasks accepted through an XHR event.""" + +import json +from pathlib import Path +from typing import Any + +from webarena_verified.api import WebArenaVerified +from webarena_verified.types.config import WebArenaVerifiedConfig +from webarena_verified.types.eval import EvalStatus + + +def _entry(url: str, *, referer: str, navigation: bool) -> dict[str, Any]: + headers = [{"name": "Referer", "value": referer}] + if navigation: + headers.extend( + [ + {"name": "Accept", "value": "text/html"}, + {"name": "Sec-Fetch-Dest", "value": "document"}, + {"name": "Sec-Fetch-Mode", "value": "navigate"}, + {"name": "Sec-Fetch-User", "value": "?1"}, + ] + ) + return { + "startedDateTime": "2026-01-01T00:00:00.000Z", + "time": 1, + "request": {"method": "GET", "url": url, "headers": headers, "cookies": [], "queryString": []}, + "response": { + "status": 200, + "headers": [], + "cookies": [], + "content": {"size": 0, "mimeType": "application/json", "text": "{}"}, + "redirectURL": "", + }, + "cache": {}, + "timings": {"send": 0, "wait": 1, "receive": 0}, + } + + +def test_navigate_task_can_be_verified_by_a_matching_xhr(tmp_path: Path) -> None: + base_url = "http://localhost:7780/admin" + trace = { + "log": { + "version": "1.2", + "creator": {"name": "pytest", "version": "1"}, + "entries": [ + _entry( + f"{base_url}/sales/order/", + referer=f"{base_url}/admin/dashboard/", + navigation=True, + ), + _entry( + f"{base_url}/mui/index/render/?namespace=sales_order_grid&search=" + "&keywordUpdated=false&filters%5Bplaceholder%5D=true&filters%5Bstatus%5D=fraud", + referer=f"{base_url}/sales/order/", + navigation=False, + ), + ], + } + } + trace_path = tmp_path / "network.har" + trace_path.write_text(json.dumps(trace)) + + evaluator = WebArenaVerified( + config=WebArenaVerifiedConfig( + environments={ + "__SHOPPING_ADMIN__": { + "urls": [base_url], + "active_url_idx": 0, + "use_header_login": True, + "credentials": {"username": "admin", "password": "admin1234"}, + } + } + ) + ) + result = evaluator.evaluate_task( + task_id=676, + agent_response={"task_type": "NAVIGATE", "status": "SUCCESS", "retrieved_data": None}, + network_trace=trace_path, + ) + + assert result.status == EvalStatus.SUCCESS + assert result.score == 1.0 diff --git a/tests/dataset/test_known_contract_corrections.py b/tests/dataset/test_known_contract_corrections.py new file mode 100644 index 0000000..7cfde4f --- /dev/null +++ b/tests/dataset/test_known_contract_corrections.py @@ -0,0 +1,53 @@ +"""Regression tests for evaluator contracts corrected from observed valid traces.""" + +from typing import Any + + +def _network_expectations(task: dict[str, Any]) -> list[dict[str, Any]]: + return [item["expected"] for item in task["eval"] if item["evaluator"] == "NetworkEventEvaluator"] + + +def test_issue_list_route_accepts_the_query_bearing_page_url( + dataset_by_task_id: dict[int, dict[str, Any]], +) -> None: + route = _network_expectations(dataset_by_task_id[339])[0]["url"] + assert route == r"^__GITLAB__/a11yproject/a11yproject\.com/-/issues/?$" + + +def test_empty_project_does_not_require_an_unchecked_checkbox_field( + dataset_by_task_id: dict[int, dict[str, Any]], +) -> None: + post_data = _network_expectations(dataset_by_task_id[475])[0]["post_data"] + assert "project[initialize_with_readme]" not in post_data + + +def test_reddit_contracts_accept_the_route_bound_to_the_required_forum_id( + dataset_by_task_id: dict[int, dict[str, Any]], +) -> None: + expected_routes = { + 600: "__REDDIT__/submit/gaming", + 605: "__REDDIT__/submit/gaming", + 609: "__REDDIT__/submit/technology", + 625: "__REDDIT__/submit/technology", + } + for task_id, route in expected_routes.items(): + assert route in _network_expectations(dataset_by_task_id[task_id])[0]["url"] + + +def test_order_grid_xhr_is_not_replaced_by_the_last_same_route_event( + dataset_by_task_id: dict[int, dict[str, Any]], +) -> None: + task = dataset_by_task_id[676] + network_evaluator = next(item for item in task["eval"] if item["evaluator"] == "NetworkEventEvaluator") + assert network_evaluator["last_event_only"] is False + + +def test_indexed_form_fields_are_not_encoded_as_singleton_alternatives( + dataset_by_task_id: dict[int, dict[str, Any]], +) -> None: + for task_id in (699, 700): + post_data = _network_expectations(dataset_by_task_id[task_id])[0]["post_data"] + assert post_data["website_ids[0]"] == "1" + assert post_data["customer_group_ids[0]"] == "1" + assert "website_ids" not in post_data + assert "customer_group_ids" not in post_data From e3695e9a1eb24fee12d645640126de7c9372d66b Mon Sep 17 00:00:00 2001 From: kevin Date: Thu, 17 Sep 2026 12:55:40 +0800 Subject: [PATCH 2/2] Preserve final navigation and narrow task contract corrections --- assets/dataset/webarena-verified.json | 22 ++-- assets/dataset/webarna-verfied-hard.json | 7 +- .../network_event_based_evaluation.md | 18 ++- .../evaluators/network_event_evaluator.py | 28 ++++- .../core/evaluation/value_comparator.py | 9 +- src/webarena_verified/types/task.py | 15 +++ tests/api/test_corrected_task_contracts.py | 117 ++++++++++++++++++ tests/api/test_navigation_xhr_evaluator.py | 37 +++++- .../evaluation/test_nullable_alternatives.py | 12 ++ .../test_known_contract_corrections.py | 20 +-- 10 files changed, 245 insertions(+), 40 deletions(-) create mode 100644 tests/api/test_corrected_task_contracts.py create mode 100644 tests/core/evaluation/test_nullable_alternatives.py diff --git a/assets/dataset/webarena-verified.json b/assets/dataset/webarena-verified.json index 50e8234..9a059db 100644 --- a/assets/dataset/webarena-verified.json +++ b/assets/dataset/webarena-verified.json @@ -12232,7 +12232,8 @@ "post_data": { "project[name]": "chatgpt_plugin", "project[namespace_id]": "2505", - "project[path]": "chatgpt_plugin" + "project[path]": "chatgpt_plugin", + "project[initialize_with_readme]": [null, "0"] }, "response_status": 302 } @@ -16694,7 +16695,7 @@ { "evaluator": "NetworkEventEvaluator", "expected": { - "url": ["__REDDIT__/submit", "__REDDIT__/submit/consoles", "__REDDIT__/submit/gaming"], + "url": ["__REDDIT__/submit", "__REDDIT__/submit/consoles"], "http_method": "POST", "post_data": { "submission[title]": "what is the recommended console to buy these days?", @@ -16704,7 +16705,7 @@ } } ], - "revision": 3 + "revision": 2 }, { "sites": ["reddit"], @@ -16842,7 +16843,7 @@ { "evaluator": "NetworkEventEvaluator", "expected": { - "url": ["__REDDIT__/submit", "__REDDIT__/submit/consoles", "__REDDIT__/submit/gaming"], + "url": ["__REDDIT__/submit", "__REDDIT__/submit/consoles"], "http_method": "POST", "post_data": { "submission[title]": "what is the recommended console to buy these days", @@ -16852,7 +16853,7 @@ } } ], - "revision": 3 + "revision": 2 }, { "sites": ["reddit"], @@ -16964,7 +16965,7 @@ { "evaluator": "NetworkEventEvaluator", "expected": { - "url": ["__REDDIT__/submit", "__REDDIT__/submit/deeplearning", "__REDDIT__/submit/technology"], + "url": ["__REDDIT__/submit", "__REDDIT__/submit/deeplearning"], "http_method": "POST", "post_data": { "submission[title]": "what is the SOTA web navigation agent repo", @@ -16974,7 +16975,7 @@ } } ], - "revision": 3 + "revision": 2 }, { "sites": ["reddit"], @@ -17472,7 +17473,7 @@ { "evaluator": "NetworkEventEvaluator", "expected": { - "url": ["__REDDIT__/submit", "__REDDIT__/submit/deeplearning", "__REDDIT__/submit/technology"], + "url": ["__REDDIT__/submit", "__REDDIT__/submit/deeplearning"], "http_method": "POST", "post_data": { "submission[title]": "the effectiveness of deep learning", @@ -17483,7 +17484,7 @@ } } ], - "revision": 3 + "revision": 2 }, { "sites": ["reddit"], @@ -19057,7 +19058,8 @@ }, { "evaluator": "NetworkEventEvaluator", - "last_event_only": false, + "navigation_only": false, + "event_query_params": {"namespace": "sales_order_grid"}, "expected": { "url": "^__SHOPPING_ADMIN__/mui/index/render/.*$", "headers": {"referer": "__SHOPPING_ADMIN__/sales/order/"}, diff --git a/assets/dataset/webarna-verfied-hard.json b/assets/dataset/webarna-verfied-hard.json index e7ed1e0..3aba551 100644 --- a/assets/dataset/webarna-verfied-hard.json +++ b/assets/dataset/webarna-verfied-hard.json @@ -10893,8 +10893,7 @@ "expected": { "url": [ "__REDDIT__/submit", - "__REDDIT__/submit/deeplearning", - "__REDDIT__/submit/technology" + "__REDDIT__/submit/deeplearning" ], "headers": null, "query_params": null, @@ -10923,7 +10922,7 @@ "instantiation_dict": { "topic": "the effectiveness of deep learning" }, - "revision": 3 + "revision": 2 }, { "sites": [ @@ -16525,4 +16524,4 @@ }, "revision": 2 } -] +] \ No newline at end of file diff --git a/docs/evaluation/network_event_based_evaluation.md b/docs/evaluation/network_event_based_evaluation.md index dc40a2f..fc33171 100644 --- a/docs/evaluation/network_event_based_evaluation.md +++ b/docs/evaluation/network_event_based_evaluation.md @@ -116,9 +116,21 @@ right page: ### 4. Event Types and Sequencing -Set `event_type` to `"navigation"` to focus on page loads, or to `"modification"` for form -submissions. The `last_event_only` flag instructs the evaluator to match the most recent event; -disabling it means "any matching event is sufficient". +GET checks on navigation tasks validate the final document navigation by default. +For a page whose state is loaded through XHR or fetch, set `navigation_only` to `false`. +Only requests after the final document navigation are then eligible, so leaving or +reloading the page invalidates earlier requests. + +`event_query_params` selects a request stream by exact query values, for example +`{"namespace": "sales_order_grid"}` when several grids share one endpoint. Keep +the state being checked, such as `filters[status]`, in `expected.query_params`. +`last_event_only` defaults to `true` and checks the last request in that stream; +setting it to `false` accepts any matching request and is unsuitable when the task +requires the final page state. + +Expected scalar alternatives may explicitly include `null`, such as `[null, "0"]` +for an unchecked checkbox that may be omitted or sent as zero. An empty string +alternative alone does not permit a missing value. ## Network Trace Structure diff --git a/src/webarena_verified/core/evaluation/evaluators/network_event_evaluator.py b/src/webarena_verified/core/evaluation/evaluators/network_event_evaluator.py index a08ad79..32c17c2 100644 --- a/src/webarena_verified/core/evaluation/evaluators/network_event_evaluator.py +++ b/src/webarena_verified/core/evaluation/evaluators/network_event_evaluator.py @@ -8,6 +8,7 @@ from functools import partial from types import MappingProxyType from typing import TYPE_CHECKING, Any, cast +from urllib.parse import parse_qs, urlsplit from webarena_verified.core.evaluation.data_types import URL from webarena_verified.core.utils import logger @@ -558,6 +559,26 @@ def _extract_event_data( response_cookies=response_cookies, ) + @staticmethod + def _select_navigation_events( + events: tuple[NetworkEvent, ...], context: TaskEvalContext, config: NetworkEventEvaluatorCfg + ) -> tuple[tuple[NetworkEvent, ...], bool]: + if not (context.task.is_navigate_task and config.expected.http_method == "GET"): + return events, False + navigations = [index for index, event in enumerate(events) if event.is_navigation_event] + if not navigations: + return (), True + last_navigation = navigations[-1] + if config.navigation_only: + return (events[last_navigation],), True + # Earlier XHRs describe a page the agent has left or reloaded. + return events[last_navigation + 1 :], False + + @staticmethod + def _matches_event_query(event: NetworkEvent, config: NetworkEventEvaluatorCfg) -> bool: + query = parse_qs(urlsplit(event.url).query, keep_blank_values=True) + return all(query.get(key) == [value] for key, value in (config.event_query_params or {}).items()) + def _filter_events_by_criteria( self, events: tuple[NetworkEvent, ...], context: TaskEvalContext, config: NetworkEventEvaluatorCfg ) -> tuple[NetworkEvent, ...]: @@ -572,8 +593,9 @@ def _filter_events_by_criteria( Filtered list of events matching URL and headers criteria """ - if not events: - return () + events, final_navigation = self._select_navigation_events(events, context, config) + if final_navigation or not events: + return events matches = [] try: @@ -588,6 +610,8 @@ def _filter_events_by_criteria( raise for event in events: + if not self._matches_event_query(event, config): + continue # Check HTTP method if config.expected.http_method and event.http_method.lower() != config.expected.http_method.lower(): continue diff --git a/src/webarena_verified/core/evaluation/value_comparator.py b/src/webarena_verified/core/evaluation/value_comparator.py index 2f4dfdf..5f74e9f 100644 --- a/src/webarena_verified/core/evaluation/value_comparator.py +++ b/src/webarena_verified/core/evaluation/value_comparator.py @@ -458,7 +458,14 @@ def _compare_recursive( visited.add(actual_id) # Handle None values - if expected is None and actual is None: + if actual is None and ( + expected is None + or ( + isinstance(expected, NormalizedType) + and isinstance(expected.raw, (list, tuple)) + and any(value is None for value in expected.raw) + ) + ): return [] if expected is None or actual is None: diff --git a/src/webarena_verified/types/task.py b/src/webarena_verified/types/task.py index 115253c..fffdfd4 100644 --- a/src/webarena_verified/types/task.py +++ b/src/webarena_verified/types/task.py @@ -228,6 +228,21 @@ class NetworkEventEvaluatorCfg(BaseEval[NetworkEventSpec]): last_event_only: bool = True """If True, validate only the last matching event. If False, validate if ANY event matches.""" + navigation_only: bool = True + """For GET checks on navigation tasks, validate the final document navigation. + + Set False for XHR/fetch checks on the final page. Those checks consider only + events after the final document navigation and still honor last_event_only. + Other task types and HTTP methods are unaffected. + """ + + event_query_params: dict[str, str] | None = None + """Exact query values that identify an event stream before selecting its last event. + + Use stable identifiers such as a grid namespace, not the state being verified. + Expected query_params still validate the selected event's complete state. + """ + ignored_query_params: tuple[str, ...] | None = None """Query parameter names to ignore during comparison (literal matching).""" diff --git a/tests/api/test_corrected_task_contracts.py b/tests/api/test_corrected_task_contracts.py new file mode 100644 index 0000000..d096350 --- /dev/null +++ b/tests/api/test_corrected_task_contracts.py @@ -0,0 +1,117 @@ +"""Exercise corrected contracts through the public HAR evaluation API.""" + +import json +from pathlib import Path +from urllib.parse import urlencode + +import pytest + +from webarena_verified.api import WebArenaVerified +from webarena_verified.types.config import WebArenaVerifiedConfig +from webarena_verified.types.eval import EvalStatus + + +def _event(url, *, method="GET", form=None, referer=None, navigation=False): + headers = [{"name": "Referer", "value": referer}] if referer else [] + if navigation: + headers += [ + {"name": "Accept", "value": "text/html"}, + {"name": "Sec-Fetch-Dest", "value": "document"}, + {"name": "Sec-Fetch-Mode", "value": "navigate"}, + {"name": "Sec-Fetch-User", "value": "?1"}, + ] + request = {"method": method, "url": url, "headers": headers, "cookies": [], "queryString": []} + if form is not None: + request["postData"] = {"mimeType": "application/x-www-form-urlencoded", "text": urlencode(form)} + return { + "startedDateTime": "2026-01-01T00:00:00.000Z", + "time": 1, + "request": request, + "response": { + "status": 302 if form is not None else 200, + "headers": [], + "cookies": [], + "content": {"size": 0, "mimeType": "text/html", "text": ""}, + "redirectURL": "", + }, + "cache": {}, + "timings": {"send": 0, "wait": 1, "receive": 0}, + } + + +def _evaluate(tmp_path, task, entries, task_type): + path = tmp_path / "network.har" + path.write_text( + json.dumps({"log": {"version": "1.2", "creator": {"name": "pytest", "version": "1"}, "entries": entries}}) + ) + evaluator = WebArenaVerified( + config=WebArenaVerifiedConfig( + test_data_file=Path(__file__).parents[2] / "assets/dataset/webarena-verified.json", + environments={ + "__GITLAB__": {"urls": ["http://localhost:8023"]}, + "__SHOPPING_ADMIN__": {"urls": ["http://localhost:7780/admin"]}, + }, + ) + ) + return ( + evaluator.evaluate_task( + task_id=task, + agent_response={ + "task_type": task_type, + "status": "SUCCESS", + "retrieved_data": None, + }, + network_trace=path, + ).status + == EvalStatus.SUCCESS + ) + + +@pytest.mark.parametrize("left_target", [False, True]) +def test_navigation_must_end_on_the_requested_page(tmp_path, left_target): + entries = [_event("http://localhost:8023/dashboard/todos", navigation=True)] + if left_target: + entries.append(_event("http://localhost:8023/dashboard/projects", navigation=True)) + assert _evaluate(tmp_path, 44, entries, "NAVIGATE") is (not left_target) + + +@pytest.mark.parametrize("route", ["issues", "issues/", "issues/123"]) +def test_issue_list_route_does_not_accept_an_issue_detail(tmp_path, route): + base = "http://localhost:8023" + expected = base + "/a11yproject/a11yproject.com/-/issues/?state=opened&label_name%5B%5D=bug" + entries = [ + _event( + base + "/a11yproject/a11yproject.com/-/" + route + "?state=opened&label_name%5B%5D=bug", navigation=True + ), + _event(base + "/api/graphql", method="POST", referer=expected), + ] + assert _evaluate(tmp_path, 339, entries, "NAVIGATE") is (route != "issues/123") + + +@pytest.mark.parametrize("readme", [None, "0", "1"]) +def test_empty_project_accepts_unchecked_but_rejects_enabled_readme(tmp_path, readme): + form = {"project[name]": "chatgpt_plugin", "project[path]": "chatgpt_plugin", "project[namespace_id]": "2505"} + if readme is not None: + form["project[initialize_with_readme]"] = readme + entries = [_event("http://localhost:8023/projects", method="POST", form=form)] + assert _evaluate(tmp_path, 475, entries, "MUTATE") is (readme != "1") + + +@pytest.mark.parametrize( + ("task", "name", "action", "amount"), + [ + (699, "spring sale", "by_percent", 20), + (700, "fall discount", "cart_fixed", 10), + ], +) +@pytest.mark.parametrize("customer_group", ["1", "2"]) +def test_price_rule_form_requires_the_requested_customer_group(tmp_path, task, name, action, amount, customer_group): + form = { + "name": name, + "website_ids[0]": "1", + "customer_group_ids[0]": customer_group, + "simple_action": action, + "discount_amount": amount, + } + entries = [_event("http://localhost:7780/admin/sales_rule/promo_quote/save/", method="POST", form=form)] + assert _evaluate(tmp_path, task, entries, "MUTATE") is (customer_group == "1") diff --git a/tests/api/test_navigation_xhr_evaluator.py b/tests/api/test_navigation_xhr_evaluator.py index 4369bc8..b082455 100644 --- a/tests/api/test_navigation_xhr_evaluator.py +++ b/tests/api/test_navigation_xhr_evaluator.py @@ -4,6 +4,8 @@ from pathlib import Path from typing import Any +import pytest + from webarena_verified.api import WebArenaVerified from webarena_verified.types.config import WebArenaVerifiedConfig from webarena_verified.types.eval import EvalStatus @@ -36,7 +38,8 @@ def _entry(url: str, *, referer: str, navigation: bool) -> dict[str, Any]: } -def test_navigate_task_can_be_verified_by_a_matching_xhr(tmp_path: Path) -> None: +@pytest.mark.parametrize("ending", ["fraud", "notification", "cleared", "left_page", "reloaded"]) +def test_navigate_task_checks_the_final_order_grid(tmp_path: Path, ending: str) -> None: base_url = "http://localhost:7780/admin" trace = { "log": { @@ -57,11 +60,38 @@ def test_navigate_task_can_be_verified_by_a_matching_xhr(tmp_path: Path) -> None ], } } + if ending == "notification": + trace["log"]["entries"].append( + _entry( + f"{base_url}/mui/index/render/?namespace=notification_area", + referer=f"{base_url}/sales/order/", + navigation=False, + ) + ) + elif ending == "cleared": + trace["log"]["entries"].append( + _entry( + f"{base_url}/mui/index/render/?namespace=sales_order_grid&search=" + "&keywordUpdated=false&filters%5Bplaceholder%5D=true", + referer=f"{base_url}/sales/order/", + navigation=False, + ) + ) + elif ending in {"left_page", "reloaded"}: + path = "admin/dashboard/" if ending == "left_page" else "sales/order/" + trace["log"]["entries"].append( + _entry( + f"{base_url}/{path}", + referer=f"{base_url}/sales/order/", + navigation=True, + ) + ) trace_path = tmp_path / "network.har" trace_path.write_text(json.dumps(trace)) evaluator = WebArenaVerified( config=WebArenaVerifiedConfig( + test_data_file=Path(__file__).parents[2] / "assets/dataset/webarena-verified.json", environments={ "__SHOPPING_ADMIN__": { "urls": [base_url], @@ -69,7 +99,7 @@ def test_navigate_task_can_be_verified_by_a_matching_xhr(tmp_path: Path) -> None "use_header_login": True, "credentials": {"username": "admin", "password": "admin1234"}, } - } + }, ) ) result = evaluator.evaluate_task( @@ -78,5 +108,4 @@ def test_navigate_task_can_be_verified_by_a_matching_xhr(tmp_path: Path) -> None network_trace=trace_path, ) - assert result.status == EvalStatus.SUCCESS - assert result.score == 1.0 + assert (result.status == EvalStatus.SUCCESS) is (ending in {"fraud", "notification"}) diff --git a/tests/core/evaluation/test_nullable_alternatives.py b/tests/core/evaluation/test_nullable_alternatives.py new file mode 100644 index 0000000..0e40541 --- /dev/null +++ b/tests/core/evaluation/test_nullable_alternatives.py @@ -0,0 +1,12 @@ +"""Null alternatives must be explicit; empty strings do not make a field optional.""" + +import pytest + +from webarena_verified.core.evaluation.data_types import NormalizedString +from webarena_verified.core.evaluation.value_comparator import ValueComparator + + +@pytest.mark.parametrize(("expected", "accepted"), [([None, "0"], True), (["", "0"], False), (["0", "1"], False)]) +def test_absence_requires_an_explicit_null_alternative(expected, accepted): + failures = ValueComparator().compare(actual=None, expected=NormalizedString(expected)) + assert (not failures) is accepted diff --git a/tests/dataset/test_known_contract_corrections.py b/tests/dataset/test_known_contract_corrections.py index 7cfde4f..12c2f64 100644 --- a/tests/dataset/test_known_contract_corrections.py +++ b/tests/dataset/test_known_contract_corrections.py @@ -18,28 +18,16 @@ def test_empty_project_does_not_require_an_unchecked_checkbox_field( dataset_by_task_id: dict[int, dict[str, Any]], ) -> None: post_data = _network_expectations(dataset_by_task_id[475])[0]["post_data"] - assert "project[initialize_with_readme]" not in post_data + assert post_data["project[initialize_with_readme]"] == [None, "0"] -def test_reddit_contracts_accept_the_route_bound_to_the_required_forum_id( - dataset_by_task_id: dict[int, dict[str, Any]], -) -> None: - expected_routes = { - 600: "__REDDIT__/submit/gaming", - 605: "__REDDIT__/submit/gaming", - 609: "__REDDIT__/submit/technology", - 625: "__REDDIT__/submit/technology", - } - for task_id, route in expected_routes.items(): - assert route in _network_expectations(dataset_by_task_id[task_id])[0]["url"] - - -def test_order_grid_xhr_is_not_replaced_by_the_last_same_route_event( +def test_order_grid_checks_the_last_xhr_on_the_final_page( dataset_by_task_id: dict[int, dict[str, Any]], ) -> None: task = dataset_by_task_id[676] network_evaluator = next(item for item in task["eval"] if item["evaluator"] == "NetworkEventEvaluator") - assert network_evaluator["last_event_only"] is False + assert network_evaluator["navigation_only"] is False + assert network_evaluator.get("last_event_only", True) is True def test_indexed_form_fields_are_not_encoded_as_singleton_alternatives(