2026-06-13 21:44:42 +01:00
|
|
|
"""Corpus-backed tests for dom round-trip, free-branch ownership, and fitness parity.
|
Geometry inner loop: batched full-objective ratio optimiser (CMA-ES)
innerloop.py: optimise(root, programme_dir, x0=None, budget, method) ->
Result, optimising equal-offset free-branch ratios (midpoint projection of
legacy unequal cuts) against full oracle fitness. OracleEvaluator scores
each population in one batched perl call. Methods: cma (default) — multi-
start sigma ladder (0.05 local, 0.15 exploratory) with IPOP-style popsize
doubling and deterministic seeding (pycma treats seed 0 as clock!) — and
compass with Hooke-Jeeves pattern moves, kept for the d0s bake-off.
Acceptance (experiments/accept_innerloop.py, §4.5 bars vs unprojected
originals, within-noise tolerance 1%): x1.65 / x1.66 / x1.58 against bars
x1.24 / x1.67 / x1.59, no new failures, 46 oracle calls vs Nelder-Mead's
200. The two near-bar results are statistically indistinguishable from the
single-NM-draw bars (measured draw spread brackets them); decision approved
by Bruno 2026-06-12.
Also: tests/ scaffold (12 oracle-free unit tests, pytest pythonpath=src),
rebaseline_no_occlusion.py for homemaker-py-gp2, cma>=3.0 dependency
(installed via dnf), dead-variable cleanup in solver.py.
Closes homemaker-py-1p0.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-06-12 09:42:24 +01:00
|
|
|
|
|
|
|
|
Skipped when the Urb checkout is absent (these need only its .dom files, not
|
2026-06-13 21:44:42 +01:00
|
|
|
perl). The parity tests compare native Python fitness against cached oracle
|
|
|
|
|
scores and failure sets (generated with URB_NO_OCCLUSION=1).
|
Geometry inner loop: batched full-objective ratio optimiser (CMA-ES)
innerloop.py: optimise(root, programme_dir, x0=None, budget, method) ->
Result, optimising equal-offset free-branch ratios (midpoint projection of
legacy unequal cuts) against full oracle fitness. OracleEvaluator scores
each population in one batched perl call. Methods: cma (default) — multi-
start sigma ladder (0.05 local, 0.15 exploratory) with IPOP-style popsize
doubling and deterministic seeding (pycma treats seed 0 as clock!) — and
compass with Hooke-Jeeves pattern moves, kept for the d0s bake-off.
Acceptance (experiments/accept_innerloop.py, §4.5 bars vs unprojected
originals, within-noise tolerance 1%): x1.65 / x1.66 / x1.58 against bars
x1.24 / x1.67 / x1.59, no new failures, 46 oracle calls vs Nelder-Mead's
200. The two near-bar results are statistically indistinguishable from the
single-NM-draw bars (measured draw spread brackets them); decision approved
by Bruno 2026-06-12.
Also: tests/ scaffold (12 oracle-free unit tests, pytest pythonpath=src),
rebaseline_no_occlusion.py for homemaker-py-gp2, cma>=3.0 dependency
(installed via dnf), dead-variable cleanup in solver.py.
Closes homemaker-py-1p0.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-06-12 09:42:24 +01:00
|
|
|
"""
|
|
|
|
|
|
2026-06-13 21:44:42 +01:00
|
|
|
import math
|
Geometry inner loop: batched full-objective ratio optimiser (CMA-ES)
innerloop.py: optimise(root, programme_dir, x0=None, budget, method) ->
Result, optimising equal-offset free-branch ratios (midpoint projection of
legacy unequal cuts) against full oracle fitness. OracleEvaluator scores
each population in one batched perl call. Methods: cma (default) — multi-
start sigma ladder (0.05 local, 0.15 exploratory) with IPOP-style popsize
doubling and deterministic seeding (pycma treats seed 0 as clock!) — and
compass with Hooke-Jeeves pattern moves, kept for the d0s bake-off.
Acceptance (experiments/accept_innerloop.py, §4.5 bars vs unprojected
originals, within-noise tolerance 1%): x1.65 / x1.66 / x1.58 against bars
x1.24 / x1.67 / x1.59, no new failures, 46 oracle calls vs Nelder-Mead's
200. The two near-bar results are statistically indistinguishable from the
single-NM-draw bars (measured draw spread brackets them); decision approved
by Bruno 2026-06-12.
Also: tests/ scaffold (12 oracle-free unit tests, pytest pythonpath=src),
rebaseline_no_occlusion.py for homemaker-py-gp2, cma>=3.0 dependency
(installed via dnf), dead-variable cleanup in solver.py.
Closes homemaker-py-1p0.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-06-12 09:42:24 +01:00
|
|
|
from pathlib import Path
|
|
|
|
|
|
|
|
|
|
import pytest
|
|
|
|
|
|
2026-06-14 08:18:06 +01:00
|
|
|
from homemaker_layout import dom, solver
|
Geometry inner loop: batched full-objective ratio optimiser (CMA-ES)
innerloop.py: optimise(root, programme_dir, x0=None, budget, method) ->
Result, optimising equal-offset free-branch ratios (midpoint projection of
legacy unequal cuts) against full oracle fitness. OracleEvaluator scores
each population in one batched perl call. Methods: cma (default) — multi-
start sigma ladder (0.05 local, 0.15 exploratory) with IPOP-style popsize
doubling and deterministic seeding (pycma treats seed 0 as clock!) — and
compass with Hooke-Jeeves pattern moves, kept for the d0s bake-off.
Acceptance (experiments/accept_innerloop.py, §4.5 bars vs unprojected
originals, within-noise tolerance 1%): x1.65 / x1.66 / x1.58 against bars
x1.24 / x1.67 / x1.59, no new failures, 46 oracle calls vs Nelder-Mead's
200. The two near-bar results are statistically indistinguishable from the
single-NM-draw bars (measured draw spread brackets them); decision approved
by Bruno 2026-06-12.
Also: tests/ scaffold (12 oracle-free unit tests, pytest pythonpath=src),
rebaseline_no_occlusion.py for homemaker-py-gp2, cma>=3.0 dependency
(installed via dnf), dead-variable cleanup in solver.py.
Closes homemaker-py-1p0.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-06-12 09:42:24 +01:00
|
|
|
|
2026-06-13 23:39:20 +01:00
|
|
|
CORPUS = Path(__file__).parent.parent / "examples" / "programme-house"
|
Geometry inner loop: batched full-objective ratio optimiser (CMA-ES)
innerloop.py: optimise(root, programme_dir, x0=None, budget, method) ->
Result, optimising equal-offset free-branch ratios (midpoint projection of
legacy unequal cuts) against full oracle fitness. OracleEvaluator scores
each population in one batched perl call. Methods: cma (default) — multi-
start sigma ladder (0.05 local, 0.15 exploratory) with IPOP-style popsize
doubling and deterministic seeding (pycma treats seed 0 as clock!) — and
compass with Hooke-Jeeves pattern moves, kept for the d0s bake-off.
Acceptance (experiments/accept_innerloop.py, §4.5 bars vs unprojected
originals, within-noise tolerance 1%): x1.65 / x1.66 / x1.58 against bars
x1.24 / x1.67 / x1.59, no new failures, 46 oracle calls vs Nelder-Mead's
200. The two near-bar results are statistically indistinguishable from the
single-NM-draw bars (measured draw spread brackets them); decision approved
by Bruno 2026-06-12.
Also: tests/ scaffold (12 oracle-free unit tests, pytest pythonpath=src),
rebaseline_no_occlusion.py for homemaker-py-gp2, cma>=3.0 dependency
(installed via dnf), dead-variable cleanup in solver.py.
Closes homemaker-py-1p0.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-06-12 09:42:24 +01:00
|
|
|
|
2026-06-13 23:39:20 +01:00
|
|
|
pytestmark = pytest.mark.skipif(not CORPUS.is_dir(), reason="Corpus not available")
|
Geometry inner loop: batched full-objective ratio optimiser (CMA-ES)
innerloop.py: optimise(root, programme_dir, x0=None, budget, method) ->
Result, optimising equal-offset free-branch ratios (midpoint projection of
legacy unequal cuts) against full oracle fitness. OracleEvaluator scores
each population in one batched perl call. Methods: cma (default) — multi-
start sigma ladder (0.05 local, 0.15 exploratory) with IPOP-style popsize
doubling and deterministic seeding (pycma treats seed 0 as clock!) — and
compass with Hooke-Jeeves pattern moves, kept for the d0s bake-off.
Acceptance (experiments/accept_innerloop.py, §4.5 bars vs unprojected
originals, within-noise tolerance 1%): x1.65 / x1.66 / x1.58 against bars
x1.24 / x1.67 / x1.59, no new failures, 46 oracle calls vs Nelder-Mead's
200. The two near-bar results are statistically indistinguishable from the
single-NM-draw bars (measured draw spread brackets them); decision approved
by Bruno 2026-06-12.
Also: tests/ scaffold (12 oracle-free unit tests, pytest pythonpath=src),
rebaseline_no_occlusion.py for homemaker-py-gp2, cma>=3.0 dependency
(installed via dnf), dead-variable cleanup in solver.py.
Closes homemaker-py-1p0.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-06-12 09:42:24 +01:00
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_roundtrip_idempotent_and_area_preserving(tmp_path):
|
|
|
|
|
# dump() does not reproduce the source bytes (different YAML style); the
|
|
|
|
|
# real invariants are that a dumped file reloads to the same dump (stable
|
|
|
|
|
# fixed point) and that per-leaf geometry survives the trip (§4.1 is the
|
|
|
|
|
# area validation against Urb itself).
|
2026-06-14 08:18:06 +01:00
|
|
|
from homemaker_layout import geometry
|
Geometry inner loop: batched full-objective ratio optimiser (CMA-ES)
innerloop.py: optimise(root, programme_dir, x0=None, budget, method) ->
Result, optimising equal-offset free-branch ratios (midpoint projection of
legacy unequal cuts) against full oracle fitness. OracleEvaluator scores
each population in one batched perl call. Methods: cma (default) — multi-
start sigma ladder (0.05 local, 0.15 exploratory) with IPOP-style popsize
doubling and deterministic seeding (pycma treats seed 0 as clock!) — and
compass with Hooke-Jeeves pattern moves, kept for the d0s bake-off.
Acceptance (experiments/accept_innerloop.py, §4.5 bars vs unprojected
originals, within-noise tolerance 1%): x1.65 / x1.66 / x1.58 against bars
x1.24 / x1.67 / x1.59, no new failures, 46 oracle calls vs Nelder-Mead's
200. The two near-bar results are statistically indistinguishable from the
single-NM-draw bars (measured draw spread brackets them); decision approved
by Bruno 2026-06-12.
Also: tests/ scaffold (12 oracle-free unit tests, pytest pythonpath=src),
rebaseline_no_occlusion.py for homemaker-py-gp2, cma>=3.0 dependency
(installed via dnf), dead-variable cleanup in solver.py.
Closes homemaker-py-1p0.
Co-Authored-By: Claude Fable 5 <noreply@anthropic.com>
2026-06-12 09:42:24 +01:00
|
|
|
|
|
|
|
|
for src in sorted(CORPUS.glob("*.dom")):
|
|
|
|
|
root = dom.load(str(src))
|
|
|
|
|
areas = [geometry.area(leaf) for lvl in dom.levels(root) for leaf in lvl.leaves()]
|
|
|
|
|
once = tmp_path / ("once_" + src.name)
|
|
|
|
|
dom.dump(root, str(once))
|
|
|
|
|
|
|
|
|
|
root2 = dom.load(str(once))
|
|
|
|
|
areas2 = [geometry.area(leaf) for lvl in dom.levels(root2) for leaf in lvl.leaves()]
|
|
|
|
|
assert areas == pytest.approx(areas2, abs=1e-9), src.name
|
|
|
|
|
|
|
|
|
|
twice = tmp_path / ("twice_" + src.name)
|
|
|
|
|
dom.dump(root2, str(twice))
|
|
|
|
|
assert twice.read_bytes() == once.read_bytes(), src.name
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_free_branches_known_dof():
|
|
|
|
|
# DOF figures from DESIGN.md §4.5
|
|
|
|
|
expected = {
|
|
|
|
|
"2f45907abd9accac2a124d311732f749.dom": 7,
|
|
|
|
|
"candidate-002.dom": 6,
|
|
|
|
|
"c964435454c459f86c3ed9a5a7621132.dom": 6,
|
|
|
|
|
}
|
|
|
|
|
for name, dof in expected.items():
|
|
|
|
|
root = dom.load(str(CORPUS / name))
|
|
|
|
|
assert len(solver.free_branches(root)) == dof, name
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_free_branches_are_lowest_storey_owners():
|
|
|
|
|
for src in sorted(CORPUS.glob("*.dom")):
|
|
|
|
|
root = dom.load(str(src))
|
|
|
|
|
for b in solver.free_branches(root):
|
|
|
|
|
assert b.divided
|
|
|
|
|
assert b.below is None or not b.below.divided
|
2026-06-13 21:44:42 +01:00
|
|
|
|
|
|
|
|
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
# Phase 3 gate: native fitness parity vs oracle (homemaker-py-uxz)
|
|
|
|
|
# Oracle scores and failure sets cached as <file>.dom.score / <file>.dom.fails
|
|
|
|
|
# generated with URB_NO_OCCLUSION=1 (DESIGN.md §6 descope).
|
|
|
|
|
# ---------------------------------------------------------------------------
|
|
|
|
|
|
|
|
|
|
def _native_evaluate(src: Path):
|
|
|
|
|
"""Run native Fitness.evaluate and return (score, frozenset[fail_lines])."""
|
2026-06-14 08:18:06 +01:00
|
|
|
from homemaker_layout import fitness as fit_mod, graph as graph_mod, geometry
|
2026-06-13 21:44:42 +01:00
|
|
|
|
|
|
|
|
root = dom.load(str(src))
|
|
|
|
|
conf, cost = fit_mod.load_config(CORPUS)
|
|
|
|
|
fit = fit_mod.Fitness(conf, cost)
|
|
|
|
|
|
|
|
|
|
failures: list[str] = []
|
|
|
|
|
tracking: dict = {
|
|
|
|
|
"has_public_access_outside": False,
|
|
|
|
|
"has_public_access_inside": False,
|
|
|
|
|
"public_length_all": 0.0,
|
|
|
|
|
"public_length_outside": 0.0,
|
|
|
|
|
"private_length_all": 0.0,
|
|
|
|
|
"private_length_outside": 0.0,
|
|
|
|
|
"stair_fit": [],
|
|
|
|
|
"_failures": failures,
|
|
|
|
|
}
|
|
|
|
|
|
|
|
|
|
programme = fit._programme or {}
|
|
|
|
|
geometry.clear_cache()
|
|
|
|
|
|
|
|
|
|
check_f, missing = graph_mod.check_space_counts(root, programme)
|
|
|
|
|
failures.extend(check_f)
|
|
|
|
|
fit.preprocess_building(root)
|
§39.7: access requirements become a declared `usage:` attribute (homemaker-py-sel)
Closes the second namespace sharing a first character with programme codes: the
usage prefixes b/t/l/k, under which a room silently inherited another room's
connectivity rules from its spelling.
usage is a plain, MANDATORY attribute of the space definition -- not a lookup
table. An interim design proposed a top-level usage_classes: table binding
author-coined names to behaviour; withdrawn, because an indirect name->behaviour
mapping living apart from the thing it describes is exactly the shape of the
prefix rule §39 exists to remove, it would be the only such table in a schema
where every other space property is a plain attribute, and the need it served
was already met -- "building specific" is about what a room is CALLED, and
name: is already free text.
Rule that settles it: a usage value exists iff the engine treats it differently
somewhere. Config selects among behaviours; it cannot invent them.
- programme.USAGES (living/kitchen/bedroom/toilet/utility/none) plus the
behaviour groupings PRIVATE_USAGES / PRIVATE_STRIPS / TOILET_STRIPS /
SOCIABLE_USAGES. Missing or unknown usage is a load error naming the code,
from BOTH parse paths.
- Code-level, never leaf-level: usage_of(leaf.type) is looked up fresh, so a
retype changes the class automatically. 51 sites assign leaf.type, and
share/share_type plus the r5a resurrection are the precedent for why
leaf-level attributes rot.
- graph.has_circulation takes the usage map and trims on declared class;
fitness.access and the public-access check likewise. fitness._t0 is DELETED --
no first-character type test remains anywhere in the codebase.
- utility is distinct from bedroom (same access requirements today) because it
is a different use and gives derive_interchange_classes an axis to relax on.
- A toilet now keeps its edge to a terminal room -- the Brand adjacency, which
the old b-before-t loop ordering severed.
- All 107 corpus entries migrated by experiments/migrate_usage_key.py, comments
and layout preserved.
MEASURED -- the connectivity model was ~4x too permissive. `none` is not
neutral: nothing is trimmed, so the graph may route THROUGH the room, and 34 of
52 codes had no class (Dental Surgery, Records Room, Utilities Closet all served
as corridors). Edges trimmed, prefix-inferred vs declared, 3 seeds each:
harbor-house 18 (9%) -> 79 (39%) inaccessible fails 0 -> 4
health-centre 12 (8%) -> 59 (40%) inaccessible fails 2 -> 3
maple-court 53 (17%) -> 123 (39%) inaccessible fails 1 -> 5
Re-baseline (seed 1, 20k, harbor): 58 fails (15h/43s) -> 61 (16h/45s), now
reporting 1-inaccessible-usable-space x2 plus level 0 and level 1 not connected.
The count rose because the objective got honest -- those failures were always
true of the layout and the old model could not see them. Every harbor number
before this was measured against a graph crediting routes through store
cupboards.
Sharpens §38.2: the objective pays x60-85 to delete circulation, and until now
the deleted corridors were not missed because storage stood in for them. With
that substitution gone, homemaker-py-2v1 is the remaining half -- and now
measurable, because the fails it should prevent actually fire.
350 passed (+5 new), same 7 pre-existing fixture failures, lint unchanged.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01MJ84Feep79Hhm3E4zZJmnB
2026-08-26 13:39:41 +00:00
|
|
|
_, gcpre = graph_mod.build_graphs_with_circ(root, fit.conf("door_width") or 1.2, failures.append, fit.usages())
|
2026-06-13 21:44:42 +01:00
|
|
|
gbpre = graph_mod.build_graphs(root, fit.conf("door_width") or 1.2)
|
|
|
|
|
failures.extend(graph_mod.check_adjacency(root, programme, gbpre, missing))
|
|
|
|
|
failures.extend(graph_mod.check_level_constraints(root, programme, missing))
|
|
|
|
|
failures.extend(graph_mod.check_vertical_connectivity(root, programme, missing))
|
|
|
|
|
|
|
|
|
|
dom.merge_divided(root)
|
|
|
|
|
geometry.clear_cache()
|
§39.7: access requirements become a declared `usage:` attribute (homemaker-py-sel)
Closes the second namespace sharing a first character with programme codes: the
usage prefixes b/t/l/k, under which a room silently inherited another room's
connectivity rules from its spelling.
usage is a plain, MANDATORY attribute of the space definition -- not a lookup
table. An interim design proposed a top-level usage_classes: table binding
author-coined names to behaviour; withdrawn, because an indirect name->behaviour
mapping living apart from the thing it describes is exactly the shape of the
prefix rule §39 exists to remove, it would be the only such table in a schema
where every other space property is a plain attribute, and the need it served
was already met -- "building specific" is about what a room is CALLED, and
name: is already free text.
Rule that settles it: a usage value exists iff the engine treats it differently
somewhere. Config selects among behaviours; it cannot invent them.
- programme.USAGES (living/kitchen/bedroom/toilet/utility/none) plus the
behaviour groupings PRIVATE_USAGES / PRIVATE_STRIPS / TOILET_STRIPS /
SOCIABLE_USAGES. Missing or unknown usage is a load error naming the code,
from BOTH parse paths.
- Code-level, never leaf-level: usage_of(leaf.type) is looked up fresh, so a
retype changes the class automatically. 51 sites assign leaf.type, and
share/share_type plus the r5a resurrection are the precedent for why
leaf-level attributes rot.
- graph.has_circulation takes the usage map and trims on declared class;
fitness.access and the public-access check likewise. fitness._t0 is DELETED --
no first-character type test remains anywhere in the codebase.
- utility is distinct from bedroom (same access requirements today) because it
is a different use and gives derive_interchange_classes an axis to relax on.
- A toilet now keeps its edge to a terminal room -- the Brand adjacency, which
the old b-before-t loop ordering severed.
- All 107 corpus entries migrated by experiments/migrate_usage_key.py, comments
and layout preserved.
MEASURED -- the connectivity model was ~4x too permissive. `none` is not
neutral: nothing is trimmed, so the graph may route THROUGH the room, and 34 of
52 codes had no class (Dental Surgery, Records Room, Utilities Closet all served
as corridors). Edges trimmed, prefix-inferred vs declared, 3 seeds each:
harbor-house 18 (9%) -> 79 (39%) inaccessible fails 0 -> 4
health-centre 12 (8%) -> 59 (40%) inaccessible fails 2 -> 3
maple-court 53 (17%) -> 123 (39%) inaccessible fails 1 -> 5
Re-baseline (seed 1, 20k, harbor): 58 fails (15h/43s) -> 61 (16h/45s), now
reporting 1-inaccessible-usable-space x2 plus level 0 and level 1 not connected.
The count rose because the objective got honest -- those failures were always
true of the layout and the old model could not see them. Every harbor number
before this was measured against a graph crediting routes through store
cupboards.
Sharpens §38.2: the objective pays x60-85 to delete circulation, and until now
the deleted corridors were not missed because storage stood in for them. With
that substitution gone, homemaker-py-2v1 is the remaining half -- and now
measurable, because the fails it should prevent actually fire.
350 passed (+5 new), same 7 pre-existing fixture failures, lint unchanged.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01MJ84Feep79Hhm3E4zZJmnB
2026-08-26 13:39:41 +00:00
|
|
|
_, gc = graph_mod.build_graphs_with_circ(root, fit.conf("door_width") or 1.2, failures.append, fit.usages())
|
2026-06-13 21:44:42 +01:00
|
|
|
gb = graph_mod.build_graphs(root, fit.conf("door_width") or 1.2)
|
|
|
|
|
|
|
|
|
|
cost_v = fit.plot_cost(root)
|
|
|
|
|
value = 0.0
|
|
|
|
|
lvls = dom.levels(root)
|
|
|
|
|
for li, lvl in enumerate(lvls):
|
|
|
|
|
se = fit.process_storey(
|
|
|
|
|
lvl, gb[li], li, failures.append,
|
|
|
|
|
graph_circ=gc, tracking=tracking, lvls=lvls, root=root,
|
|
|
|
|
)
|
|
|
|
|
cost_v += se.cost
|
|
|
|
|
value += se.value
|
|
|
|
|
|
|
|
|
|
bf = fit.evaluate_building(root, tracking)
|
|
|
|
|
value *= bf
|
|
|
|
|
value *= 0.5 ** len(failures)
|
|
|
|
|
score = value / cost_v if cost_v else 0.0
|
|
|
|
|
|
|
|
|
|
return score, frozenset(failures)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _oracle_result(src: Path):
|
|
|
|
|
"""Read cached oracle score and failure set (URB_NO_OCCLUSION=1)."""
|
2026-06-14 08:18:06 +01:00
|
|
|
from homemaker_layout.oracle import Score
|
2026-06-13 21:44:42 +01:00
|
|
|
|
|
|
|
|
score_file = Path(str(src) + ".score")
|
|
|
|
|
fails_file = Path(str(src) + ".fails")
|
|
|
|
|
if not score_file.exists():
|
|
|
|
|
pytest.skip(f"No cached oracle score for {src.name}")
|
|
|
|
|
oracle_score = float(score_file.read_text().strip())
|
|
|
|
|
oracle_fails = Score(
|
|
|
|
|
fitness=oracle_score,
|
|
|
|
|
fails=fails_file.read_text() if fails_file.exists() else "",
|
|
|
|
|
).fail_lines
|
|
|
|
|
return oracle_score, frozenset(oracle_fails)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize("src", sorted(CORPUS.glob("*.dom")), ids=lambda p: p.name)
|
|
|
|
|
def test_native_fitness_score_parity(src):
|
|
|
|
|
"""Native score matches oracle within 1e-4 relative tolerance."""
|
|
|
|
|
native_score, _ = _native_evaluate(src)
|
|
|
|
|
oracle_score, _ = _oracle_result(src)
|
|
|
|
|
assert math.isclose(native_score, oracle_score, rel_tol=1e-4, abs_tol=1e-15), (
|
|
|
|
|
f"{src.name}: native={native_score:.6e} oracle={oracle_score:.6e}"
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize("src", sorted(CORPUS.glob("*.dom")), ids=lambda p: p.name)
|
|
|
|
|
def test_native_fitness_fail_set_parity(src):
|
|
|
|
|
"""Native failure set matches oracle failure set exactly."""
|
|
|
|
|
_, native_fails = _native_evaluate(src)
|
|
|
|
|
_, oracle_fails = _oracle_result(src)
|
|
|
|
|
only_native = native_fails - oracle_fails
|
|
|
|
|
only_oracle = oracle_fails - native_fails
|
|
|
|
|
assert not only_native and not only_oracle, (
|
|
|
|
|
f"{src.name}: only_native={sorted(only_native)} only_oracle={sorted(only_oracle)}"
|
|
|
|
|
)
|