Add unit tests for geometry and fitness modules
26 tests for geometry (area, angles, aspect, boundary ids, centroid,
offset, etc.) and 35 tests for fitness (gaussian, config lookup,
quality terms, value rates, costs, stair helpers). Suite: 175 passed.
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-14 00:06:02 +01:00
|
|
|
|
"""Unit tests for fitness.py quality terms and helpers (oracle-free)."""
|
|
|
|
|
|
|
|
|
|
|
|
import pytest
|
|
|
|
|
|
|
2026-06-14 08:18:06 +01:00
|
|
|
|
from homemaker_layout import dom, geometry
|
|
|
|
|
|
from homemaker_layout.dom import Node
|
2026-06-18 22:33:29 +01:00
|
|
|
|
from homemaker_layout.fitness import (
|
|
|
|
|
|
CONF_DEFAULTS,
|
|
|
|
|
|
COST_DEFAULTS,
|
|
|
|
|
|
FAIL_THRESHOLD,
|
|
|
|
|
|
Fitness,
|
|
|
|
|
|
_leaf_grade,
|
homemaker-py-2g7.3: hard/soft fail tiering behind --use-tiers flag
Splits the flat outer-search comparator (-n_fails, fitness) into a tiered
(-n_hard, -n_soft, fitness) so search budget stops being spent polishing
SOFT shape fails (crinkliness/proportion/size/width/edge-too-long/
staircase-volume) while HARD structural fails (missing space, wrong/
required level, level/circulation/vertical connectivity, adjacency,
stairs, covered-outside, storey limits, public access) remain unfixed.
fitness.classify_fail_tier/tier_counts classify every fail string emitted
across fitness.py and graph.py, raising on anything unrecognised so new
fail sites must declare a tier. Validated against all real fail strings in
the checked-in corpus plus every fail-emission call site read from source.
driver.Individual gains n_hard/n_soft (populated from innerloop.Result.
fail_lines); search(use_tiers=...) swaps the comparator when set (default
off, so existing runs are unaffected — inner-loop 0.5^n cliff untouched).
evolve.py exposes --use-tiers / HOMEMAKER_USE_TIERS.
experiments/tier_ab_2g7_3.py runs the acceptance A/B (harbor+maple, 3
seeds, 20k evals) in the background; results pending.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LSwQwpEaHFBkeVSDDWd75S
2026-08-02 16:00:39 +01:00
|
|
|
|
classify_fail_tier,
|
2026-06-18 22:33:29 +01:00
|
|
|
|
gaussian,
|
homemaker-py-2g7.3: hard/soft fail tiering behind --use-tiers flag
Splits the flat outer-search comparator (-n_fails, fitness) into a tiered
(-n_hard, -n_soft, fitness) so search budget stops being spent polishing
SOFT shape fails (crinkliness/proportion/size/width/edge-too-long/
staircase-volume) while HARD structural fails (missing space, wrong/
required level, level/circulation/vertical connectivity, adjacency,
stairs, covered-outside, storey limits, public access) remain unfixed.
fitness.classify_fail_tier/tier_counts classify every fail string emitted
across fitness.py and graph.py, raising on anything unrecognised so new
fail sites must declare a tier. Validated against all real fail strings in
the checked-in corpus plus every fail-emission call site read from source.
driver.Individual gains n_hard/n_soft (populated from innerloop.Result.
fail_lines); search(use_tiers=...) swaps the comparator when set (default
off, so existing runs are unaffected — inner-loop 0.5^n cliff untouched).
evolve.py exposes --use-tiers / HOMEMAKER_USE_TIERS.
experiments/tier_ab_2g7_3.py runs the acceptance A/B (harbor+maple, 3
seeds, 20k evals) in the background; results pending.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LSwQwpEaHFBkeVSDDWd75S
2026-08-02 16:00:39 +01:00
|
|
|
|
tier_counts,
|
2026-06-18 22:33:29 +01:00
|
|
|
|
)
|
Add unit tests for geometry and fitness modules
26 tests for geometry (area, angles, aspect, boundary ids, centroid,
offset, etc.) and 35 tests for fitness (gaussian, config lookup,
quality terms, value rates, costs, stair helpers). Suite: 175 passed.
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-14 00:06:02 +01:00
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _leaf(type_: str, size: float = 4.0) -> Node:
|
|
|
|
|
|
"""Undivided level-root leaf with a square plot of side `size`."""
|
|
|
|
|
|
geometry.clear_cache()
|
|
|
|
|
|
return Node(
|
|
|
|
|
|
node=[[0.0, 0.0], [size, 0.0], [size, size], [0.0, size]],
|
|
|
|
|
|
type=type_,
|
|
|
|
|
|
)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# gaussian
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_gaussian_peak_returns_a():
|
|
|
|
|
|
assert gaussian(5.0, 1.0, 5.0, 1.0) == pytest.approx(1.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_gaussian_peak_scales_by_a():
|
|
|
|
|
|
assert gaussian(3.0, 2.5, 3.0, 1.0) == pytest.approx(2.5)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_gaussian_one_sigma_uses_truncated_e():
|
|
|
|
|
|
# Urb uses e=2.718281828, not math.e; at one sigma the factor is e^-0.5
|
|
|
|
|
|
e = 2.718281828
|
|
|
|
|
|
expected = e ** -0.5
|
|
|
|
|
|
assert gaussian(6.0, 1.0, 5.0, 1.0) == pytest.approx(expected, rel=1e-9)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_gaussian_symmetry():
|
|
|
|
|
|
assert gaussian(4.0, 1.0, 5.0, 1.0) == pytest.approx(gaussian(6.0, 1.0, 5.0, 1.0))
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# Fitness.conf / cost
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_conf_falls_back_to_defaults():
|
|
|
|
|
|
assert Fitness().conf("value_inside") == CONF_DEFAULTS["value_inside"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_conf_override_wins():
|
|
|
|
|
|
assert Fitness(conf={"value_inside": 999.0}).conf("value_inside") == 999.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_conf_unknown_key_returns_none():
|
|
|
|
|
|
assert Fitness().conf("no_such_key") is None
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_cost_falls_back_to_defaults():
|
|
|
|
|
|
assert Fitness().cost("inside") == COST_DEFAULTS["inside"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_cost_override_wins():
|
|
|
|
|
|
assert Fitness(cost={"inside": 42.0}).cost("inside") == 42.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_cost_unknown_key_returns_zero():
|
|
|
|
|
|
assert Fitness().cost("no_such_key") == 0.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# get_space_params lookup chain
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_get_space_params_circulation_size():
|
|
|
|
|
|
assert Fitness().get_space_params("C", "size") == CONF_DEFAULTS["size_circulation"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_get_space_params_outside_width():
|
|
|
|
|
|
assert Fitness().get_space_params("O", "width") == CONF_DEFAULTS["width_outside"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_get_space_params_sahn_proportion():
|
|
|
|
|
|
assert Fitness().get_space_params("S", "proportion") == CONF_DEFAULTS["proportion_outside"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_get_space_params_inside_falls_back_to_inside_defaults():
|
|
|
|
|
|
assert Fitness().get_space_params("k1", "proportion") == CONF_DEFAULTS["proportion_inside"]
|
|
|
|
|
|
assert Fitness().get_space_params("k1", "size") == CONF_DEFAULTS["size_inside"]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_get_space_params_named_space_overrides_default():
|
|
|
|
|
|
f = Fitness(conf={"spaces": {"k1": {"size": [20.0, 4.0]}}})
|
|
|
|
|
|
assert f.get_space_params("k1", "size") == [20.0, 4.0]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# quality_proportion
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_quality_proportion_square_inside_returns_one():
|
|
|
|
|
|
# aspect=1.0 < proportion_inside[0]=1.5 → 1.0
|
|
|
|
|
|
assert Fitness().quality_proportion(_leaf("k1")) == pytest.approx(1.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_quality_proportion_square_outside_returns_one():
|
|
|
|
|
|
# aspect=1.0 < proportion_outside[0]=1.5 → 1.0
|
|
|
|
|
|
assert Fitness().quality_proportion(_leaf("O")) == pytest.approx(1.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_quality_proportion_square_circulation_returns_one():
|
|
|
|
|
|
assert Fitness().quality_proportion(_leaf("C")) == pytest.approx(1.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# quality_size
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_quality_size_outside_always_one():
|
|
|
|
|
|
assert Fitness().quality_size(_leaf("O")) == 1.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_quality_size_sahn_always_one():
|
|
|
|
|
|
assert Fitness().quality_size(_leaf("S")) == 1.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_quality_size_inside_at_peak():
|
|
|
|
|
|
# size_inside=[16.0,3.5]; leaf is 4×4=16 m² → gaussian at peak → 1.0
|
|
|
|
|
|
leaf = _leaf("k1", size=4.0)
|
|
|
|
|
|
assert geometry.area(leaf) == pytest.approx(16.0)
|
|
|
|
|
|
assert Fitness().quality_size(leaf) == pytest.approx(1.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_quality_size_circulation_at_peak():
|
|
|
|
|
|
# size_circulation=[0.0,14.0]; peak at 0, gaussian(area,1,0,14) → always <1 for area>0
|
|
|
|
|
|
# Just verify it returns a value in [0,1]
|
|
|
|
|
|
f = Fitness().quality_size(_leaf("C", size=4.0))
|
|
|
|
|
|
assert 0.0 < f <= 1.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# quality_width
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_quality_width_wide_inside_returns_one():
|
|
|
|
|
|
# width_inside=[4.0,1.0]; 10m side > 4.0 → 1.0
|
|
|
|
|
|
assert Fitness().quality_width(_leaf("k1", size=10.0)) == pytest.approx(1.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_quality_width_wide_circulation_returns_one():
|
|
|
|
|
|
# width_circulation=[2.4,0.2]; 10m > 2.4 → 1.0
|
|
|
|
|
|
assert Fitness().quality_width(_leaf("C", size=10.0)) == pytest.approx(1.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_quality_width_wide_outside_ground_uses_gaussian():
|
|
|
|
|
|
# outside at level 0 falls through to gaussian; 10m > width_outside[0]=3.0 → 1.0
|
|
|
|
|
|
assert Fitness().quality_width(_leaf("O", size=10.0)) == pytest.approx(1.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# quality_perpendicular
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_quality_perpendicular_rectangle_near_one():
|
|
|
|
|
|
# All four corners of the square are pi/2; perpendicular formula gives ≈1
|
|
|
|
|
|
leaf = _leaf("k1", size=4.0)
|
|
|
|
|
|
result = Fitness().quality_perpendicular(leaf)
|
|
|
|
|
|
assert result == pytest.approx(1.0, abs=1e-6)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# value_rate
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_value_rate_outside_ground():
|
|
|
|
|
|
leaf = _leaf("O")
|
|
|
|
|
|
assert dom.level_of(leaf) == 0
|
|
|
|
|
|
assert Fitness().value_rate(leaf) == pytest.approx(CONF_DEFAULTS["value_outside"])
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_value_rate_circulation():
|
|
|
|
|
|
assert Fitness().value_rate(_leaf("C")) == pytest.approx(CONF_DEFAULTS["value_circulation"])
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_value_rate_inside():
|
|
|
|
|
|
assert Fitness().value_rate(_leaf("k1")) == pytest.approx(CONF_DEFAULTS["value_inside"])
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# leaf_cost
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_leaf_cost_outside_bare():
|
|
|
|
|
|
# not covered, not supported → outside rate × area
|
|
|
|
|
|
leaf = _leaf("O", size=4.0) # area = 16.0
|
|
|
|
|
|
assert Fitness().leaf_cost(leaf) == pytest.approx(COST_DEFAULTS["outside"] * 16.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_leaf_cost_inside():
|
|
|
|
|
|
leaf = _leaf("k1", size=4.0)
|
|
|
|
|
|
assert Fitness().leaf_cost(leaf) == pytest.approx(COST_DEFAULTS["inside"] * 16.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
hph/§13.8: share-aware edge-too-long cap — shared leaves no longer penalised for aggregate wall length
§13.7 flagged edge-too-long as harbor's top fail class. Dissection showed the
bulk are a leaf-sharing REPRESENTATION ARTIFACT: a share=k leaf aggregates k
same-code rooms, so its walls run ~k× the flat 8 m cap purely for being big —
the same §13.3 leak (size/missing relaxed for shared leaves) on the wall measure,
since edge_cost/outside_edge_cost ignored leaf.share.
Fix: Fitness._edge_cap(*leaves) scales the 8 m cap by the largest type-guarded
leaf_share among adjoining leaves, mirroring quality_size's k×target; non-shared
leaves keep the flat cap so genuine narrow/oversize pathologies stay flagged.
Gated behind a share_edge_cap config knob (SHAREEDGE env), default OFF so the
§13.x controls reproduce.
A/B (full Phase-8 stack, staged, 20k evals, seeds 0/1/2): control reproduces
§13.7 (maple 80.3 exact, harbor 34.7≈34.0); share-aware arm maple 80.3→74.0
(−7.9%), harbor 34.7→31.0 (−10.6%), zero regressions across 6 seeds. Positive
and monotone-harmless (only ever removes a false-positive fail). Verdict:
recommend default-ON; follow-up issue flips the default + rebaselines the floor.
Tests: 6 new unit tests for _edge_cap (221 pass).
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01JygRv4n2dcyDQqMiDRe7TN
2026-06-28 21:24:51 +01:00
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# Share-aware edge-too-long cap (hph §13.7)
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _shared_leaf(type_: str = "k1", k: int = 3) -> Node:
|
|
|
|
|
|
leaf = _leaf(type_)
|
|
|
|
|
|
leaf.share = k
|
|
|
|
|
|
leaf.share_type = type_
|
|
|
|
|
|
return leaf
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_edge_cap_flat_by_default():
|
|
|
|
|
|
# no leaf_sharing → flat 8 m regardless of any share stamp
|
|
|
|
|
|
fit = Fitness()
|
|
|
|
|
|
assert fit._edge_cap(_shared_leaf(k=3)) == pytest.approx(8.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_edge_cap_flat_when_lever_off_even_with_sharing():
|
2026-06-28 21:38:53 +01:00
|
|
|
|
# leaf_sharing on but the hph lever explicitly off → still flat (control arm).
|
|
|
|
|
|
# Post-§13.8 the lever defaults ON under sharing, so the control must pin it.
|
|
|
|
|
|
fit = Fitness(conf={"leaf_sharing": True, "share_edge_cap": False})
|
hph/§13.8: share-aware edge-too-long cap — shared leaves no longer penalised for aggregate wall length
§13.7 flagged edge-too-long as harbor's top fail class. Dissection showed the
bulk are a leaf-sharing REPRESENTATION ARTIFACT: a share=k leaf aggregates k
same-code rooms, so its walls run ~k× the flat 8 m cap purely for being big —
the same §13.3 leak (size/missing relaxed for shared leaves) on the wall measure,
since edge_cost/outside_edge_cost ignored leaf.share.
Fix: Fitness._edge_cap(*leaves) scales the 8 m cap by the largest type-guarded
leaf_share among adjoining leaves, mirroring quality_size's k×target; non-shared
leaves keep the flat cap so genuine narrow/oversize pathologies stay flagged.
Gated behind a share_edge_cap config knob (SHAREEDGE env), default OFF so the
§13.x controls reproduce.
A/B (full Phase-8 stack, staged, 20k evals, seeds 0/1/2): control reproduces
§13.7 (maple 80.3 exact, harbor 34.7≈34.0); share-aware arm maple 80.3→74.0
(−7.9%), harbor 34.7→31.0 (−10.6%), zero regressions across 6 seeds. Positive
and monotone-harmless (only ever removes a false-positive fail). Verdict:
recommend default-ON; follow-up issue flips the default + rebaselines the floor.
Tests: 6 new unit tests for _edge_cap (221 pass).
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01JygRv4n2dcyDQqMiDRe7TN
2026-06-28 21:24:51 +01:00
|
|
|
|
assert fit._edge_cap(_shared_leaf(k=3)) == pytest.approx(8.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_edge_cap_scales_by_share_when_lever_on():
|
|
|
|
|
|
fit = Fitness(conf={"leaf_sharing": True, "share_edge_cap": True})
|
|
|
|
|
|
assert fit._edge_cap(_shared_leaf(k=3)) == pytest.approx(24.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
2026-06-28 21:38:53 +01:00
|
|
|
|
def test_edge_cap_defaults_on_under_leaf_sharing():
|
|
|
|
|
|
# §13.8 default flip: leaf_sharing on, lever unset → cap scales by share
|
|
|
|
|
|
fit = Fitness(conf={"leaf_sharing": True})
|
|
|
|
|
|
assert fit._edge_cap(_shared_leaf(k=3)) == pytest.approx(24.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
hph/§13.8: share-aware edge-too-long cap — shared leaves no longer penalised for aggregate wall length
§13.7 flagged edge-too-long as harbor's top fail class. Dissection showed the
bulk are a leaf-sharing REPRESENTATION ARTIFACT: a share=k leaf aggregates k
same-code rooms, so its walls run ~k× the flat 8 m cap purely for being big —
the same §13.3 leak (size/missing relaxed for shared leaves) on the wall measure,
since edge_cost/outside_edge_cost ignored leaf.share.
Fix: Fitness._edge_cap(*leaves) scales the 8 m cap by the largest type-guarded
leaf_share among adjoining leaves, mirroring quality_size's k×target; non-shared
leaves keep the flat cap so genuine narrow/oversize pathologies stay flagged.
Gated behind a share_edge_cap config knob (SHAREEDGE env), default OFF so the
§13.x controls reproduce.
A/B (full Phase-8 stack, staged, 20k evals, seeds 0/1/2): control reproduces
§13.7 (maple 80.3 exact, harbor 34.7≈34.0); share-aware arm maple 80.3→74.0
(−7.9%), harbor 34.7→31.0 (−10.6%), zero regressions across 6 seeds. Positive
and monotone-harmless (only ever removes a false-positive fail). Verdict:
recommend default-ON; follow-up issue flips the default + rebaselines the floor.
Tests: 6 new unit tests for _edge_cap (221 pass).
Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01JygRv4n2dcyDQqMiDRe7TN
2026-06-28 21:24:51 +01:00
|
|
|
|
def test_edge_cap_unshared_leaf_keeps_flat_cap():
|
|
|
|
|
|
# a non-shared leaf (the narrow-sliver pathology) is never relaxed
|
|
|
|
|
|
fit = Fitness(conf={"leaf_sharing": True, "share_edge_cap": True})
|
|
|
|
|
|
assert fit._edge_cap(_leaf("k1")) == pytest.approx(8.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_edge_cap_stale_share_type_ignored():
|
|
|
|
|
|
# retyped leaf whose stamp no longer matches type → share invalid → flat
|
|
|
|
|
|
fit = Fitness(conf={"leaf_sharing": True, "share_edge_cap": True})
|
|
|
|
|
|
leaf = _shared_leaf("k1", k=3)
|
|
|
|
|
|
leaf.type = "b1" # retyped; share_type still "k1"
|
|
|
|
|
|
assert fit._edge_cap(leaf) == pytest.approx(8.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_edge_cap_uses_largest_share_among_adjoining_leaves():
|
|
|
|
|
|
# an interior wall takes the max share of the two leaves it separates
|
|
|
|
|
|
fit = Fitness(conf={"leaf_sharing": True, "share_edge_cap": True})
|
|
|
|
|
|
cap = fit._edge_cap(_leaf("k1"), _shared_leaf("b1", k=2))
|
|
|
|
|
|
assert cap == pytest.approx(16.0)
|
|
|
|
|
|
|
|
|
|
|
|
|
Add unit tests for geometry and fitness modules
26 tests for geometry (area, angles, aspect, boundary ids, centroid,
offset, etc.) and 35 tests for fitness (gaussian, config lookup,
quality terms, value rates, costs, stair helpers). Suite: 175 passed.
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-14 00:06:02 +01:00
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# Stair helpers
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_risers_number_exact_division():
|
|
|
|
|
|
# 2.0 / 0.25 = 8.0 exactly → returns 8
|
|
|
|
|
|
assert Fitness._risers_number(2.0, 0.25) == 8
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_risers_number_rounds_up():
|
|
|
|
|
|
# 3.0 / 0.19 ≈ 15.789 → rounds up to 16
|
|
|
|
|
|
assert Fitness._risers_number(3.0, 0.19) == 16
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_ideal_going_clamps_to_minimum():
|
|
|
|
|
|
# riser=0.25 → going=0.125 < 0.22 → clamp
|
|
|
|
|
|
assert Fitness._ideal_going(0.25) == 0.22
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_ideal_going_above_minimum():
|
|
|
|
|
|
# riser=0.15 → going=0.325 > 0.22; result should be in valid range
|
|
|
|
|
|
result = Fitness._ideal_going(0.15)
|
|
|
|
|
|
assert result >= 0.22
|
|
|
|
|
|
assert result <= 0.625
|
2026-06-18 22:33:29 +01:00
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# Graded high-fail objective (§11.4)
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_leaf_grade_no_failing_factors_is_zero():
|
|
|
|
|
|
# All factors above FAIL_THRESHOLD → no proximity credit.
|
|
|
|
|
|
assert _leaf_grade({"size": 0.9, "width": 1.0, "access": 1.0}) == 0.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_leaf_grade_credits_only_failing_factors():
|
|
|
|
|
|
# Only size fails (0.05 < 0.1); credit = 0.05 / 0.1 = 0.5.
|
|
|
|
|
|
g = _leaf_grade({"size": 0.05, "width": 0.5, "proportion": 1.0})
|
|
|
|
|
|
assert g == pytest.approx(0.05 / FAIL_THRESHOLD)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_leaf_grade_monotone_in_proximity():
|
|
|
|
|
|
# A failing factor closer to the threshold scores higher (better).
|
|
|
|
|
|
deep = _leaf_grade({"size": 0.01})
|
|
|
|
|
|
shallow = _leaf_grade({"size": 0.09})
|
|
|
|
|
|
assert shallow > deep
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_leaf_grade_sums_over_failing_factors():
|
|
|
|
|
|
g = _leaf_grade({"size": 0.04, "width": 0.06, "access": 1.0})
|
|
|
|
|
|
assert g == pytest.approx((0.04 + 0.06) / FAIL_THRESHOLD)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_leaf_grade_ignores_non_graded_keys():
|
|
|
|
|
|
# daylight is pinned and never a graded factor even if below threshold.
|
|
|
|
|
|
assert _leaf_grade({"daylight": 0.0}) == 0.0
|
2026-06-28 22:04:35 +01:00
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# load_config overrides (homemaker-py-x3b)
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_load_config_overrides_merge_last(tmp_path):
|
|
|
|
|
|
# The CLI/driver injects run-level knobs (leaf_sharing) without editing any
|
|
|
|
|
|
# on-disk patterns.config, so §13.3 example programmes stay reproducible.
|
|
|
|
|
|
import yaml
|
|
|
|
|
|
|
|
|
|
|
|
from homemaker_layout.fitness import load_config
|
|
|
|
|
|
|
|
|
|
|
|
(tmp_path / "patterns.config").write_text(
|
|
|
|
|
|
yaml.safe_dump({"spaces": {"b": {"size": [12.0, 1.0]}}}))
|
|
|
|
|
|
|
|
|
|
|
|
conf, _ = load_config(tmp_path)
|
|
|
|
|
|
assert "leaf_sharing" not in conf # absent on disk
|
|
|
|
|
|
|
|
|
|
|
|
conf2, _ = load_config(tmp_path, overrides={"leaf_sharing": True})
|
|
|
|
|
|
assert conf2["leaf_sharing"] is True
|
|
|
|
|
|
assert conf2["spaces"]["b"] == {"size": [12.0, 1.0]} # disk content preserved
|
|
|
|
|
|
|
|
|
|
|
|
# None / empty overrides are a no-op (default-OFF parity).
|
|
|
|
|
|
assert "leaf_sharing" not in load_config(tmp_path, overrides=None)[0]
|
|
|
|
|
|
assert "leaf_sharing" not in load_config(tmp_path, overrides={})[0]
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_programme_parses_per_code_share(tmp_path):
|
|
|
|
|
|
# homemaker-py-x3b: SpaceReq carries the optional per-code 'share' grain and a
|
|
|
|
|
|
# has_share flag distinguishing an explicit share:1 (opt out) from the default.
|
|
|
|
|
|
import yaml
|
|
|
|
|
|
|
|
|
|
|
|
from homemaker_layout.programme import load_programme
|
|
|
|
|
|
|
|
|
|
|
|
p = tmp_path / "patterns.config"
|
|
|
|
|
|
p.write_text(yaml.safe_dump({"spaces": {
|
|
|
|
|
|
"b": {"size": [12.0, 1.0], "share": 3},
|
|
|
|
|
|
"k": {"size": [20.0, 1.0]}, # no share key
|
|
|
|
|
|
}}))
|
|
|
|
|
|
reqs = load_programme(str(p))
|
|
|
|
|
|
assert reqs["b"].share == 3 and reqs["b"].has_share is True
|
|
|
|
|
|
assert reqs["k"].share == 1 and reqs["k"].has_share is False
|
homemaker-py-2g7.3: hard/soft fail tiering behind --use-tiers flag
Splits the flat outer-search comparator (-n_fails, fitness) into a tiered
(-n_hard, -n_soft, fitness) so search budget stops being spent polishing
SOFT shape fails (crinkliness/proportion/size/width/edge-too-long/
staircase-volume) while HARD structural fails (missing space, wrong/
required level, level/circulation/vertical connectivity, adjacency,
stairs, covered-outside, storey limits, public access) remain unfixed.
fitness.classify_fail_tier/tier_counts classify every fail string emitted
across fitness.py and graph.py, raising on anything unrecognised so new
fail sites must declare a tier. Validated against all real fail strings in
the checked-in corpus plus every fail-emission call site read from source.
driver.Individual gains n_hard/n_soft (populated from innerloop.Result.
fail_lines); search(use_tiers=...) swaps the comparator when set (default
off, so existing runs are unaffected — inner-loop 0.5^n cliff untouched).
evolve.py exposes --use-tiers / HOMEMAKER_USE_TIERS.
experiments/tier_ab_2g7_3.py runs the acceptance A/B (harbor+maple, 3
seeds, 20k evals) in the background; results pending.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01LSwQwpEaHFBkeVSDDWd75S
2026-08-02 16:00:39 +01:00
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# Hard/soft fail tiering (homemaker-py-2g7.3)
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize("fail_str", [
|
|
|
|
|
|
"missing required space: la1",
|
|
|
|
|
|
"missing required space: la1 (critical)",
|
|
|
|
|
|
"too many spaces: k (found 3, expected 2)",
|
|
|
|
|
|
"missing ef1: would need size check",
|
|
|
|
|
|
"missing ef1: would need width check",
|
|
|
|
|
|
"missing ef1: would need proportion check",
|
|
|
|
|
|
"missing m: would need adjacency to c",
|
|
|
|
|
|
"missing r: would need to be on level 1",
|
|
|
|
|
|
"missing t1: would need connection to c below",
|
|
|
|
|
|
"0/lr (cr1) not adjacent to c",
|
|
|
|
|
|
"li1 on wrong level (level 0, expected 1)",
|
|
|
|
|
|
"t1 not connected to c below",
|
|
|
|
|
|
"level 0 not connected",
|
|
|
|
|
|
"0 inaccessible usable space",
|
|
|
|
|
|
"level 0 no outside space",
|
|
|
|
|
|
"0/lr unsupported covered outside",
|
|
|
|
|
|
"0/lr covered outside above ground",
|
|
|
|
|
|
"too few stairs (0, min 1)",
|
|
|
|
|
|
"too many stairs (2, max 1)",
|
|
|
|
|
|
"storey limit",
|
|
|
|
|
|
"storey minimum",
|
|
|
|
|
|
"no outside public access",
|
|
|
|
|
|
])
|
|
|
|
|
|
def test_classify_fail_tier_hard(fail_str):
|
|
|
|
|
|
assert classify_fail_tier(fail_str) == "hard"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
@pytest.mark.parametrize("fail_str", [
|
|
|
|
|
|
"0/lr perpendicular",
|
|
|
|
|
|
"0/lr proportion",
|
|
|
|
|
|
"0/lr size",
|
|
|
|
|
|
"0/lr width",
|
|
|
|
|
|
"0/lr crinkliness",
|
|
|
|
|
|
"0/lr access",
|
|
|
|
|
|
"0/lr lrr edge too long",
|
|
|
|
|
|
"lr outside edge too long",
|
|
|
|
|
|
"staircase volume",
|
|
|
|
|
|
])
|
|
|
|
|
|
def test_classify_fail_tier_soft(fail_str):
|
|
|
|
|
|
assert classify_fail_tier(fail_str) == "soft"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_classify_fail_tier_missing_cascade_is_hard_not_soft():
|
|
|
|
|
|
# "missing X: would need size check" contains the SOFT " size" substring,
|
|
|
|
|
|
# but is a consequence of a HARD missing-space fail, not a shape defect —
|
|
|
|
|
|
# the HARD markers must be checked first (fitness.py ordering).
|
|
|
|
|
|
assert classify_fail_tier("missing m#2: would need size check") == "hard"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_classify_fail_tier_unknown_raises():
|
|
|
|
|
|
with pytest.raises(ValueError):
|
|
|
|
|
|
classify_fail_tier("some brand new fail string nobody tiered yet")
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_tier_counts_splits_hard_and_soft():
|
|
|
|
|
|
fails = ("level 0 not connected", "0/lr proportion", "0/lr crinkliness",
|
|
|
|
|
|
"missing required space: k1")
|
|
|
|
|
|
assert tier_counts(fails) == (2, 2)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_tier_counts_empty():
|
|
|
|
|
|
assert tier_counts(()) == (0, 0)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_classify_fail_tier_covers_full_corpus():
|
|
|
|
|
|
"""Regression guard: every fail string ever emitted into a checked-in
|
|
|
|
|
|
native (non-YAML) .fails file must still classify without error."""
|
|
|
|
|
|
import glob
|
|
|
|
|
|
from pathlib import Path
|
|
|
|
|
|
|
|
|
|
|
|
repo_root = Path(__file__).resolve().parent.parent
|
|
|
|
|
|
checked = 0
|
|
|
|
|
|
for path in glob.glob(str(repo_root / "examples" / "**" / "*.fails"), recursive=True):
|
|
|
|
|
|
with open(path) as f:
|
|
|
|
|
|
first = f.readline()
|
|
|
|
|
|
if first.startswith("---"):
|
|
|
|
|
|
continue # legacy Perl-oracle YAML .fails, not this evaluator's output
|
|
|
|
|
|
lines = [first.rstrip("\n")] + [ln.rstrip("\n") for ln in f]
|
|
|
|
|
|
for line in lines:
|
|
|
|
|
|
if not line:
|
|
|
|
|
|
continue
|
|
|
|
|
|
classify_fail_tier(line) # raises on failure
|
|
|
|
|
|
checked += 1
|
|
|
|
|
|
assert checked > 0
|
§38.2 refinement: connectivity is under-priced ~3x, not just a crinkliness bug
Follow-up measurement corrects the first draft of §38 in two ways.
1. Harbor-house's floor is 15 fails (evolved-3M-nols-3, 1.7M evals), not the
30-40 I quoted from §13.11's 20k-budget runs. Frontage deficit predicts the
COST of solving, not impossibility: ~150x budget gap between a
frontage-short and a frontage-surplus programme. Table corrected.
2. Zero-exposure is only half the mechanism, and not the dominant half.
Splitting the deletion test by lit vs buried shows a WELL-DAYLIT corridor
(q_crink=0.736) is still worth x4.06 to delete. Cause: value_circulation=50
vs value_inside=300, so merging corridor into room is a flat x6 gain, while
'level N not connected' costs only x0.5. Break-even needs 0.5^k < 50/300,
i.e. k > 2.58 -- severing must cost at least 3 fails and costs 1. Net x3.0
predicted, x4.06 measured. The objective is net-positive on severing the
spine even when the circulation is perfectly lit, which explains why both
'level N not connected' fails survive in the best layout after 1.7M evals.
Adds fitness.quality_uncrinkliness crinkliness_mode (EXPERIMENTAL, default
"urb" = stock hard 0.0, byte-identical: 336 passed vs 331 before, same 7
pre-existing fixture failures). A/B harness ab_crinkliness_mode_ssz.py shows
none of the three modes removes the incentive, and the lit column is 3/8 under
every mode including stock -- clean isolation of the two mechanisms.
Filed homemaker-py-2v1 (P0) for the pricing fix; ssz/hxi now depend on it.
Acceptance test recorded up front: harbor must reach 15 fails in materially
fewer than 1.7M evals AND without either not-connected fail.
Co-Authored-By: Claude Opus 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01MJ84Feep79Hhm3E4zZJmnB
2026-08-26 07:40:37 +00:00
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
# homemaker-py-ssz / DESIGN.md §38.1 — crinkliness_mode (EXPERIMENTAL)
|
|
|
|
|
|
# --------------------------------------------------------------------------- #
|
|
|
|
|
|
class _StubCrink(Fitness):
|
|
|
|
|
|
"""Fitness with ``crinkliness`` stubbed, so the modes can be tested without
|
|
|
|
|
|
building a real tree/graph (the value under test is the branch, not the
|
|
|
|
|
|
geometry)."""
|
|
|
|
|
|
|
|
|
|
|
|
_stub = 0.0
|
|
|
|
|
|
|
|
|
|
|
|
def crinkliness(self, leaf, G, groups): # noqa: D102 - test stub
|
|
|
|
|
|
return self._stub
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def _stub_fit(mode=None, stub=0.0, type_="t1"):
|
|
|
|
|
|
conf = dict(CONF_DEFAULTS)
|
|
|
|
|
|
if mode is not None:
|
|
|
|
|
|
conf["crinkliness_mode"] = mode
|
|
|
|
|
|
f = _StubCrink(conf, dict(COST_DEFAULTS))
|
|
|
|
|
|
f._stub = stub
|
|
|
|
|
|
return f, _leaf(type_)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_crinkliness_mode_defaults_to_urb_and_reproduces_hard_zero():
|
|
|
|
|
|
"""Default must be byte-identical to stock Urb: buried leaf -> exactly 0.0."""
|
|
|
|
|
|
f, leaf = _stub_fit()
|
|
|
|
|
|
assert f._crinkliness_mode == "urb"
|
|
|
|
|
|
assert f.quality_uncrinkliness(leaf, None, {}) == 0.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_crinkliness_floor_restores_gradient_but_keeps_the_failure():
|
|
|
|
|
|
"""The floor must stay BELOW FAIL_THRESHOLD: it restores a value gradient
|
|
|
|
|
|
without silently deleting a whole fail category."""
|
|
|
|
|
|
f, leaf = _stub_fit("floor")
|
|
|
|
|
|
q = f.quality_uncrinkliness(leaf, None, {})
|
|
|
|
|
|
assert q > 0.0, "buried leaf should no longer be worth exactly nothing"
|
|
|
|
|
|
assert q < FAIL_THRESHOLD, "buried leaf must still emit its crinkliness fail"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_crinkliness_compact_ok_clips_on_the_compact_side_only():
|
|
|
|
|
|
"""Being more compact than target is not a defect; being over-exposed is."""
|
|
|
|
|
|
target = CONF_DEFAULTS["uncrinkliness"][0]
|
|
|
|
|
|
# 1/crink > target => more compact than target => clipped to 1.0
|
|
|
|
|
|
f, leaf = _stub_fit("compact_ok", stub=1.0 / (target * 2))
|
|
|
|
|
|
assert f.quality_uncrinkliness(leaf, None, {}) == 1.0
|
|
|
|
|
|
# 1/crink < target => over-exposed => still decays
|
|
|
|
|
|
f, leaf = _stub_fit("compact_ok", stub=1.0 / (target / 2))
|
|
|
|
|
|
assert f.quality_uncrinkliness(leaf, None, {}) < 1.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_crinkliness_exempt_circulation_only_exempts_circulation():
|
|
|
|
|
|
f, circ = _stub_fit("exempt_circulation", type_="C")
|
|
|
|
|
|
assert f.quality_uncrinkliness(circ, None, {}) == 1.0
|
|
|
|
|
|
f, room = _stub_fit("exempt_circulation", type_="t1")
|
|
|
|
|
|
assert f.quality_uncrinkliness(room, None, {}) == 0.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
def test_crinkliness_mode_unknown_raises():
|
|
|
|
|
|
with pytest.raises(ValueError, match="crinkliness_mode"):
|
|
|
|
|
|
_stub_fit("nonsense")
|