From 929be5b144f4cf6536d619d0a9e3364ae8be2398 Mon Sep 17 00:00:00 2001 From: Bruno Postle Date: Sat, 1 Aug 2026 21:08:32 +0100 Subject: [PATCH] homemaker-py-iio: fix stale leaf-share leak in collapse_global's probe valuation _collapse_value and _usage_quality temporarily overwrite leaf.type to probe a hypothetical candidate code, but graph.leaf_share reads that overwritten type against leaf.share_type -- so a stale share (left over from a code the leaf was since retyped away from) spuriously reactivates whenever the probed candidate happens to equal the old share_type, skewing the Hungarian assignment's cell value for that (leaf, code) pair. dom.dump/dom.load drops such stale metadata on reload (dom._emit only serialises share when share_type==type), so a live search tree carrying it and its dump/reload round trip fed different values into the same collapse_global call and landed on different optimal matchings. Fix: neutralise share_type during the probe whenever the candidate differs from the leaf's real current type, restoring it in the finally block. The leaf's own current type still legitimately carries a live share. Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01R8agJBT2ZpmF3ErW7wi2wY --- .beads/issues.jsonl | 38 ++++++++-------- src/homemaker_layout/fitness.py | 22 +++++++++ tests/test_collapse_global.py | 81 +++++++++++++++++++++++++++++++++ 3 files changed, 122 insertions(+), 19 deletions(-) diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index 0b5be71..2609edf 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -22,7 +22,7 @@ {"id":"homemaker-py-1p0","title":"Geometry inner loop: full-objective equal-offset ratio optimiser","description":"DESIGN.md §5.1, §7 Phase 1. Productionise experiments/optimize_fullfitness.py into homemaker: optimise(topology, x0=None) -\u003e (geometry, fitness). DOF = equal-offset division ratios of free branches (solver.free_branches, lowest-storey cut ownership), clipped to [eps, 1-eps]. Objective = full oracle fitness (never a proxy — §4.2 falsified). Must support warm-start x0 (§5.6) and a population/batch evaluation mode so each iteration scores via one batched oracle call (§4.6).","acceptance_criteria":"Reproduces or exceeds §4.5 gains (x1.24–x1.67, no new failures) on 2f45907, candidate-002, c964435; works as a library call on any corpus .dom","status":"closed","priority":1,"issue_type":"feature","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-06-11T23:36:58Z","created_by":"Bruno Postle","updated_at":"2026-06-12T08:46:31Z","started_at":"2026-06-12T00:14:19Z","closed_at":"2026-06-12T08:46:31Z","close_reason":"innerloop.optimise() lands: batched CMA-ES sigma ladder (0.05/0.15, IPOP popsize doubling, deterministic seeding) over equal-offset free-branch ratios vs full oracle fitness; warm-start x0 supported. Acceptance vs unprojected originals: x1.65/x1.66/x1.58 against bars x1.24/x1.67/x1.59, no new failures, 46 oracle calls vs NM's 200. Two near-bar results accepted as reproduced-within-noise (1% tol) — draw spread brackets the single-NM-draw bars; approved by Bruno 2026-06-12. Gotchas: equal-offset projection of legacy unequal cuts loses fitness/adds failures (midpoint projection used); pycma seed=0 means clock-seeded.","dependencies":[{"issue_id":"homemaker-py-1p0","depends_on_id":"homemaker-py-av5","type":"blocks","created_at":"2026-06-12T00:39:33Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":3,"comment_count":0} {"id":"homemaker-py-8cs","title":"Experiment: warm-vs-cold start of inner loop (Lamarckian inheritance)","description":"DESIGN.md §5.6, §4.6. Warm-starting a child topology's inner loop from the parent's optimised ratios is the main lever for cutting per-topology cost (~3 min/topology cold). Apply single topology mutations to optimised corpus designs, re-optimise warm (surviving cuts keep values, new cuts get heuristic defaults) vs cold, compare oracle-call counts to convergence at equal final fitness.","acceptance_criteria":"Speedup factor measured across \u003e=10 mutated topologies; decision recorded (expect order-of-magnitude; if \u003c2x, revisit §4.6 Phase-2 scoping)","notes":"Experiment script committed (experiments/warm_vs_cold.py, 1cc86c8) and machinery validated oracle-free; one mutated child scored through the oracle OK. Waiting on homemaker-py-gp2 reference run to finish, then execute under URB_NO_OCCLUSION=1 (3 parents x 400 evals + 12 children x 2 x 200 evals, ~1.5-2 h oracle time). Default budgets: parent 400, child 200; target = evals to 95% of best final.","status":"closed","priority":1,"issue_type":"task","owner":"bruno@postle.net","created_at":"2026-06-11T23:36:58Z","created_by":"Bruno Postle","updated_at":"2026-06-12T11:44:45Z","closed_at":"2026-06-12T11:44:45Z","close_reason":"Measured (URB_NO_OCCLUSION=1, parent budget 400, child 200, 12 single mutations across 3 designs): cold start reached 95% of warm final in 0/12 cases within budget — speedup unbounded at practical budgets; warm finals beat cold finals x1.2-x4 in 12/12; 6/12 warm starts were within 95% at 1 eval (near-neutral mutations). Decision: Lamarckian warm-starting is MANDATORY in the memetic driver (homemaker-py-b39), not an optimisation; cold starts produce strictly worse geometry at equal budget. Note: 2 undivides were exactly fitness-neutral (same-type merge == Merge_Divided equivalence) — locality datum for homemaker-py-nyb.","dependencies":[{"issue_id":"homemaker-py-8cs","depends_on_id":"homemaker-py-1p0","type":"blocks","created_at":"2026-06-12T00:39:34Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-av5","title":"Batched oracle: score many .dom files per invocation","description":"oracle.py currently scores one .dom per urb-fitness.pl call (~1.65 s/dom). DESIGN.md §4.6: batching amortises Perl startup to ~0.99 s/dom and is required so population/batch optimisers can score a whole generation in one oracle call. Extend oracle.py with a batch API: write N .dom files, one perl invocation, parse N .score/.fails pairs. Keep the single-file path for compatibility.","acceptance_criteria":"Batch of 35 corpus files scores in one perl invocation; per-file results identical to single-file calls; measured s/dom reported","status":"closed","priority":1,"issue_type":"task","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-06-11T23:36:56Z","created_by":"Bruno Postle","updated_at":"2026-06-12T00:14:06Z","started_at":"2026-06-11T23:50:40Z","closed_at":"2026-06-12T00:14:06Z","close_reason":"score_batch() lands in oracle.py; 35-file corpus parity verified single-vs-batch (1e-12 rel fitness, exact fail sets); 0.98 s/dom batched vs 1.27 single, x1.30","dependency_count":0,"dependent_count":1,"comment_count":0} -{"id":"homemaker-py-iio","title":"Rescoring a dumped .dom under leaf_sharing+collapse_insearch does not reproduce the search's own reported n_fails","description":"Discovered during homemaker-py-91f. driver.search_staged's own r.best.n_fails (computed in-process via NativeEvaluator -\u003e Fitness.score_with_fails on copy.deepcopy(self.root) each eval) is NOT reproduced by dom.dump(r.best.root)+dom.load()+Fitness.score_with_fails on a fresh deepcopy, even with an IDENTICAL, fully-correct conf (leaf_sharing=True, share_edge_cap=True, collapse_insearch=True) and even within the SAME process (no cross-process/hash-seed effects -- verified PYTHONHASHSEED 0-4 all give the identical, stable, WRONG number). Concretely (harbor-house seed=0, budget=20000, full default stack): search reports 37 fails; copy.deepcopy(r.best.root) rescored immediately in-process also gives 37 (exact match, verified 5x); but dom.dump(r.best.root, f)+dom.load(f) then rescored gives a stable, reproducible 53 -- 15 extra fails, dominated by 'missing required space: m*' / 'missing m: would need adjacency/level' for a level-0 count=3 code ('m', Meeting Room) that must be getting satisfied via collapse_insearch's collapse_global relabelling in the live tree but is NOT literally present as a raw leaf.type in the tree (grep of the dumped .dom confirms no leaf typed exactly 'm' anywhere). Ruled out: hash-seed randomness (stable across PYTHONHASHSEED 0-4), naive YAML float-precision loss (dom.load+dom.dump round-trip is byte-stable once loaded), and the known separate collapse_insearch-conf-omission bug in run_staged_search.py's own rescore (homemaker-py-7ua, which produces a DIFFERENT wrong number, 55, via a different mechanism -- missing the collapse_insearch override entirely). This is a THIRD, distinct issue: even with the conf fully correct, dump/reload of the raw (pre-collapse) topology changes what collapse_global's Jacobi-relaxation/adjacency-relabelling converges to. Leading hypothesis (not yet confirmed): dom.py's _link()-reconstructed parent/below/position linkage after a fresh parse does not exactly match the linkage the live, incrementally-mutated search tree carries (module docstring notes 'multi-storey wall-stacking where an upper quad inherits its coordinates from the matching quad below' -- a below-pointer or traversal-order difference could change collapse_global's adjacency graph or leaf iteration order). Needs focused investigation with debug instrumentation inside collapse_global comparing the live vs reloaded tree's leaf order/adjacency graph on the SAME topology. Impact: any workflow that dumps a .dom under the leaf_sharing+collapse_insearch stack and later rescores it from disk (homemaker-fitness CLI, homemaker-collapse, ad-hoc diagnostics) gets a WRONG, but stable/reproducible-looking, fail count -- silently more pessimistic than what the search actually achieved. homemaker-py-91f's fail-category tally works around this by scoring driver.search_staged's r.best.root in-process, immediately, never via a dump/reload round trip (see experiments/run_and_capture_91f.py).","status":"open","priority":2,"issue_type":"bug","owner":"bruno@postle.net","created_at":"2026-08-01T15:29:46Z","created_by":"Bruno Postle","updated_at":"2026-08-01T15:29:46Z","dependency_count":0,"dependent_count":0,"comment_count":0} +{"id":"homemaker-py-iio","title":"Rescoring a dumped .dom under leaf_sharing+collapse_insearch does not reproduce the search's own reported n_fails","description":"Discovered during homemaker-py-91f. driver.search_staged's own r.best.n_fails (computed in-process via NativeEvaluator -\u003e Fitness.score_with_fails on copy.deepcopy(self.root) each eval) is NOT reproduced by dom.dump(r.best.root)+dom.load()+Fitness.score_with_fails on a fresh deepcopy, even with an IDENTICAL, fully-correct conf (leaf_sharing=True, share_edge_cap=True, collapse_insearch=True) and even within the SAME process (no cross-process/hash-seed effects -- verified PYTHONHASHSEED 0-4 all give the identical, stable, WRONG number). Concretely (harbor-house seed=0, budget=20000, full default stack): search reports 37 fails; copy.deepcopy(r.best.root) rescored immediately in-process also gives 37 (exact match, verified 5x); but dom.dump(r.best.root, f)+dom.load(f) then rescored gives a stable, reproducible 53 -- 15 extra fails, dominated by 'missing required space: m*' / 'missing m: would need adjacency/level' for a level-0 count=3 code ('m', Meeting Room) that must be getting satisfied via collapse_insearch's collapse_global relabelling in the live tree but is NOT literally present as a raw leaf.type in the tree (grep of the dumped .dom confirms no leaf typed exactly 'm' anywhere). Ruled out: hash-seed randomness (stable across PYTHONHASHSEED 0-4), naive YAML float-precision loss (dom.load+dom.dump round-trip is byte-stable once loaded), and the known separate collapse_insearch-conf-omission bug in run_staged_search.py's own rescore (homemaker-py-7ua, which produces a DIFFERENT wrong number, 55, via a different mechanism -- missing the collapse_insearch override entirely). This is a THIRD, distinct issue: even with the conf fully correct, dump/reload of the raw (pre-collapse) topology changes what collapse_global's Jacobi-relaxation/adjacency-relabelling converges to. Leading hypothesis (not yet confirmed): dom.py's _link()-reconstructed parent/below/position linkage after a fresh parse does not exactly match the linkage the live, incrementally-mutated search tree carries (module docstring notes 'multi-storey wall-stacking where an upper quad inherits its coordinates from the matching quad below' -- a below-pointer or traversal-order difference could change collapse_global's adjacency graph or leaf iteration order). Needs focused investigation with debug instrumentation inside collapse_global comparing the live vs reloaded tree's leaf order/adjacency graph on the SAME topology. Impact: any workflow that dumps a .dom under the leaf_sharing+collapse_insearch stack and later rescores it from disk (homemaker-fitness CLI, homemaker-collapse, ad-hoc diagnostics) gets a WRONG, but stable/reproducible-looking, fail count -- silently more pessimistic than what the search actually achieved. homemaker-py-91f's fail-category tally works around this by scoring driver.search_staged's r.best.root in-process, immediately, never via a dump/reload round trip (see experiments/run_and_capture_91f.py).","notes":"ROOT CAUSE FOUND: not a dump/reload geometry or linkage bug -- a stale\nleaf_sharing metadata leak inside collapse_global's Hungarian candidate\nvaluation.\n\nFitness._collapse_value (fitness.py) and Fitness._usage_quality both probe a\nHYPOTHETICAL candidate code by temporarily overwriting leaf.type, then call\nquality_size -\u003e graph.leaf_share(leaf, max_share), which reads leaf.type\n(the overridden candidate) and compares it to leaf.share_type. A leaf that\nonce held a live share (share\u003e1, share_type==its type at the time) but was\nsince retyped away carries stale share/share_type metadata by design (the\ndocstrings say retype silently invalidates it, guarded everywhere by\nshare_type==type). But because _collapse_value/_usage_quality overwrite\n.type for the PROBE, a stale share_type that happens to equal the CANDIDATE\ncode under evaluation spuriously reactivates the old k-multiplier credit --\neven though the leaf never actually committed to that code. This inflates\n(or deflates) that one (leaf, candidate) cell of the assignment matrix,\nwhich can flip which leaf the Hungarian solver picks for a given room code.\n\ndom.dump/dom.load round-trips structure/type/geometry perfectly (confirmed:\nnot a float-precision or below-linkage issue as originally suspected) but\ndom._emit only serialises `share` when `share_type == type` -- so a stale\nshare/share_type combo is silently dropped on reload. That's why the LIVE\ntree (carrying the stale metadata) and the RELOADED tree (clean) fed\ndifferent values into the SAME collapse_global assignment and landed on\ndifferent optimal matchings -- reproducing exactly at harbor-house seed=0\nbudget=20000 (structural diff showed exactly 2 leaves with stale\nshare/share_type; live search reported 35 fails, dump/reload rescore gave\n64).\n\nFIX (src/homemaker_layout/fitness.py): in both _collapse_value and\n_usage_quality, when the probed code differs from the leaf's real current\ntype, temporarily clear leaf.share_type (restored in the finally block) so\ngraph.leaf_share can never match a stale share_type against an unrelated\nhypothetical candidate. The leaf's own real current type still legitimately\ncarries a live share (code == orig case is untouched).\n\nVerified: 2 new regression tests in tests/test_collapse_global.py both fail\nwithout the fix and pass with it (one direct _collapse_value unit test, one\nfull collapse_global + dom.dump/dom.load end-to-end test with tuned\ngeometry that reliably flips the assignment pre-fix). Full test suite:\n337/337 pass. Re-ran the exact harbor-house seed=0 budget=20000 repro from\nthe original report: search/in-process-rescore/dump-reload-rescore all now\nagree at 37/37/37 (previously 35/35/64).","status":"in_progress","priority":2,"issue_type":"bug","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-08-01T15:29:46Z","created_by":"Bruno Postle","updated_at":"2026-08-01T20:06:50Z","started_at":"2026-08-01T19:05:07Z","dependency_count":0,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-91f","title":"Residual diagnostic on current full default construction stack","description":"Re-run the §13.1/§13.2-style per-leaf fail-breakdown diagnostic (experiments/diag_leaf_shapefail.py, diag_slack_localization.py) on the CURRENT full default stack (proportion-aware + adjacency-aware seeding, depth-balanced, leaf-sharing factor 3, interior-O odiv=3, share-aware edge cap — post homemaker-py-rq2/x3b), on harbor-house and maple-court. The last such diagnostic predates hph/rq2 (share-aware edge cap) and the erc.7 depth-balance+leaf-sharing synergy default flip, so the current floor (harbor 31.0, maple 74.0 per §13.9) has never been decomposed by failure category/leaf. This is the same read-only methodology that found leaf-sharing (erc.3), depth-balancing (erc.4), interior-O (ld2), and the edge-cap fix (hph) — DESIGN.md's own diagnostic-first discipline. Expected output: which fail category now dominates the residual, informing the next concrete construction lever (the way §13.7's edge-too-long finding directly produced hph). No code changes, no A/B — pure measurement.","design":"Reference: DESIGN.md §13.1 (erc.1), §13.2 (erc.2), §13.7 (71d.1), §13.9 (rq2). Scripts to reuse/extend: experiments/diag_leaf_shapefail.py, experiments/diag_slack_localization.py, experiments/diag_edge_too_long.py.","notes":"Progress note (2026-08-01T16:31:04+01:00): found a real reproducibility gap (filed as homemaker-py-iio, P2) -- rescoring a dumped .dom under the full leaf_sharing+collapse_insearch stack does NOT reproduce the search's own in-process reported n_fails (verified: 37 search-time vs 53 stable-but-wrong after dump+reload, same conf, same process, not hash-seed noise). Also filed homemaker-py-7ua (P3) for a narrower, separate bug: run_staged_search.py's own final sanity rescore omits the collapse_insearch override entirely. Both mean the first batch of 6 staged-search runs I did via the external run_staged_search.py + dump + reload-rescore pipeline (results in /tmp scratchpad .../91f/*.dom) are NOT trustworthy for a fail-category tally -- reloading them and rescoring inflates certain categories (missing-required-space cascade) artificially. Relaunched all 6 (harbor-house + maple-court, seeds 0/1/2, budget 20000, full default stack: leaf_sharing/leaf_share_factor=3/depth_balanced/interior_outside/outside_divisor=3/share_edge_cap) via a new script (experiments/run_and_capture_91f.py) that captures the TRUE in-process fails list immediately off driver.search_staged's r.best.root (verified this matches the search's own reported n_fails exactly, 5/5 repeats) instead of dumping+reloading. This is now running in the background; ETA ~2-2.5h total. Will tally fail categories from the *.fails.json outputs once complete.","status":"closed","priority":2,"issue_type":"task","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-08-01T10:06:37Z","created_by":"Bruno Postle","updated_at":"2026-08-01T18:37:15Z","started_at":"2026-08-01T11:19:09Z","closed_at":"2026-08-01T18:37:15Z","close_reason":"Diagnosed via real driver.search_staged runs (budget 20000, seeds 0/1/2, harbor-house + maple-court, full default stack). Fails/seed: harbor 37/33/30 (mean 33.3, vs cited 31.0); maple 82/84/78 (mean 81.3, vs cited 74.0) -- good sanity check on the stack wiring. Fail-category breakdown (combined): crinkliness 48.0%, size 20.6%, everything else (adjacency/proportion/access/edge-too-long/missing-space/connectivity/stairs) each \u003c=6%. Shape-intrinsic fails (crinkliness+size ~69%) now completely dominate the residual; construction-completeness fails are a small tail. This revises erc.1's (§13.1) old recommendation to deprioritise compactness-cuts (erc.5) in favour of leaf-sharing (erc.3) -- leaf-sharing/depth-balancing/interior-O/edge-cap are now all fully deployed as defaults, yet crinkliness is proportionally MORE dominant than ever. Recommendation written up in DESIGN.md §13.11: reopen erc.5-style compactness-aware cutting, or a crinkliness-targeted lever specifically (crinkliness:size ~2.3:1), as the next concrete construction lever. Two bugs found and filed along the way: homemaker-py-iio (P2, open) -- dumping+reloading a .dom under leaf_sharing+collapse_insearch does not reproduce the search's own in-process fail count (root cause not yet found); homemaker-py-7ua (P3, open) -- run_staged_search.py's own final sanity rescore omits the collapse_insearch override. New scripts: experiments/run_and_capture_91f.py (in-process fails capture, avoids the iio pitfall), experiments/diag_residual_91f.py (category tally from the *.fails.json sidecars).","dependency_count":0,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-1s3","title":"Multi-use leaves as permanent design goal (§26 path b, never attempted)","description":"§26 (homemaker-py-9o5/xi7/b3v) scoped two readings of multi-use leaves (a leaf legitimately serving several DIFFERENT compatible programme codes at once, e.g. study+guest-bedroom, kitchen+dining — Stewart Brand's 'loose-fit' rooms): (a) superposition as a SEARCH RELAXATION — carry uncommitted candidate types per leaf, collapse to one usage only at scoring time; (b) multi-use as the PERMANENT DESIGN GOAL, surviving into the output with no collapse. Only (a) was built and measured, and it was NULL/NEGATIVE — diagnosed as underperforming not from a relaxation gap (measured small, gap_ratio 1.01-1.23) but because the geometry floor dominates: type labels are not the binding constraint on these programmes, so easing them buys nothing while re-typing adds feasibility noise (fitness.py per-eval collapse perturbs counts/adjacency).\n\nPath (b) was never built. It's structurally different from (a): rather than a per-eval relabelling relaxation on top of the existing leaf count, it would permanently REDUCE leaf count by having one leaf serve two rooms' worth of programme requirement simultaneously — the same structural mechanism as leaf-sharing (§13.3, homemaker-py-x3b), which is the single biggest positive lever in the whole DESIGN.md log (harbor-house −21% to −32% at various stages, because it cuts leaf count in a way the search cannot mutate back — §13.4/13.5's key finding that levers the search 'cannot erode' compound, unlike shape-only levers that wash out over a 20k-eval budget).\n\nMultiple independent diagnostics (§12.3 calibration, §12.4's conclusion, §13.1's per-leaf saturation analysis) converge on: the residual fail floor at harbor/maple scale is driven by having as many leaves as there are distinct rooms (52 rooms -\u003e 73 leaves at 44% utilisation gives every leaf a high perimeter/area ratio). Leaf-sharing already exploits this for SAME-code multiplicities; path (b) would extend the same leverage to DIFFERENT-but-compatible codes, which is a materially larger addressable set on programmes with many small single-instance rooms (offices, WCs, meeting rooms — see health-centre, examples/health-centre, 19 distinct codes).\n\nTask: design + build path (b) — likely: SpaceReq gains a compatibility/co-location relation (reuse or extend programme.derive_interchange_classes' S1-S4 guards, or a new explicit 'co_locate' declaration since 'interchangeable' semantics don't fit two DIFFERENT codes coexisting), a leaf can be permanently typed as serving code-pair (or code-set) X, and check_space_counts/quality functions treat the areas as shared per the k-instances-per-leaf model §13.3 already uses for same-code sharing. Gate behind a new conf flag (default OFF, bit-identical when off, per this project's established pattern for every §13.x/§20+ lever). A/B against the current default stack on harbor-house and health-centre (the diverse-room-type programme built for xyu/9yx, §31/§32) before considering a default flip — same discipline as every other lever in this log.","notes":"FINAL VERDICT: NULL. The N=3 positive result did not replicate.\n\nThree measurements collected:\n1. Original (N=3, staged, 20k budget): harbor -1.4%, health-centre -13.9% -- looked promising\n2. Confirm #1 (N=15, plain search, 3k budget, mirrors xyu/9yx protocol): harbor +6.1% worse (p=0.30),\n health-centre +6.6% worse (Wilcoxon p=0.044) -- different protocol (budget+algorithm), but negative\n3. Confirm #2 (N=15, staged, 20k budget -- TRUE same-conditions replication): harbor +6.6% worse (p=0.15),\n health-centre +4.7% worse (p=0.48) -- both trend negative, neither significant\n\nConfirm #2 is the one that actually matches the original protocol (only seed count differs), and it\ndisagrees with the original's direction on both programmes. Conclusion: the N=3 result was sampling\nnoise (health-centre's -13.9% was driven substantially by one seed swinging 71-\u003e43; didn't hold at N=15).\n\nmulti_use stays default OFF, not recommended even as a \"promising\" lever -- this is a clean NULL result,\nnot mixed/promising. Mechanism is complete, tested (335/335), gated off, left in the codebase as a\ndocumented option an architect could opt into per-programme, but no further investment planned.\n\nDESIGN.md §33 fully rewritten with all three measurements and the honest conclusion. This closes out both\nhalves of §26's multi-use-leaves question: path (a) search-relaxation was NULL/NEGATIVE, path (b)\npermanent-fusion is NULL after replication.\n\nTotal compute across this investigation: ~2h (original) + ~2h (precision-weighted rerun) + ~2h (mixture\nrerun) + ~1.5h (N=15 plain confirm) + ~10h (N=15 staged confirm) = ~17.5h across 5 A/B runs.","status":"closed","priority":2,"issue_type":"task","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-07-30T07:55:56Z","created_by":"Bruno Postle","updated_at":"2026-08-01T07:48:32Z","started_at":"2026-07-30T09:20:03Z","closed_at":"2026-08-01T07:48:32Z","close_reason":"Closed","dependency_count":0,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-h10","title":"Re-run §12.3 reassociate/shape-feasibility A/B at fixed worker count","description":"§12.3 (homemaker-py-9gp) measured mutate_reassociate (M3 Wong-Liu move) and the shape-feasibility pre-filter as negative: +3.3/+4.0 fails on maple/harbor. But that A/B ran BEFORE §12.4 (homemaker-py-c3g) found and fixed a real nondeterminism bug — driver._run_batch admitted parallel futures in completion order rather than submission order, producing ±3-6 fail noise between otherwise-identical runs. §12.4's own writeup flags this explicitly: 'sub-±3 effects (the §12.3 +3-4 negatives, the §12.4 ±1.7) should be re-run at a single fixed worker count before being trusted as magnitudes.' That re-run was never done for §12.3.\n\nReassociate is the only search-machinery move in the whole DESIGN.md log verified to reach genuinely new tree topologies (confirmed on synthetic cases in §12.3's own tests) — every other outer-search-machinery lever tried (niching, graded objective, island model, grain annealing, graded connectivity, circulation repair, beam search, ruin-recreate, bubble-diagram signal) is independently null-to-negative for other reasons, so this is the one candidate whose 'negative' verdict might be pure measurement artifact rather than a real finding.\n\nTask: re-run experiments/run_9gp_ab.sh (maple-court + harbor, seeds 0/1/2, 20000 evals, staged) with the post-c3g determinism fix in place, at a single fixed worker count (matching whatever the original run used — check the script/log for workers=N). If the negative holds at fixed worker count, close as confirmed-null (upgrade §12.3's confidence). If it flips positive or neutral, this reopens the reachability question closed in §12.3/§12.4's 'residual is geometry floor, not search-reachability' conclusion — would need a larger-N confirmation before any default flip, per this project's own evidentiary bar (cf. f1d/1ph/e01 pattern).","status":"closed","priority":2,"issue_type":"task","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-07-30T07:55:15Z","created_by":"Bruno Postle","updated_at":"2026-07-30T08:14:43Z","started_at":"2026-07-30T08:06:25Z","closed_at":"2026-07-30T08:14:43Z","close_reason":"Confirmed without a full re-run: run_9gp_ab.sh/run_staged_search.py never threaded a worker count, so every §12.3 arm already ran at n_workers=1 (serial) — the mode §12.4 already proved byte-for-byte reproducible even before the completion-order fix (that bug is ProcessPoolExecutor-as_completed-only). Spot-checked empirically too: same config run twice (harbor-house s0, budget 300) gave identical fail counts at every checkpoint. §12.3's negative verdict is CONFIRMED at fixed worker count; DESIGN.md §12.4 updated with the finding.","dependency_count":0,"dependent_count":0,"comment_count":0} @@ -104,27 +104,27 @@ {"id":"homemaker-py-erc.6","title":"Experiment: inner-loop slack-expansion objective term","description":"Inner-loop counterpart to plot-fill construction. If Diagnostic B shows the inner loop has room to expand leaves into slack but no objective gradient to do so (the scalar rewards hitting target area but not exceeding it where slack exists), add a term/incentive so the ratio optimiser pushes leaf boundaries out to consume neighbouring slack and satisfy size, rather than parking at target.\n\nCONDITIONAL on Diagnostic B: build this only if B localizes the gap to the inner loop (room to expand, no gradient); if B shows construction targets too-small dims, prefer the plot-fill construction sibling. Must preserve the §5.4 inner-loop cliff / §4.9 lexicographic protection — the term sits where it cannot displace the fail-count ordering. A/B vs §12.2 baseline, seeds 0/1/2, 20000 evals, staged, default-OFF. Record DESIGN.md §13.6.","notes":"DEPRIORITISED by Diagnostic B (§13.2). B shows the inner loop CANNOT repair undersize: the slack is depth-driven maldistribution baked into the frozen topology, and the equal-offset ratio DOF cannot shrink a 14x leaf to feed a starved one without trading into shape fails (0.5^n cliff). Wrong DOF and wrong direction — the blocker is slicing POSITION, not a missing expansion reward. Fix belongs upstream in construction/topology (erc.4 re-scoped, erc.3). Keep as a low-priority follow-up only if a depth-balanced construction still leaves a residual size gradient the inner loop could pick up.","status":"closed","priority":4,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-06-22T23:16:24Z","created_by":"Bruno Postle","updated_at":"2026-06-28T13:22:22Z","closed_at":"2026-06-28T13:22:22Z","close_reason":"wont-fix (DESIGN §13.7): Diag B (§13.2) showed the inner loop cannot repair undersize (wrong DOF — slicing position, frozen-topology ratios). Superseded by depth-balanced construction (erc.4). Condition unmet.","dependencies":[{"issue_id":"homemaker-py-erc.6","depends_on_id":"homemaker-py-erc","type":"parent-child","created_at":"2026-06-23T00:16:23Z","created_by":"Bruno Postle","metadata":"{}"},{"issue_id":"homemaker-py-erc.6","depends_on_id":"homemaker-py-erc.2","type":"blocks","created_at":"2026-06-23T00:16:47Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-erc.5","title":"Experiment: compactness-aware cuts (minimize leaf perimeter/area)","description":"Attacks the #1 factor, crinkliness (346) — a per-leaf perimeter/area property DISTINCT from proportion (aspect ratio). Proportion-aware seeding (leu.2) sizes splits but does not bias toward balanced, square-ish subdivision. Add a KD-tree-style 'keep both children compact' cut rule (prefer the cut orientation/position that minimises summed child perimeter/area) in construction.\n\nCONDITIONAL on Diagnostic A: if A shows per-leaf shape-fail is FLAT across densities (floor intrinsic to slicing density), better cuts at the same leaf count will not pay → this should be closed wont-fix in favour of leaf-sharing. Only build if A shows shape-fail RISES with density. A/B vs §12.2 baseline, seeds 0/1/2, 20000 evals, staged, default-OFF. Record DESIGN.md §13.5.","notes":"DEPRIORITISED by erc.1 verdict (§13.1): per-leaf shape-fail flat vs slicing density and cuts already squarest (_size_divisions_from_targets picks squarest rotation) yet still ~1.8 fails/leaf =\u003e little compactness headroom at fixed leaf count. Floor is intrinsic to leaf COUNT, not cut quality. Revisit only if leaf-sharing (erc.3) underdelivers.","status":"closed","priority":4,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-06-22T23:16:21Z","created_by":"Bruno Postle","updated_at":"2026-06-28T13:22:17Z","closed_at":"2026-06-28T13:22:17Z","close_reason":"wont-fix (DESIGN §13.7): Diag A (§13.1) showed the floor is intrinsic to leaf COUNT not cut quality; revisit condition was 'only if leaf-sharing underdelivers' but leaf-sharing OVER-delivered (−32…−39%, §13.3). Condition unmet.","dependencies":[{"issue_id":"homemaker-py-erc.5","depends_on_id":"homemaker-py-erc","type":"parent-child","created_at":"2026-06-23T00:16:21Z","created_by":"Bruno Postle","metadata":"{}"},{"issue_id":"homemaker-py-erc.5","depends_on_id":"homemaker-py-erc.1","type":"blocks","created_at":"2026-06-23T00:16:43Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-2g5","title":"Rebuild occlusion/daylight/sun subsystem in Python (post-Phase-5, after optimisation fully native)","description":"DESIGN.md §6 port scope — a whole subsystem, not a term. quality_daylight (Leaf.pm:281-296) needs Urb::Misc::Sun + Urb::Field::Occlusion (+CIESky); quality_uncrinkliness also takes the occlusion object. Indoor spaces return 1 for daylight; cost is outdoor spaces + crinkliness. Port Sun_horizontal (262980-minute normalisation) and the occlusion wall set from Dom-\u003eWalls.","acceptance_criteria":"Daylight and crinkliness factors match Perl (float tolerance) across the corpus, including multi-storey cases","notes":"Re-scoped 2026-06-12: occlusion disabled in the Urb oracle instead of ported (see homemaker-py-gp2). Native fitness ships with simple crinkliness (illumination factor = 1, in homemaker-py-gnw). This issue is now the eventual Python occlusion rebuild, only after optimisation works entirely in Python. Restores outdoor-daylight and shaded-wall selection pressure.\nReframed 2026-06-17: orthogonal to epic homemaker-py-c4c. This is fitness FIDELITY (restoring daylight + shaded-wall selection pressure to match Perl), not search CAPABILITY — it changes what 'good' means, not the search's ability to find good. It will NOT improve final designs in the sense currently sought. Stays P4, deferred until the topology-search-quality epic lands and optimisation is fully native.","status":"open","priority":4,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-06-11T23:38:25Z","created_by":"Bruno Postle","updated_at":"2026-06-17T19:14:48Z","dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"memory","key":"experiment-harness-gotcha-the-leaf-sharing-relaxed-objective","value":"Experiment harness gotcha: the leaf-sharing RELAXED objective (§13.3) is injected ONLY by monkeypatching fitness.load_config in the parent process (run_staged_search.py / probe scripts). This is parent-process-only and does NOT propagate into ProcessPoolExecutor workers (n_workers\u003e1), which re-import fitness fresh and score under the STRICT on-disk patterns.config -\u003e r.n_fails MISMATCH (worker strict vs parent relaxed re-score). ALL §13.x floor runs were therefore SERIAL. Any future PARALLEL leaf-sharing experiment will silently mis-score until leaf_sharing lives on disk/CLI (tracked: homemaker-py-x3b). The parallel driver itself is correct; both paths score via load_config(programme_dir)."} -{"_type":"memory","key":"never-use-corpus-filenames-candidate-001-dom-candidate","value":"Never use corpus filenames (candidate-001.dom, candidate-002.dom, generated.dom, init.dom, etc.) as --output targets when running experiments. These are test fixtures. Always write experimental outputs to scratch/ or a timestamped path. Lesson from 2026-06-14: warm-start runs overwrote candidate-001/002.dom and broke graph tests."} -{"_type":"memory","key":"proportion-aware-constructive-seeding-leu-2-12-2","value":"Proportion-aware constructive seeding (leu.2/§12.2): sizing seed cuts from target AREAS only regresses (thin slivers wreck aspect); you must ALSO pick each cut's rotation for child squareness. It is a convergence ACCELERATOR via a deeper local optimum around the constructed topology: wins where that topology is roughly right and budget is scarce (harbor -13%, maple -10% at 20k evals) but DELAYS small programmes where the seed must be restructured by undivide (programme-house regresses at fixed budget, yet reaches the floor given budget - speed, not asymptote). Default-on. Also: n_storeys must honour storey_minimum, not just level: keys (programme-house storey_minimum:2, all rooms level:0 - was seeded 1 storey short; cq1)."} {"_type":"memory","key":"run-to-run-reproducibility-in-homemaker-layout-serial","value":"Run-to-run reproducibility in homemaker-layout: serial search (workers=1) is byte-for-byte deterministic; parallel (workers\u003e1) is now deterministic too AFTER fixing driver._run_batch to admit futures in submission order (was as_completed/completion order, bug xcy). Reproducibility holds only for a FIXED worker count — serial vs parallel differ because children-per-iteration is 1 vs n_workers (different batch granularity), which is expected, not a bug. The constructive seeder was NEVER nondeterministic: _assign_adjacency_aware has unique idx tiebreaks; comparing topologies with Python builtin hash() of the signature STRING is invalid (PYTHONHASHSEED salts str hashing per process) — use a stable hash (sha1) or genome.signature equality."} -{"_type":"memory","key":"urb-fitness-bug-found-fixed-2026-06-12","value":"Urb fitness bug found+fixed 2026-06-12 (patch in /home/bruno/src/urb, uncommitted): ProgrammeDriven.pm ratio_o/ratio_type grepped case-insensitively over the ratios hash and took the FIRST key — nondeterministic (x4.5 score swings) for designs with mixed-case type classes (both 'c' circulation and 'C' covered). Fixed to SUM the class (matches Is_Circulation//Is_Outside semantics); 35/35 corpus scores unchanged. CRITICAL for homemaker-py-3y7/gnw: the native port must implement class-SUM ratios. Building.pm has the same unpatched pattern (site-driven path, not used by our oracle). Also: the memetic search reward-hacked this bug before the fix — search results predating it are noise artifacts."} -{"_type":"memory","key":"collapse-global-94g-and-any-label-usage-optimisation","value":"collapse_global (94g) and any label/usage optimisation CANNOT fix geometry-intrinsic fails. The harbor-house 15-fail best layout contains long-thin cells that are useless whatever room usage is assigned — their width/proportion/crinkliness fails are shape-bound, not label slack. Two consequences: (1) do not over-claim collapse gains — only ~2-3 of that layout's fails are reclaimable relabel slack, the rest are geometry- or building-level bound; (2) the threshold objective must not be tuned to 'pass' a degenerate cell via a permissive room type — a metric-pass on a physically useless space is gaming, not a fix. Real remedies for these are geometry/topology search (cell shape) and circulation placement, filed separately, not the collapse."} -{"_type":"memory","key":"island-model-psk-14-is-a-null-priming","value":"Island model (psk, §14) is a NULL: priming a population from N converged independent elites + crossover-heavy migration does not beat best-of-N at equal total budget (maple island 124 vs control 116). The child_probe instrument shows WHY: area-matched crossover across independently-converged elites almost never synthesizes (1-3 of ~64 children beat the better parent, max drop 2-5) because the slicing encoding is non-canonical (9gp), so splices are disruptive not combinatorial. Search-machinery null #3 after graded-objective and niching/restarts; residual stays geometry/shape-bound."} -{"_type":"memory","key":"programme-house-optimisation-result-2026-06-14-15","value":"Programme-house optimisation result (2026-06-14/15): best achievable is 1 fail (l1 wrong level, score ~0.005). 0 fails is geometrically impossible: l1 (min 27m²) must occupy ll (~23m²) at level 0, which eliminates the t3-adj-C provider; dividing ll into lll(l1)+llr(C) gives llr proportion ~6:1 (fails). Python memetic optimizer achieves 1 fail in 50k evals vs Perl optimiser's 2-3 fails. Winning topology: TWO C nodes at level 0 — ll(C) for t3-adj-C via geometric contact, rl(C) for staircase via tree-sibling adjacency to rrr(O). Best .dom: scratch/from-warmstart-fixed.dom and scratch/from-compound3-fixed.dom."} -{"_type":"memory","key":"unfold-strategy-for-shared-leaves-homemaker-py-8iv","value":"Unfold strategy for shared leaves (homemaker-py-8iv, resolved 2026-07-16): use the BALANCED GRID (operators._grow_balanced/_size_subtree_equal), NOT circulation-aware slicing. Slicing a shared leaf perpendicular to its access edge so every child touches the corridor was implemented + A/B-tested and LOST decisively (150k-eval warm-start polish from evolved-3M: slice 41 fails/3.5e-14 vs grid 25 fails/2.4e-09, grid ahead at every milestone). Reason: k rooms all touching one wall are intrinsically thin slices; that geometric debt (proportion/long/width) is unfixable without topology change, while the grid's squarer children let local search re-route access cheaply via level_retype/place_missing/level_fix. Lesson: at the sharing-\u003eno-sharing transition, prioritise squarer children and leave access to local search; do not reintroduce slicing in Schedule B (kpu)."} {"_type":"memory","key":"9o5-multi-use-leaves-is-path-a-superposition","value":"9o5 multi-use leaves is path (a) — superposition as SEARCH RELAXATION that COLLAPSES to specific usage at the end, NOT path (b) loose-fit/no-collapse. Bruno's intent: codes with SIMILAR leaf requirements form an interchangeable equivalence class; during evolution the solver doesn't commit which leaf serves which specific usage (smoother landscape, no fighting over exact leaf usage); at the end the layout is CONDENSED to specific usages by brute-forcing the in-class assignment (3 interchangeable usages over 3 leaves = 3! = 6 combinations to check, pick best). 'Derive automatically' compatibility = requirement-similarity grouping. This reverses the issue's stated 'path b preferred' note."} -{"_type":"memory","key":"deceptive-valleys-in-topology-search-when-every-single","value":"Deceptive valleys in topology search: when every single-step mutation from a target state passes through a high-fail intermediary (e.g. level_fix displaces a room into 5+ new fails), a compound operator that atomically applies two coordinated changes can escape. Design compound operators to land on the low-fail state directly, bypassing the deceptive gradient. Programme-house example: level_compound_fix atomically moves the level-constrained room AND re-inserts the displaced room adjacent to C in one step (operators.py, 2026-06-14)."} -{"_type":"memory","key":"homemaker-py-pythonpath-set-pythonpath-home-bruno-src","value":"homemaker-layout PYTHONPATH: package installed as 'homemaker-layout' via pip install -e . so 'import homemaker_layout' works from anywhere without PYTHONPATH. For running tests use 'python -m pytest' from project root /home/bruno/src/homemaker-layout (pyproject.toml adds src/ automatically). Never try pip show homemaker — that's the old homemaker-addon conflict."} -{"_type":"memory","key":"multi-storey-staircase-consistency-when-dividing-or-retyping","value":"Multi-storey staircase consistency: when dividing or retyping a circulation (C) leaf at one level, the same structural change should be propagated to the matching leaf on ALL other storeys so the stair core path is maintained. The optimizer cannot fix staircase disruptions through trial-and-error geometry alone — it requires a synchronized multi-level operator that applies the same topology change to every storey simultaneously."} -{"_type":"memory","key":"adjacency-in-binary-slicing-tree-is-structural-not","value":"Adjacency in binary slicing tree is structural, not geometric: the inner-loop NM cannot fix topological adjacency failures. Two paths exist: (1) tree-sibling adjacency — a node is adjacent to its sibling in the tree; (2) cross-zone geometric adjacency — leaves from different subtrees that happen to share a boundary. Staircase/adjacency fails require a topology mutation that changes which nodes are siblings or which zones touch. This was proved empirically on programme-house: staircase fail from rot=0 layout could not be fixed by NM but was fixed by level_retype creating a two-C topology (2026-06-14/15)."} -{"_type":"memory","key":"collapse-global-s-jacobi-adjacency-relaxation-homemaker-py","value":"collapse_global's Jacobi adjacency relaxation (homemaker-py-94g) is a synchronous per-round linear-assignment re-solve, which can 2-cycle indefinitely between two labellings that each satisfy ZERO adjacency requirements even though a permutation satisfying ALL of them exists -- proven on a minimal 4-cell chain (p1-q1-p2-q2, two disjoint adjacency pairs p1\u003c-\u003ep2/q1\u003c-\u003eq2) in test_two_opt_polish_escapes_jacobi_plateau. homemaker-py-9wi added Fitness._two_opt_adjacency_polish: a same-level pairwise-swap local search run after the Jacobi fixpoint, gated behind collapse_global(local_search=True) (default off, exposed as homemaker-collapse --local-search). Monotone by construction (a swap is kept only if it strictly increases total reward). Empirically on the 11 harbor-house evolved-*.dom/3m.dom/materialised-3M.dom layouts: 10 matched Jacobi-only exactly, 0 regressed, and evolved-anneal-3M.dom improved 21-\u003e19 fails (fixed a genuine mutual da1\u003c-\u003ek1 adjacency miss the Jacobi loop couldn't reach)."} -{"_type":"memory","key":"urb-oracle-nondeterminism-urb-fitness-pl-output-varies","value":"Urb oracle nondeterminism: urb-fitness.pl output varies run-to-run from Perl hash-order randomisation — .fails line ORDER shuffles (compare sorted, use oracle.Score.fail_lines) and the score float can flip by ~1 ULP (compare with math.isclose rel_tol=1e-12, never ==). Not a batching artifact; affects single runs too. Matters for the Phase 3 native-fitness parity gate (homemaker-py-uxz)."} -{"_type":"memory","key":"correction-to-urb-fitness-bug-memory-bruno-2026","value":"CORRECTION to urb-fitness-bug memory (Bruno, 2026-06-12): 'C' is NOT a 'covered' type — Is_Covered is a geometric predicate (indoor space above). Urb's generic types are canonically UPPERCASE: C=circulation, O=outside, S=sahn (get_space_types qw/C O S/; corpus is 100% uppercase, never 'c'/'o' leaves). The mixed-case designs that fired the latent ratio_type first-match bug were created by homemaker's own operator type pool emitting lowercase 'c'/'o' — fixed: driver/operators now emit uppercase generics only, and class checks use t[0].lower() in 'cos'. The Urb class-sum patch stays as defensive hardening (zero impact on canonical designs). Native port (3y7/gnw): treat type classes case-insensitively, generics canonically uppercase."} -{"_type":"memory","key":"experiment-seeding-pitfall-run-search-scaled-py-s","value":"Experiment seeding pitfall: run_search_scaled.py's default PH_SEED (c964…dom) is a FINISHED programme-house design — passing it warm-starts and floors at ~3 fails, NOT a blank-slate topology search. For blank-slate runs comparable to §11.5/§11.6 baselines, seed from examples/programme-house/init.dom (a bare undivided plot; driver bootstrap auto-triggers only on bare plots). Bit the 6zy sweep — first pass used c964 and falsely showed 3-fail floor across the whole grid."} +{"_type":"memory","key":"collapse-global-94g-and-any-label-usage-optimisation","value":"collapse_global (94g) and any label/usage optimisation CANNOT fix geometry-intrinsic fails. The harbor-house 15-fail best layout contains long-thin cells that are useless whatever room usage is assigned — their width/proportion/crinkliness fails are shape-bound, not label slack. Two consequences: (1) do not over-claim collapse gains — only ~2-3 of that layout's fails are reclaimable relabel slack, the rest are geometry- or building-level bound; (2) the threshold objective must not be tuned to 'pass' a degenerate cell via a permissive room type — a metric-pass on a physically useless space is gaming, not a fix. Real remedies for these are geometry/topology search (cell shape) and circulation placement, filed separately, not the collapse."} {"_type":"memory","key":"homemaker-py-3l6-fix-leaf-sharing-evolve-runs","value":"homemaker-py-3l6 fix: leaf-sharing evolve runs now auto-finish before write via driver.polish_finish — unfold_shared_leaves() then a warm-started leaf_sharing=False polish search (--polish-budget, default budget//2). Makes the written .dom honest under canonical homemaker-fitness (internal==canonical when leaf_sharing off). Interrupt path forces polish_budget=0 (unfold+rescore only). This is yaa's unfold-then-polish, made automatic; Schedule B annealing is still kpu."} -{"_type":"memory","key":"strategy-decision-2026-06-12-bruno-occlusion-daylight","value":"Strategy decision 2026-06-12 (Bruno): occlusion/daylight is ORTHOGONAL to building a scalable optimiser. Disable it in Urb (env flag, homemaker-py-gp2) rather than port it; native fitness uses simple crinkliness (illumination factor = 1); rebuild occlusion in Python only after optimisation is fully native (homemaker-py-2g5, now P4). Consequence: all scores change when the flag flips — re-baseline corpus/.score, DESIGN \\$4.5 gains, gate bars at one clean boundary AFTER homemaker-py-1p0 closes; Phase-2 urb-evolve benchmark must run with the same flag."} +{"_type":"memory","key":"island-model-psk-14-is-a-null-priming","value":"Island model (psk, §14) is a NULL: priming a population from N converged independent elites + crossover-heavy migration does not beat best-of-N at equal total budget (maple island 124 vs control 116). The child_probe instrument shows WHY: area-matched crossover across independently-converged elites almost never synthesizes (1-3 of ~64 children beat the better parent, max drop 2-5) because the slicing encoding is non-canonical (9gp), so splices are disruptive not combinatorial. Search-machinery null #3 after graded-objective and niching/restarts; residual stays geometry/shape-bound."} +{"_type":"memory","key":"never-use-corpus-filenames-candidate-001-dom-candidate","value":"Never use corpus filenames (candidate-001.dom, candidate-002.dom, generated.dom, init.dom, etc.) as --output targets when running experiments. These are test fixtures. Always write experimental outputs to scratch/ or a timestamped path. Lesson from 2026-06-14: warm-start runs overwrote candidate-001/002.dom and broke graph tests."} +{"_type":"memory","key":"correction-to-urb-fitness-bug-memory-bruno-2026","value":"CORRECTION to urb-fitness-bug memory (Bruno, 2026-06-12): 'C' is NOT a 'covered' type — Is_Covered is a geometric predicate (indoor space above). Urb's generic types are canonically UPPERCASE: C=circulation, O=outside, S=sahn (get_space_types qw/C O S/; corpus is 100% uppercase, never 'c'/'o' leaves). The mixed-case designs that fired the latent ratio_type first-match bug were created by homemaker's own operator type pool emitting lowercase 'c'/'o' — fixed: driver/operators now emit uppercase generics only, and class checks use t[0].lower() in 'cos'. The Urb class-sum patch stays as defensive hardening (zero impact on canonical designs). Native port (3y7/gnw): treat type classes case-insensitively, generics canonically uppercase."} +{"_type":"memory","key":"deceptive-valleys-in-topology-search-when-every-single","value":"Deceptive valleys in topology search: when every single-step mutation from a target state passes through a high-fail intermediary (e.g. level_fix displaces a room into 5+ new fails), a compound operator that atomically applies two coordinated changes can escape. Design compound operators to land on the low-fail state directly, bypassing the deceptive gradient. Programme-house example: level_compound_fix atomically moves the level-constrained room AND re-inserts the displaced room adjacent to C in one step (operators.py, 2026-06-14)."} +{"_type":"memory","key":"unfold-strategy-for-shared-leaves-homemaker-py-8iv","value":"Unfold strategy for shared leaves (homemaker-py-8iv, resolved 2026-07-16): use the BALANCED GRID (operators._grow_balanced/_size_subtree_equal), NOT circulation-aware slicing. Slicing a shared leaf perpendicular to its access edge so every child touches the corridor was implemented + A/B-tested and LOST decisively (150k-eval warm-start polish from evolved-3M: slice 41 fails/3.5e-14 vs grid 25 fails/2.4e-09, grid ahead at every milestone). Reason: k rooms all touching one wall are intrinsically thin slices; that geometric debt (proportion/long/width) is unfixable without topology change, while the grid's squarer children let local search re-route access cheaply via level_retype/place_missing/level_fix. Lesson: at the sharing-\u003eno-sharing transition, prioritise squarer children and leave access to local search; do not reintroduce slicing in Schedule B (kpu)."} +{"_type":"memory","key":"user-preference-bruno-this-is-a-fedora-system","value":"User preference (Bruno): this is a Fedora system — NEVER install Python packages via pip without asking first; always ask whether to install the rpm via dnf (e.g. python3-cma) before considering pip. Applies to any dependency additions."} +{"_type":"memory","key":"homemaker-py-pythonpath-set-pythonpath-home-bruno-src","value":"homemaker-layout PYTHONPATH: package installed as 'homemaker-layout' via pip install -e . so 'import homemaker_layout' works from anywhere without PYTHONPATH. For running tests use 'python -m pytest' from project root /home/bruno/src/homemaker-layout (pyproject.toml adds src/ automatically). Never try pip show homemaker — that's the old homemaker-addon conflict."} +{"_type":"memory","key":"urb-oracle-nondeterminism-urb-fitness-pl-output-varies","value":"Urb oracle nondeterminism: urb-fitness.pl output varies run-to-run from Perl hash-order randomisation — .fails line ORDER shuffles (compare sorted, use oracle.Score.fail_lines) and the score float can flip by ~1 ULP (compare with math.isclose rel_tol=1e-12, never ==). Not a batching artifact; affects single runs too. Matters for the Phase 3 native-fitness parity gate (homemaker-py-uxz)."} +{"_type":"memory","key":"collapse-global-s-jacobi-adjacency-relaxation-homemaker-py","value":"collapse_global's Jacobi adjacency relaxation (homemaker-py-94g) is a synchronous per-round linear-assignment re-solve, which can 2-cycle indefinitely between two labellings that each satisfy ZERO adjacency requirements even though a permutation satisfying ALL of them exists -- proven on a minimal 4-cell chain (p1-q1-p2-q2, two disjoint adjacency pairs p1\u003c-\u003ep2/q1\u003c-\u003eq2) in test_two_opt_polish_escapes_jacobi_plateau. homemaker-py-9wi added Fitness._two_opt_adjacency_polish: a same-level pairwise-swap local search run after the Jacobi fixpoint, gated behind collapse_global(local_search=True) (default off, exposed as homemaker-collapse --local-search). Monotone by construction (a swap is kept only if it strictly increases total reward). Empirically on the 11 harbor-house evolved-*.dom/3m.dom/materialised-3M.dom layouts: 10 matched Jacobi-only exactly, 0 regressed, and evolved-anneal-3M.dom improved 21-\u003e19 fails (fixed a genuine mutual da1\u003c-\u003ek1 adjacency miss the Jacobi loop couldn't reach)."} +{"_type":"memory","key":"programme-house-optimisation-result-2026-06-14-15","value":"Programme-house optimisation result (2026-06-14/15): best achievable is 1 fail (l1 wrong level, score ~0.005). 0 fails is geometrically impossible: l1 (min 27m²) must occupy ll (~23m²) at level 0, which eliminates the t3-adj-C provider; dividing ll into lll(l1)+llr(C) gives llr proportion ~6:1 (fails). Python memetic optimizer achieves 1 fail in 50k evals vs Perl optimiser's 2-3 fails. Winning topology: TWO C nodes at level 0 — ll(C) for t3-adj-C via geometric contact, rl(C) for staircase via tree-sibling adjacency to rrr(O). Best .dom: scratch/from-warmstart-fixed.dom and scratch/from-compound3-fixed.dom."} +{"_type":"memory","key":"proportion-aware-constructive-seeding-leu-2-12-2","value":"Proportion-aware constructive seeding (leu.2/§12.2): sizing seed cuts from target AREAS only regresses (thin slivers wreck aspect); you must ALSO pick each cut's rotation for child squareness. It is a convergence ACCELERATOR via a deeper local optimum around the constructed topology: wins where that topology is roughly right and budget is scarce (harbor -13%, maple -10% at 20k evals) but DELAYS small programmes where the seed must be restructured by undivide (programme-house regresses at fixed budget, yet reaches the floor given budget - speed, not asymptote). Default-on. Also: n_storeys must honour storey_minimum, not just level: keys (programme-house storey_minimum:2, all rooms level:0 - was seeded 1 storey short; cq1)."} +{"_type":"memory","key":"adjacency-in-binary-slicing-tree-is-structural-not","value":"Adjacency in binary slicing tree is structural, not geometric: the inner-loop NM cannot fix topological adjacency failures. Two paths exist: (1) tree-sibling adjacency — a node is adjacent to its sibling in the tree; (2) cross-zone geometric adjacency — leaves from different subtrees that happen to share a boundary. Staircase/adjacency fails require a topology mutation that changes which nodes are siblings or which zones touch. This was proved empirically on programme-house: staircase fail from rot=0 layout could not be fixed by NM but was fixed by level_retype creating a two-C topology (2026-06-14/15)."} {"_type":"memory","key":"ld2-13-6-interior-o-seed-diagnostic-all","value":"ld2/§13.6 interior-O seed diagnostic: ALL crinkliness fails in the constructed bal+share seed are UNDER-exposed (crink\u003c0.62, landlocked rooms with no facade + no uncovered-O neighbour) — zero over-exposed sliver fails. So the erc crinkliness residual is genuine under-daylighting, validating the interior light-well premise. Default outside_divisor=6 was too sparse (null: harbor 147-\u003e142, crinkliness even rose). odiv=3 is the seed-optimal joint setting: harbor seed fails 147-\u003e129 (-18), maple 219-\u003e206 (-14), landlocked fails drop, at cost of more leaves (harbor +4, maple +8). Because it ADDS leaves it carries the §13.4 wash-out risk; A/B to convergence pending."} +{"_type":"memory","key":"multi-storey-staircase-consistency-when-dividing-or-retyping","value":"Multi-storey staircase consistency: when dividing or retyping a circulation (C) leaf at one level, the same structural change should be propagated to the matching leaf on ALL other storeys so the stair core path is maintained. The optimizer cannot fix staircase disruptions through trial-and-error geometry alone — it requires a synchronized multi-level operator that applies the same topology change to every storey simultaneously."} +{"_type":"memory","key":"urb-fitness-bug-found-fixed-2026-06-12","value":"Urb fitness bug found+fixed 2026-06-12 (patch in /home/bruno/src/urb, uncommitted): ProgrammeDriven.pm ratio_o/ratio_type grepped case-insensitively over the ratios hash and took the FIRST key — nondeterministic (x4.5 score swings) for designs with mixed-case type classes (both 'c' circulation and 'C' covered). Fixed to SUM the class (matches Is_Circulation//Is_Outside semantics); 35/35 corpus scores unchanged. CRITICAL for homemaker-py-3y7/gnw: the native port must implement class-SUM ratios. Building.pm has the same unpatched pattern (site-driven path, not used by our oracle). Also: the memetic search reward-hacked this bug before the fix — search results predating it are noise artifacts."} {"_type":"memory","key":"warm-x0-initialization-bug-pattern-when-a-topology","value":"warm_x0 initialization bug pattern: when a topology operator explicitly sets division ratios on a newly-created node (e.g. compound_fix sets node.division=[0.25,0.25] for t3), parent.ratios has no entry for that node (it was a leaf). warm_x0 defaults it to 0.5, corrupting the inner loop's starting point and making the operator invisible to lex comparison. Fix: only propagate child ratios for nodes where the parent node was NOT already divided; stale hidden nodes revealed by structural mutations (swap flipping b.below) must NOT contribute their pre-writeback values. See driver.py lines 259-267 (fixed 2026-06-14)."} {"_type":"memory","key":"cli-tool-style-prefer-python-m-homemaker-module","value":"CLI tool style: prefer python -m homemaker.module --parameters pattern, installable via pip install -e . with pyproject.toml entry_points. Not standalone bin/ scripts."} -{"_type":"memory","key":"user-preference-bruno-this-is-a-fedora-system","value":"User preference (Bruno): this is a Fedora system — NEVER install Python packages via pip without asking first; always ask whether to install the rpm via dnf (e.g. python3-cma) before considering pip. Applies to any dependency additions."} +{"_type":"memory","key":"experiment-harness-gotcha-the-leaf-sharing-relaxed-objective","value":"Experiment harness gotcha: the leaf-sharing RELAXED objective (§13.3) is injected ONLY by monkeypatching fitness.load_config in the parent process (run_staged_search.py / probe scripts). This is parent-process-only and does NOT propagate into ProcessPoolExecutor workers (n_workers\u003e1), which re-import fitness fresh and score under the STRICT on-disk patterns.config -\u003e r.n_fails MISMATCH (worker strict vs parent relaxed re-score). ALL §13.x floor runs were therefore SERIAL. Any future PARALLEL leaf-sharing experiment will silently mis-score until leaf_sharing lives on disk/CLI (tracked: homemaker-py-x3b). The parallel driver itself is correct; both paths score via load_config(programme_dir)."} +{"_type":"memory","key":"experiment-seeding-pitfall-run-search-scaled-py-s","value":"Experiment seeding pitfall: run_search_scaled.py's default PH_SEED (c964…dom) is a FINISHED programme-house design — passing it warm-starts and floors at ~3 fails, NOT a blank-slate topology search. For blank-slate runs comparable to §11.5/§11.6 baselines, seed from examples/programme-house/init.dom (a bare undivided plot; driver bootstrap auto-triggers only on bare plots). Bit the 6zy sweep — first pass used c964 and falsely showed 3-fail floor across the whole grid."} +{"_type":"memory","key":"strategy-decision-2026-06-12-bruno-occlusion-daylight","value":"Strategy decision 2026-06-12 (Bruno): occlusion/daylight is ORTHOGONAL to building a scalable optimiser. Disable it in Urb (env flag, homemaker-py-gp2) rather than port it; native fitness uses simple crinkliness (illumination factor = 1); rebuild occlusion in Python only after optimisation is fully native (homemaker-py-2g5, now P4). Consequence: all scores change when the flag flips — re-baseline corpus/.score, DESIGN \\$4.5 gains, gate bars at one clean boundary AFTER homemaker-py-1p0 closes; Phase-2 urb-evolve benchmark must run with the same flag."} diff --git a/src/homemaker_layout/fitness.py b/src/homemaker_layout/fitness.py index ebf206d..59af911 100644 --- a/src/homemaker_layout/fitness.py +++ b/src/homemaker_layout/fitness.py @@ -334,7 +334,18 @@ class Fitness: (perpendicular, crinkliness, access) and value rate are usage-invariant within a class, so this is the separable per-leaf collapse objective.""" orig = leaf.type + orig_share_type = leaf.share_type leaf.type = usage + if usage != orig: + # homemaker-py-iio: a stale share (share>1, share_type left over + # from a code this leaf no longer holds) must not spuriously + # reactivate just because THIS hypothetical usage probe happens to + # match the old share_type -- graph.leaf_share reads leaf.type, + # which we have just overridden, so it would otherwise compare the + # stale share_type against the candidate usage instead of the + # leaf's real committed type. Only the leaf's OWN current type + # (usage == orig) may legitimately carry a live share. + leaf.share_type = None try: return ( self.quality_size(leaf) @@ -343,6 +354,7 @@ class Fitness: ) finally: leaf.type = orig + leaf.share_type = orig_share_type def _best_assignment(self, quality: list[list[float]]) -> list[tuple[int, int]]: """Maximum-total-quality matching of ``min(rows, cols)`` leaf->slot @@ -447,13 +459,23 @@ class Fitness: if req.level is not None and req.level != lvl: return forbid orig = lf.type + orig_share_type = lf.share_type lf.type = code + if code != orig: + # homemaker-py-iio: see _usage_quality -- a stale share/share_type + # left over from a code this leaf no longer holds must not + # spuriously reactivate just because this hypothetical candidate + # code happens to match it (graph.leaf_share reads leaf.type, + # which is overridden to the candidate here). Only the leaf's own + # current type (code == orig) may legitimately carry a live share. + lf.share_type = None try: qs = self.quality_size(lf) qw = self.quality_width(lf) qp = self.quality_proportion(lf) finally: lf.type = orig + lf.share_type = orig_share_type val = qs * qw * qp * geometry.area(lf) if objective == "threshold": passes = ( diff --git a/tests/test_collapse_global.py b/tests/test_collapse_global.py index a1d44d4..dfd5564 100644 --- a/tests/test_collapse_global.py +++ b/tests/test_collapse_global.py @@ -168,6 +168,87 @@ def test_two_opt_polish_escapes_jacobi_plateau(): assert satisfied(root_polished) == 4 # 2-opt reaches the true optimum +# --------------------------------------------------------------------------- # +# Stale leaf-share must not leak into a hypothetical candidate (homemaker-py-iio) +# --------------------------------------------------------------------------- # + +def test_collapse_value_ignores_stale_share_for_hypothetical_code(): + # A leaf that once held a live share (share>1, share_type==its type at the + # time) but was since retyped away carries stale share/share_type + # metadata -- graph.leaf_share's docstring: any retype silently + # invalidates a stale share, guarded everywhere by share_type==type. But + # _collapse_value probes a hypothetical candidate by temporarily + # overwriting leaf.type, and graph.leaf_share reads that overwritten + # type -- so a stale share_type that happens to equal the CANDIDATE code + # must not spuriously reactivate; only the leaf's own real current type + # may legitimately carry a live share. + conf = _conf({"b1": {"size": [16.0, 4.0], "width": [4.0, 1.0], "proportion": [1.5, 0.5]}}, + leaf_sharing=True) + fit = Fitness(conf=conf) + prog = fit._programme + forbid, fail_w = fit._COLLAPSE_FORBID, fit._COLLAPSE_FAIL_W + + stale_leaf, _ = _two_leaf_root("other", "other").leaves() + stale_leaf.type = "other" + stale_leaf.share = 3 + stale_leaf.share_type = "b1" # stale: leaf is not currently typed "b1" + val_stale = fit._collapse_value(stale_leaf, "b1", 0, prog, "quality", forbid, fail_w) + + clean_leaf, _ = _two_leaf_root("other", "other").leaves() + clean_leaf.type = "other" # share stays at the default 1 / share_type None + val_clean = fit._collapse_value(clean_leaf, "b1", 0, prog, "quality", forbid, fail_w) + + assert val_stale == val_clean + + # But the leaf's OWN current type still legitimately carries a live share. + live_leaf, _ = _two_leaf_root("b1", "b1").leaves() + live_leaf.type = "b1" + live_leaf.share = 3 + live_leaf.share_type = "b1" + val_live_self = fit._collapse_value(live_leaf, "b1", 0, prog, "quality", forbid, fail_w) + assert val_live_self != val_clean + + +def test_collapse_global_dump_reload_agree_with_stale_share(tmp_path): + # End-to-end regression for the bug: a stale share/share_type surviving + # in-memory but dropped by dom.dump/dom.load (dom._emit only serialises + # share while share_type==type) must not change collapse_global's + # relabelling -- before the fix it did, because the stale metadata leaked + # into the Hungarian assignment's candidate valuation and swayed it to + # relabel the WRONG leaf (right, physically a poor fit for "b2") instead + # of the size-appropriate one, purely because right's stale share_type + # happened to equal that candidate code. + from homemaker_layout import dom + + conf = _conf({ + "b1": {"size": [16.0, 4.0], "width": [4.0, 1.0], "proportion": [1.5, 0.5]}, + "b2": {"size": [10.8, 2.0], "width": [3.5, 0.8], "proportion": [1.5, 0.5]}, + }, leaf_sharing=True) + + def _make_root(): + root = _two_leaf_root("b1", "b1") + _left, right = root.leaves() + right.share = 2 + right.share_type = "b2" # stale: right is currently typed "b1", not "b2" + return root + + live = _make_root() + Fitness(conf=conf).collapse_global(live) + + path = tmp_path / "stale_share.dom" + dumped = _make_root() + dom.dump(dumped, str(path)) + reloaded = dom.load(str(path)) + Fitness(conf=conf).collapse_global(reloaded) + + live_types = [lf.type for lf in live.leaves()] + reloaded_types = [lf.type for lf in reloaded.leaves()] + assert live_types == reloaded_types + # And it's the size-consistent labelling in both cases (left, the smaller + # leaf, takes the smaller-target b2; not the stale-share-swayed choice). + assert live_types == ["b2", "b1"] + + def test_collapse_finish_is_keep_better_and_unmerged(): # collapse_finish returns (tree, base, collapsed, applied); the tree it hands # back is unmerged (leaves still carry their divisions), and collapsed<=base.