From 8328ac1b695b453668f09b4bc9c515be60a798f8 Mon Sep 17 00:00:00 2001 From: Bruno Postle Date: Thu, 23 Jul 2026 18:29:45 +0100 Subject: [PATCH] qi6: full-budget A/B for graded circulation-connectivity signal, negative result MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit conn_grade ON vs OFF (qpk protocol, experiments/run_qi6_ab.sh): harbor-house (budget 2500, seeds 1-3) byte-identical output in every seed — the secondary comparator key never fired. programme-house (budget 3000, seeds 1-5) 3/5 seeds tie exactly; seeds 1/2 diverge to a different topology but the fail delta is adjacency/crinkliness/width/access/size, never connectivity. Zero of 4 cases where a not-connected fail was present got cleared by the grade. Mechanism (b)/(c) (graded proximity as tertiary comparator key) is falsified, not just unconfirmed. Kept default OFF (already was). Closed qi6; filed homemaker-py-8sh for the remaining candidate (mechanism (a): an explicit insert/relocate-circulation operator that doesn't depend on the search stumbling onto a fail-count tie). Co-Authored-By: Claude Sonnet 5 Claude-Session: https://claude.ai/code/session_01GDZjAATDWW1xFfc7xnJqSt --- .beads/issues.jsonl | 37 ++++++++++++------------ DESIGN.md | 42 +++++++++++++++++++++------ experiments/run_qi6_ab.sh | 60 +++++++++++++++++++++++++++++++++++++++ 3 files changed, 113 insertions(+), 26 deletions(-) create mode 100755 experiments/run_qi6_ab.sh diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index 2bbdbaa..a368290 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -22,7 +22,7 @@ {"id":"homemaker-py-1p0","title":"Geometry inner loop: full-objective equal-offset ratio optimiser","description":"DESIGN.md §5.1, §7 Phase 1. Productionise experiments/optimize_fullfitness.py into homemaker: optimise(topology, x0=None) -\u003e (geometry, fitness). DOF = equal-offset division ratios of free branches (solver.free_branches, lowest-storey cut ownership), clipped to [eps, 1-eps]. Objective = full oracle fitness (never a proxy — §4.2 falsified). Must support warm-start x0 (§5.6) and a population/batch evaluation mode so each iteration scores via one batched oracle call (§4.6).","acceptance_criteria":"Reproduces or exceeds §4.5 gains (x1.24–x1.67, no new failures) on 2f45907, candidate-002, c964435; works as a library call on any corpus .dom","status":"closed","priority":1,"issue_type":"feature","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-06-11T23:36:58Z","created_by":"Bruno Postle","updated_at":"2026-06-12T08:46:31Z","started_at":"2026-06-12T00:14:19Z","closed_at":"2026-06-12T08:46:31Z","close_reason":"innerloop.optimise() lands: batched CMA-ES sigma ladder (0.05/0.15, IPOP popsize doubling, deterministic seeding) over equal-offset free-branch ratios vs full oracle fitness; warm-start x0 supported. Acceptance vs unprojected originals: x1.65/x1.66/x1.58 against bars x1.24/x1.67/x1.59, no new failures, 46 oracle calls vs NM's 200. Two near-bar results accepted as reproduced-within-noise (1% tol) — draw spread brackets the single-NM-draw bars; approved by Bruno 2026-06-12. Gotchas: equal-offset projection of legacy unequal cuts loses fitness/adds failures (midpoint projection used); pycma seed=0 means clock-seeded.","dependencies":[{"issue_id":"homemaker-py-1p0","depends_on_id":"homemaker-py-av5","type":"blocks","created_at":"2026-06-12T00:39:33Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":3,"comment_count":0} {"id":"homemaker-py-8cs","title":"Experiment: warm-vs-cold start of inner loop (Lamarckian inheritance)","description":"DESIGN.md §5.6, §4.6. Warm-starting a child topology's inner loop from the parent's optimised ratios is the main lever for cutting per-topology cost (~3 min/topology cold). Apply single topology mutations to optimised corpus designs, re-optimise warm (surviving cuts keep values, new cuts get heuristic defaults) vs cold, compare oracle-call counts to convergence at equal final fitness.","acceptance_criteria":"Speedup factor measured across \u003e=10 mutated topologies; decision recorded (expect order-of-magnitude; if \u003c2x, revisit §4.6 Phase-2 scoping)","notes":"Experiment script committed (experiments/warm_vs_cold.py, 1cc86c8) and machinery validated oracle-free; one mutated child scored through the oracle OK. Waiting on homemaker-py-gp2 reference run to finish, then execute under URB_NO_OCCLUSION=1 (3 parents x 400 evals + 12 children x 2 x 200 evals, ~1.5-2 h oracle time). Default budgets: parent 400, child 200; target = evals to 95% of best final.","status":"closed","priority":1,"issue_type":"task","owner":"bruno@postle.net","created_at":"2026-06-11T23:36:58Z","created_by":"Bruno Postle","updated_at":"2026-06-12T11:44:45Z","closed_at":"2026-06-12T11:44:45Z","close_reason":"Measured (URB_NO_OCCLUSION=1, parent budget 400, child 200, 12 single mutations across 3 designs): cold start reached 95% of warm final in 0/12 cases within budget — speedup unbounded at practical budgets; warm finals beat cold finals x1.2-x4 in 12/12; 6/12 warm starts were within 95% at 1 eval (near-neutral mutations). Decision: Lamarckian warm-starting is MANDATORY in the memetic driver (homemaker-py-b39), not an optimisation; cold starts produce strictly worse geometry at equal budget. Note: 2 undivides were exactly fitness-neutral (same-type merge == Merge_Divided equivalence) — locality datum for homemaker-py-nyb.","dependencies":[{"issue_id":"homemaker-py-8cs","depends_on_id":"homemaker-py-1p0","type":"blocks","created_at":"2026-06-12T00:39:34Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-av5","title":"Batched oracle: score many .dom files per invocation","description":"oracle.py currently scores one .dom per urb-fitness.pl call (~1.65 s/dom). DESIGN.md §4.6: batching amortises Perl startup to ~0.99 s/dom and is required so population/batch optimisers can score a whole generation in one oracle call. Extend oracle.py with a batch API: write N .dom files, one perl invocation, parse N .score/.fails pairs. Keep the single-file path for compatibility.","acceptance_criteria":"Batch of 35 corpus files scores in one perl invocation; per-file results identical to single-file calls; measured s/dom reported","status":"closed","priority":1,"issue_type":"task","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-06-11T23:36:56Z","created_by":"Bruno Postle","updated_at":"2026-06-12T00:14:06Z","started_at":"2026-06-11T23:50:40Z","closed_at":"2026-06-12T00:14:06Z","close_reason":"score_batch() lands in oracle.py; 35-file corpus parity verified single-vs-batch (1e-12 rel fitness, exact fail sets); 0.98 s/dom batched vs 1.27 single, x1.30","dependency_count":0,"dependent_count":1,"comment_count":0} -{"id":"homemaker-py-qi6","title":"Circulation placement to clear not-connected / access fails","description":"The 94g finish-time collapse cannot touch the \"not-connected\" (level N not\nconnected) and related access/inaccessible fails: they are properties of the\nCIRCULATION skeleton (c/o/s cells), which the collapse deliberately never\nrelabels (they form the structure the room assignment is layered onto). On the\nharbor-house best layout 2 of the 15 fails are not-connected (levels 0 and 1);\nthese are out of scope for any label optimisation.\n\nGOAL: a search/repair step that places or reshapes circulation so every usable\nspace is reachable and each storey's circulation graph is connected. Candidate\nmechanisms: (a) a mutation/operator that inserts a circulation cell to bridge a\ndisconnected component (graph.py already computes connected components +\nconnected_circulation); (b) a finish-time repair that re-types a boundary cell to\ncirculation where it reconnects the graph at least net-fail cost; (c) bias the\nouter search toward connected topologies via the graded signal.\n\nInteracts with 94g: circulation placement changes which cells are skeleton vs\nassignable, so it should run BEFORE the label collapse (collapse then optimises\nlabels over the improved skeleton). Also interacts with the public-access pin\n(94g) — better circulation placement can supply invariant inside public access,\nremoving the need to pin a room provider.\n\nMeasure on the 6 evolved layouts from the 94g sweep (not-connected + access +\ninaccessible fail counts). Related: 94g, homemaker-py-2g5.","notes":"LANDED (2026-07-18) mechanism (c) — graded circulation-connectivity signal (DESIGN §18). graph.circulation_connectivity(G) = largest-circ-component fraction [0,1]; summed over storeys it rides the score_with_grade proximity channel, gated by conf flag conn_grade (replaces the §11.4 leaf-grade on that channel). Secondary comparator key (-n_fails, grade, fitness) only — scalar fitness and fail count byte-identical (verified). Wired conn_grade through driver _overrides_for/_fitness_for/_evaluate/search (enabling it implies the grade key); evolve --conn-grade (HOMEMAKER_CONN_GRADE, default OFF). 9 new tests (tests/test_conn_grade.py), 276 pass; 60-eval CLI smoke confirms plumbing. PENDING: full-budget A/B to confirm the gradient actually pulls runs toward connected circulation and clears 'not connected' fails; if graded key alone insufficient, follow-on is an insert/relocate-circulation mutation operator (mechanism a) which now has a gradient to climb. Keeping in_progress until the A/B verdict.","status":"in_progress","priority":2,"issue_type":"feature","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-07-18T10:12:27Z","created_by":"Bruno Postle","updated_at":"2026-07-18T17:06:53Z","started_at":"2026-07-18T12:17:08Z","dependencies":[{"issue_id":"homemaker-py-qi6","depends_on_id":"homemaker-py-94g","type":"discovered-from","created_at":"2026-07-18T11:12:27Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} +{"id":"homemaker-py-qi6","title":"Circulation placement to clear not-connected / access fails","description":"The 94g finish-time collapse cannot touch the \"not-connected\" (level N not\nconnected) and related access/inaccessible fails: they are properties of the\nCIRCULATION skeleton (c/o/s cells), which the collapse deliberately never\nrelabels (they form the structure the room assignment is layered onto). On the\nharbor-house best layout 2 of the 15 fails are not-connected (levels 0 and 1);\nthese are out of scope for any label optimisation.\n\nGOAL: a search/repair step that places or reshapes circulation so every usable\nspace is reachable and each storey's circulation graph is connected. Candidate\nmechanisms: (a) a mutation/operator that inserts a circulation cell to bridge a\ndisconnected component (graph.py already computes connected components +\nconnected_circulation); (b) a finish-time repair that re-types a boundary cell to\ncirculation where it reconnects the graph at least net-fail cost; (c) bias the\nouter search toward connected topologies via the graded signal.\n\nInteracts with 94g: circulation placement changes which cells are skeleton vs\nassignable, so it should run BEFORE the label collapse (collapse then optimises\nlabels over the improved skeleton). Also interacts with the public-access pin\n(94g) — better circulation placement can supply invariant inside public access,\nremoving the need to pin a room provider.\n\nMeasure on the 6 evolved layouts from the 94g sweep (not-connected + access +\ninaccessible fail counts). Related: 94g, homemaker-py-2g5.","notes":"LANDED (2026-07-18) mechanism (c) — graded circulation-connectivity signal (DESIGN §18). graph.circulation_connectivity(G) = largest-circ-component fraction [0,1]; summed over storeys it rides the score_with_grade proximity channel, gated by conf flag conn_grade (replaces the §11.4 leaf-grade on that channel). Secondary comparator key (-n_fails, grade, fitness) only — scalar fitness and fail count byte-identical (verified). Wired conn_grade through driver _overrides_for/_fitness_for/_evaluate/search (enabling it implies the grade key); evolve --conn-grade (HOMEMAKER_CONN_GRADE, default OFF). 9 new tests (tests/test_conn_grade.py), 276 pass; 60-eval CLI smoke confirms plumbing.\n\nA/B VERDICT (2026-07-22, qpk protocol, experiments/run_qi6_ab.sh) — NEGATIVE. conn_grade ON vs OFF, full-budget, both finished with --collapse: harbor-house (budget 2500, seeds 1-3) byte-identical output in every seed — the grade never fired. programme-house (budget 3000, seeds 1-5) 3/5 seeds tie exactly; seeds 1/2 diverge to a different topology with one fewer total fail, but the diff is adjacency/crinkliness/width/access/size, not connectivity. Zero cases (of 4) where a not-connected fail was present and cleared by the grade. Mechanism (b)/(c) (graded proximity as tertiary comparator key) is falsified, not just unconfirmed. Kept default OFF (already was). DESIGN.md §18 updated with full verdict.\n\nRemaining candidate: mechanism (a), an explicit insert/relocate-circulation mutation/repair operator that doesn't depend on the search stumbling onto a fail-count tie. Not started — filing as follow-on if this gets picked up; otherwise low priority (fitness fidelity, not search capability, per 94g framing).","status":"in_progress","priority":2,"issue_type":"feature","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-07-18T10:12:27Z","created_by":"Bruno Postle","updated_at":"2026-07-23T07:08:07Z","started_at":"2026-07-18T12:17:08Z","dependencies":[{"issue_id":"homemaker-py-qi6","depends_on_id":"homemaker-py-94g","type":"discovered-from","created_at":"2026-07-18T11:12:27Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-7fm","title":"Geometry/topology search for shape-intrinsic fails (long-thin useless cells)","description":"The finish-time collapse (94g) proved that a large share of the harbor-house best\nlayout's residual fails are GEOMETRY-INTRINSIC, not label slack: long-thin cells\nthat are useless whatever room usage is assigned. Their width / proportion /\ncrinkliness fails cannot be cleared by any relabelling (94g searches labels only,\nnever geometry) — confirmed by collapse_global clearing only ~2-3 of 15 fails on\nthe best layout, the rest shape- or building-level bound.\n\nGOAL: a search/repair operator that reshapes such cells so the space becomes\nusable — e.g. re-solving a subtree's division ratios, merging a sliver into a\nneighbour, or a targeted division-ratio mutation biased by the offending factor\n(narrowest-width, aspect, crinkliness). Unlike the collapse this MUST move\ngeometry (division ratios / tree shape), and must be evaluated for net fail-count\neffect (a reshape that fixes width may add size elsewhere — same shuffle risk the\n94g threshold objective addressed for labels).\n\nSCOPE: width/proportion/crinkliness fails on inside room cells whose geometry no\nroom type can satisfy. Explicitly NOT relabelling (that is 94g, done). Candidate\nmechanisms: (a) inner-loop solve already optimises ratios — check why it leaves\ndegenerate cells (local optimum? target-dim conflict?); (b) a finish-time\n\"deslim\" operator + re-solve; (c) an operators.py mutation weighted toward high-\naspect leaves. Measure on the same 6 evolved layouts used for the 94g sweep.\n\nSee bd memory collapse-global-94g-and-any-label-usage-optimisation for the\nlabel-vs-geometry boundary. Related: 94g (label collapse), homemaker-py-2g5\n(occlusion/daylight rebuild, which feeds crinkliness).","design":"DONE (negative) — see DESIGN.md §19 for full writeup.\n\nDiagnosis ruled out mechanism (a): re-running the ratio inner loop with 1500\nevals (vs ~80-200 in real search) on the 12-fail collapsed best layout made\nzero difference. Traced two structural causes instead: (1) area starvation\nseveral levels up the tree, (2) cut orientation parallel to the parent's long\naxis. Neither is a ratio problem.\n\nImplemented mutate_shape_rotate + mutate_deslim (operators.py) targeting each\ncause, gated on a new `fit` argument (mirrors the existing `reqs` gate\npattern). Tested as a finish-time exhaustive hill-climb (all 3 rotations +\ndeslim+reinsert per failing cut, keep-better) on the same 6 harbor-house\nlayouts §17 (94g) swept: 0 improving moves found, anywhere, under any\nvariant. Root cause: on a co-evolved layout the cut that makes a leaf thin is\nalso providing some other leaf's adjacency/public-access — straightening it\nelsewhere isn't free. This is §4.2's lesson (partial-objective repair of a\nco-evolved optimum can't win) confirmed for topology repair, not just ratio\nsolving.\n\nCode landed: operators.py (mutate_shape_rotate, mutate_deslim, _shape_failing,\nMUTATIONS/mutate() wiring), tests/test_operators.py (4 new tests + automatic\ncoverage via test_mutations_yield_canonical_genomes). 282 tests pass. Both\noperators are currently unreachable from driver.search/evolve.py (no `fit`\nthreaded through) since the finish-time evaluation found nothing worth\nwiring up further.\n\nOpen question spun out separately: whether these operators help as in-search\nGA moves, where selection pressure across generations might accept a\nlocally-worse move a later step completes — a different regime from\nsingle-step greedy hill-climbing. See follow-up issue.","status":"closed","priority":2,"issue_type":"feature","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-07-18T10:11:43Z","created_by":"Bruno Postle","updated_at":"2026-07-19T10:06:01Z","started_at":"2026-07-19T07:00:32Z","closed_at":"2026-07-19T10:06:01Z","close_reason":"Closed","dependencies":[{"issue_id":"homemaker-py-7fm","depends_on_id":"homemaker-py-94g","type":"discovered-from","created_at":"2026-07-18T11:11:42Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-94g","title":"Global cell↔room collapse: generalise 9o5 matching to all spaces (WFC-style)","description":"Generalise the 9o5 superposition/collapse from interchange-classes to a GLOBAL cell↔room assignment: evolution searches unlabelled floorplans (tree shape + circulation/room/outside), a collapse function optimally LABELS each candidate. Mechanically an assignment problem — N cells (computed area/width/proportion/level/graph-pos) ↔ M required rooms (target dims + level + adjacency) — minimising total fail-cost; reuse Fitness._best_assignment (Hungarian/brute) but over the whole leaf set instead of one class. WFC framing = the constructive algorithm: each cell a distribution over types, PROPAGATE hard constraints exactly (level, requires_below service stacks, must-have adjacency) to prune, OBSERVE lowest-entropy cell, collapse to best-fitting room weighted by fit-quality, backtrack on contradiction.\n\nMOTIVATION (harbor-house evolved-3M-nols-3.dom, best layout, 15 fails): ~11 of 15 are LABEL-RELATIVE — 4 size (0/lrrll,0/rllll,1/lrlr,1/lrrll), 3 width (0/rlrrlr,0/rrllrr,0/rrrr), 2 proportion (0/rlrlr,1/rlrlr), 2 wrong-level (me1 L1-\u003e0, r L0-\u003e1). A cell fails size/width/proportion because the room ASSIGNED to it wants dims the cell lacks; relabel to a room it fits and the fail vanishes. Wrong-level is a HARD constraint a level-respecting collapse honours for free. Only 4 are shape-intrinsic and out of scope: crinkliness x2 (0/llll,0/rlllr) + not-connected x2 (level 0/1). Realistic target: 15 -\u003e ~4-6.\n\nWHY IT MAY SUCCEED WHERE 9o5 WAS FALSIFIED (xi7): 9o5 went negative because (1) auto-derived classes were semantically wrong (harbor-house 8-code chain, see homemaker-py-b3v) and (2) collapse perturbed feasibility (ON ADDS fails, 38v33/48v43). A GLOBAL, hard-constraint-respecting collapse sidesteps both: no fragile similarity classes; never violates level/adjacency/stack so it cannot ADD feasibility fails — only improve or match.\n\nRISK: search-landscape flattening. fitness = max-over-labellings makes the objective flatter/noisier (many topologies collapse to similar best scores), removing the gradient evolution climbs — the likely cause of 9o5's negative verdict, AMPLIFIED at full scope. Mitigations to A/B: (a) collapse only at FINISH (search on committed types, one relabel pass at end — cheap, strictly cannot worsen final score); (b) warm-start collapse from evolved labels as a local polish.\n\nRECOMMENDED FIRST STEP (cheapest, ~1 day, strictly cannot worsen final layout): FINISH-TIME global collapse — after a normal run, one optimal cell-\u003eroom assignment over the full leaf set with hard constraints enforced, then re-score. Measures empirically how many of the 11 label-relative fails are real slack vs already-optimal. If it clears a meaningful chunk, justifies the in-search WFC collapse + the landscape-flattening A/B. Does NOT fix crinkliness/connectivity (need geometry + circulation-placement work, file separately).\n\nFiles: fitness.py (_best_assignment, collapse_superposition), programme.py (constraints), driver.py (finish hook), graph.py (adjacency for propagation).","notes":"WIRED + PUBLIC-ACCESS TERM DONE.\n1. Public-access pin (preserve_public_access=True, default): when the building's\n ONLY street access is an l/k ROOM neighbour of a public outside leaf (no\n circulation fallback — the existential building check the per-leaf objective\n can't see), that room leaf is PINNED (kept + its demand slot decremented) so\n the collapse can't drop \"no outside public access\". On the best layout this\n turns 15-\u003e13 into 15-\u003e12 (the lone regression removed, zero new fails). Sweep\n total 172-\u003e171, still monotone across all 6.\n2. Keep-better wrapper Fitness.collapse_finish(root, **kw) -\u003e (tree, base, coll,\n applied): scores on throwaway copies (score_with_fails merges in place),\n returns collapsed only if fails don't increase. Safety belt (config already\n monotone here, not proven so in general).\n3. Wiring: driver.collapse_best(result, programme_dir, ...) updates result.best\n (lineage +collapse, canonical re-score). evolve.py runs it after the sharing\n polish behind --collapse/--no-collapse (default ON). Standalone CLI\n homemaker-collapse file.dom (pyproject entry) writes \u003cstem\u003e.collapsed.dom;\n flags --adjacency/--public-access/--objective/--keep-better. Verified the\n written dom independently re-scores 15-\u003e12.\n4. Tests: tests/test_collapse_global.py (6) — demand-set relabel, level hard\n constraint, c/o/s exclusion, no-op safety, keep-better/unmerged. 267 pass.\n\nSTILL OPEN (separate work, not label slick): geometry-intrinsic fails — long-thin\nuseless cells (width/proportion/crinkliness) and not-connected — need geometry/\ntopology + circulation-placement search, NOT the collapse (see bd memory\ncollapse-global-94g...). In-search WFC collapse A/B also still open (xi7 risk).","status":"closed","priority":2,"issue_type":"feature","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-07-17T16:17:12Z","created_by":"Bruno Postle","updated_at":"2026-07-18T10:13:43Z","started_at":"2026-07-17T21:07:05Z","closed_at":"2026-07-18T10:13:43Z","close_reason":"FINISH-TIME global cell→room collapse DELIVERED (commits d52cce6, da18ef7, 880a214):\nFitness.collapse_global (c/o/s partition, level hard constraint, adjacency\nrelaxation, threshold objective, public-access pin) + collapse_finish keep-better\nwrapper + driver.collapse_best + evolve --collapse hook + homemaker-collapse CLI +\n6 tests (267 pass). Best harbor-house layout 15→12 fails; monotone across 6 evolved\nlayouts (sweep 195→171). Label search only — cannot fix geometry/building-intrinsic\nfails.\n\nRemaining scope spun out: homemaker-py-7fm (shape-intrinsic reshape), homemaker-py-qi6\n(circulation placement / not-connected), homemaker-py-qpk (in-search WFC collapse\nA/B, xi7 landscape-flattening risk). Closing 94g as the finish-time thrust is done.","dependencies":[{"issue_id":"homemaker-py-94g","depends_on_id":"homemaker-py-b3v","type":"related","created_at":"2026-07-17T17:23:20Z","created_by":"Bruno Postle","metadata":"{}"},{"issue_id":"homemaker-py-94g","depends_on_id":"homemaker-py-xi7","type":"related","created_at":"2026-07-17T17:23:06Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-3l6","title":"Leaf-sharing default makes internal fitness diverge from canonical homemaker-fitness score","description":"Default is --leaf-sharing (on). Leaf sharing is a fitness-evaluation knob (fitness.py:415 quality_size, plus edge cap and count check): a shared leaf of code X is credited as satisfying k programme entries, with its size Gaussian re-centred on k*target. The evolve internal objective therefore rewards genomes that under-materialise the programme. When the winning .dom is re-scored by the canonical homemaker-fitness (leaf_sharing off), those un-materialised copies become 'missing required space ... (critical)' fails.\n\nObserved on examples/harbor-house (init.dom, budget 3M, workers 2):\n - leaf sharing ON (default): internal best 1.03e-05, but canonical score 6.73e-29 with 90 fails (15 critical missing-room).\n - --no-leaf-sharing (warm-started to full budget): internal and canonical agree at 4.19e-06, 15 fails, 0 critical -- ~9500x better than the prior best 3m.dom (4.41e-10).\n\nSo the default silently optimises an objective the canonical scorer does not credit, and writes a catastrophically worse .dom than its reported internal fitness implies.\n\nOptions to consider:\n 1. Make --no-leaf-sharing the default (strict per-leaf baseline agrees with canonical scorer).\n 2. Before writing output, re-score best-so-far with leaf_sharing off and warn (or refuse) if it regresses vs internal fitness.\n 3. Materialise/unfold shared leaves into k distinct rooms when writing the .dom, so the output satisfies the per-room programme.\n 4. Keep sharing as an early-phase relaxation only and anneal leaf_share_factor down to 0 before finishing (see related annealing investigation).","notes":"FIXED (option 3+2 combined, auto-finish): leaf-sharing runs now unfold+polish+rescore before write so output is honest under the canonical scorer.\n\nImplementation:\n- driver.polish_finish(result, programme_dir, polish_budget, ...): deep-copies best, operators.unfold_shared_leaves() to materialise the count deficit, then warm-starts a leaf_sharing=False search (bootstrap=False) from the unfolded genome. polish_budget\u003c=0 -\u003e single rescore eval only (used on interrupt). Stitches evals/topologies/sigs/restarts/history onto the sharing run; history tagged share:/polish: since the two objectives are not comparable. Returned best.fitness is canonical (leaf_sharing off =\u003e internal==canonical).\n- evolve.py: new --polish-budget flag (env HOMEMAKER_POLISH_BUDGET, default -1=auto=budget//2, 0=unfold+rescore only). main() calls polish_finish when --leaf-sharing on; interrupt forces polish_budget=0 for a fast honest output.\n\nVerified end-to-end (harbor-house, budget 3000 + polish 1500): reported polish fitness 4.79788e-27 EXACTLY matches canonical homemaker-fitness, 0 critical fails (was: internal 1.03e-05 vs canonical 6.73e-29 w/ 15 critical). Tiny budget so absolute quality low but honesty restored. Tests: driver.polish_finish x3 (test_driver.py), full suite 254 pass.\n\nDefault kept --leaf-sharing on per decision (sharing's topology-search speed retained; output made honest by the finish). Schedule B in-run annealing remains kpu.","status":"closed","priority":2,"issue_type":"bug","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-07-05T16:24:34Z","created_by":"Bruno Postle","updated_at":"2026-07-15T08:00:26Z","started_at":"2026-07-15T06:56:29Z","closed_at":"2026-07-15T08:00:26Z","close_reason":"Closed","dependency_count":0,"dependent_count":1,"comment_count":0} @@ -55,6 +55,7 @@ {"id":"homemaker-py-nyb","title":"High-locality topology operators (mutation + subtree crossover)","description":"DESIGN.md §5, §7 Phase 2, §8.4. Mutation moves: divide/undivide leaf, swap children, rotate cut, retype leaf, per-floor delta edits, storey add/delete (cf. Urb Mutate.pm — but geometry sliding belongs to the inner loop, not the operator set). Crossover: area-matched subtree exchange (a subtree = a contiguous region, so crossover is meaningful — Crossover.pm). Operators must be high-locality: small genome change =\u003e small phenotype change, so warm-started inner loops stay cheap.","acceptance_criteria":"Each operator produces valid genomes (oracle scores them without error); locality measured (mean fitness/geometry perturbation per operator)","status":"closed","priority":2,"issue_type":"feature","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-06-11T23:37:27Z","created_by":"Bruno Postle","updated_at":"2026-06-12T13:07:37Z","started_at":"2026-06-12T12:54:23Z","closed_at":"2026-06-12T13:07:37Z","close_reason":"operators.py lands: 7 mutations + area-matched crossover, valid-by-construction via genome.encode repair. 115/115 oracle-valid children; locality measured: geom-pert 0.07-0.33 per op, fitness-pert 0.68-0.99 (0.5^n cliff flags raw moves — warm restart + penalty reshaping confirmed load-bearing). Also fixed dom._link stale below-links on structural mutation.","dependencies":[{"issue_id":"homemaker-py-nyb","depends_on_id":"homemaker-py-k2g","type":"blocks","created_at":"2026-06-12T00:39:36Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} {"id":"homemaker-py-k2g","title":"Topology genome: base-floor tree + per-floor deltas + type assignment","description":"DESIGN.md §5.2, §7 Phase 2. Genome = base-floor slicing topology (primary) + per-leaf type assignment + per-floor divide/undivide deltas (Below-inheritance as regulariser; cut owned by lowest storey where its path is divided — §10). Must round-trip to/from dom.py Node trees so the oracle and inner loop consume it directly. Includes storey count and per-floor type overrides.","acceptance_criteria":"Genome \u003c-\u003e .dom round-trip on all 35 corpus files preserves fitness; multi-storey wall stacking preserved","status":"closed","priority":2,"issue_type":"feature","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-06-11T23:37:26Z","created_by":"Bruno Postle","updated_at":"2026-06-12T12:52:34Z","started_at":"2026-06-12T10:55:21Z","closed_at":"2026-06-12T12:52:34Z","close_reason":"genome.py encode/decode lands. 35/35 oracle fitness parity after round-trip (flag-on); genome fixed-point + owned-projection tests. Dead-field discovery: corpus upper storeys carry drifted dead divisions (97) and rotations (187) — canonicalised by decode, validated fitness-neutral.","dependency_count":0,"dependent_count":1,"comment_count":0} {"id":"homemaker-py-d0s","title":"Experiment: inner-loop optimiser bake-off at equal oracle budgets","description":"DESIGN.md §7 Phase 1, §8.3. DOF is only ~rooms-1 (6–7 on corpus). Compare Nelder-Mead vs CMA-ES vs batched multi-start pattern search at equal oracle-call budgets, measuring fitness gained per oracle call and wall-clock (batch-friendliness matters — §4.6). Measure, don't commit blind.","acceptance_criteria":"Table of fitness-per-budget across \u003e=3 candidates; one optimiser chosen and recorded in DESIGN.md","status":"closed","priority":2,"issue_type":"task","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-06-11T23:36:59Z","created_by":"Bruno Postle","updated_at":"2026-06-13T08:48:13Z","started_at":"2026-06-12T21:22:15Z","closed_at":"2026-06-13T08:48:13Z","close_reason":"Bake-off complete: CMA-ES confirmed as Phase 1/2 optimiser. NM wins quality per eval but sequential architecture incompatible with batching (§4.6). Compass stalls on narrow valleys. Results in DESIGN.md §8.3 and experiments/bakeoff_innerloop.*","dependencies":[{"issue_id":"homemaker-py-d0s","depends_on_id":"homemaker-py-1p0","type":"blocks","created_at":"2026-06-12T00:39:35Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} +{"id":"homemaker-py-8sh","title":"Insert/relocate-circulation mutation operator for not-connected fails","description":"Follow-on from homemaker-py-qi6, mechanism (a). The graded circulation-connectivity\nsignal (mechanism (b)/(c), DESIGN.md §18) was A/B tested full-budget and measured\nNEGATIVE: it never fired on harbor-house (byte-identical output ON vs OFF across 3\nseeds) and on programme-house it only ever perturbed unrelated fails, never a\nnot-connected fail (0/4 cases cleared). A finish-time repair (bridge cells into\ncirculation) was also measured negative earlier (195-\u003e560 fails, see §18 history).\n\nGOAL: an explicit search-time mutation/repair operator that inserts or relocates a\ncirculation cell to bridge a disconnected component directly, rather than relying\non the outer GA to discover connectivity via a comparator-key gradient (which this\nissue's A/B showed doesn't work) or a finish-time relabel (which the earlier\nprototype showed is too late/costly). graph.py already computes connected\ncomponents + connected_circulation to detect where bridging is needed.\n\nMeasure on the 6 evolved layouts from the 94g sweep (not-connected + access +\ninaccessible fail counts), same protocol as qi6.","status":"open","priority":3,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-07-23T17:21:37Z","created_by":"Bruno Postle","updated_at":"2026-07-23T17:21:37Z","dependency_count":0,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-161","title":"In-search evaluation of shape_rotate/deslim GA operators (7fm follow-up)","description":"homemaker-py-7fm's finish-time hill-climb found zero improving moves for\nmutate_shape_rotate/mutate_deslim (operators.py) on the 6-layout harbor-house\nsweep — every candidate move traded a shape fail for a new adjacency/access\nfail on the already co-evolved layout. That's a different regime from\nin-search use: a full multi-generation GA run gives selection pressure and\npopulation diversity a chance to accept a locally-worse move that a later\nstep or recombination completes.\n\nTo test: thread `fit` through driver.search (currently only reqs is passed to\noperators.mutate; shape_rotate/deslim need `fit` and currently no-op inside\nthe GA). Gate with an enable_shape_repair-style flag, mirroring how\nenable_reassociate (§12.3) let 9gp.2 do a clean A/B. Run full-budget\nharbor-house search with/without across seeds, compare final fail counts.\n\nIf negative again, the geometry-intrinsic residual on harbor-house-scale\nprogrammes may be a genuine floor for this representation, not a repairable\ninefficiency (consistent with §17/§19's framing). See DESIGN.md §19 and bd\nmemory collapse-global-94g-and-any-label-usage-optimisation for full context.","notes":"RESULT (negative, confirms 7fm): full sweep at budget=1,000,000, pop=16, child_budget=80, workers=4, harbor-house/init.dom cold-start, 4 seeds (0-3):\n\nenable_shape_repair=False (baseline): fails [14,15,12,17] mean=14.50, fitness mean=1.741e-05\nenable_shape_repair=True: fails [17,14,16,12] mean=14.75, fitness mean=1.24e-05\n\nNo improvement from threading fit into the GA and letting shape_rotate/deslim fire in-search — mean fails is marginally WORSE with the operators enabled, and the 0.25 delta is far inside the seed-to-seed spread (12-17) in both arms. Matches the smaller pilot (budget=20000, 3 seeds: off mean=31.33, on mean=32.00) at a different scale, and matches 7fm's finish-time hill-climb finding that these operators trade one fail for another on already-co-evolved harbor-house layouts.\n\nConclusion: in-search selection pressure and population diversity do NOT rescue shape_rotate/deslim on harbor-house-scale programmes. Supports DESIGN.md §19's framing — the residual fails here look like a genuine floor for this representation on this programme, not a repairable inefficiency reachable by richer local operators. Consistent with bd memory collapse-global-94g-and-any-label-usage-optimisation (geometry-intrinsic fails need geometry/topology search, not local repair or relabeling).\n\nCode kept (not reverted): driver.search()/search_staged() gained enable_shape_repair: bool = False, threading a cached fitness.Fitness instance into operators.mutate() only when set, mirroring the enable_reassociate clean-toggle pattern. Default off reproduces prior runs byte-for-byte. Test: test_enable_shape_repair_threads_fit_into_mutate in tests/test_driver.py. Kept for reuse/reproducibility per the enable_reassociate precedent, not because the flag should default on.","status":"closed","priority":3,"issue_type":"task","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-07-19T10:05:51Z","created_by":"Bruno Postle","updated_at":"2026-07-22T15:54:47Z","started_at":"2026-07-19T19:57:48Z","closed_at":"2026-07-22T15:54:47Z","close_reason":"Closed","dependency_count":0,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-qpk","title":"In-search WFC collapse: run collapse_global per-eval during search (A/B vs finish-time)","description":"94g landed and validated the FINISH-TIME global cell-\u003eroom collapse (label search\nover a fixed geometry, monotone, best layout 15-\u003e12). The original 94g thrust was\na per-eval IN-SEARCH collapse: evolution searches unlabelled floorplans and the\nfitness collapses (optimally labels) each candidate before scoring, so search\noptimises the condensed objective directly. This issue is that step.\n\nMECHANISM: call collapse_global (or a cheaper incremental variant) inside\n_evaluate_full before the checks, same as the 9o5 collapse_superposition hook\n(fitness.py:_evaluate_full gates on self._superpose). Reuse the 94g substrate:\nc/o/s partition, level hard constraint, adjacency relaxation, public-access pin,\nthreshold objective.\n\nRISK (carried from homemaker-py-xi7, why 9o5 went negative): fitness =\nmax-over-labellings flattens/roughens the landscape — many topologies collapse to\nsimilar best scores, removing the gradient evolution climbs. AMPLIFIED at full\n(global) scope. The finish-time result does NOT de-risk this: finish-time is\nstrictly cannot-worsen by construction; in-search changes the objective every\neval. MUST A/B superpose-global ON vs OFF with a relaxation-gap log, exactly like\nxi7 did for 9o5, before adopting.\n\nCOST: collapse_global builds graphs + an assignment relaxation per eval — far more\nthan 9o5's per-class collapse. Needs an incremental/cheap variant or caching to be\naffordable in the inner loop; profile first.\n\nPrereq ordering: circulation placement (homemaker-py-qi6) and shape repair change\nthe skeleton/geometry the collapse labels over, so ideally sequence those first.\nRelated: 94g (finish-time, done), xi7 (9o5 A/B + relaxation-gap log), 9o5.","notes":"A/B VALIDATION COMPLETE (2026-07-19), xi7 protocol, 4 workers, equal eval\nbudget, both arms finished with standard --collapse (94g finish-time) so\ncomparison is on the final COLLAPSED score:\n\nharbor-house (init.dom, budget=2500, seeds 1-3): ON WINS 3/3.\n mean fails 80.3 -\u003e 72.0 (s1 85-\u003e74, s2 76-\u003e65, s3 80-\u003e77). No losses.\nprogramme-house (init.dom, budget=3000, seeds 1-5): ON wins 3/5.\n mean fails 8.4 -\u003e 7.8 (s1 8-\u003e5, s2 8-\u003e7, s4 10-\u003e9 win; s3 8-\u003e9, s5 8-\u003e9\n loss by 1 fail). Weaker/noisier on this much smaller building (already\n near its geometry floor, see section13/section19).\nCOMBINED head-to-head: ON 6, OFF 2.\n\nVERDICT: POSITIVE, and the OPPOSITE of the 9o5/xi7 prior (which was\nNULL/NEGATIVE for the per-class interchange relaxation). Unlike 9o5, this is\nthe SAME global WFC-style matching section17/94g already proved\nmonotone/positive at finish time -- running it every eval lets the outer\nsearch see the condensed objective instead of discovering it only once, and\nthe gradient survives rather than flattening. Effect scales WITH building\nsize (harbor-house clean 3/3 vs programme-house mixed), opposite of the 9o5\nlandscape-flattening fear.\n\nCOST: harbor-house wall-clock 102.6s(OFF)-\u003e177.8s(ON) ~1.73x;\nprogramme-house 39.0s-\u003e43.7s ~1.12x. Matches the profiled 1.5-1.9x/eval\nfigure.\n\nDECISION: kept default OFF (programme-house sample too mixed/small to flip\ndefault; 9o5/xi7 scar warrants a second larger-budget confirmation first),\nbut --collapse-insearch is a genuine, tested, working opt-in for\nharbor-house-scale-or-larger programmes. DESIGN.md section20 has full\nwriteup + per-seed numbers. Not filing a follow-up issue -- a larger-N\nprogramme-house seed sweep would be the natural next step if this gets\nrevisited, noted in DESIGN.md as a low-priority idea, not a blocker.\n\nRaw run logs/doms: /tmp scratchpad qpk_ab/ (not committed, ephemeral).","status":"closed","priority":3,"issue_type":"feature","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-07-18T10:12:54Z","created_by":"Bruno Postle","updated_at":"2026-07-19T19:18:44Z","started_at":"2026-07-19T10:11:11Z","closed_at":"2026-07-19T19:18:44Z","close_reason":"A/B validation complete: POSITIVE (opposite of 9o5/xi7 prior). Harbor-house ON wins 3/3 (mean fails 80.3-\u003e72.0); programme-house mixed 3/5. Combined 6/2. Kept default OFF pending a larger programme-house sample, but --collapse-insearch is a working, documented opt-in. See DESIGN.md section20.","dependencies":[{"issue_id":"homemaker-py-qpk","depends_on_id":"homemaker-py-94g","type":"discovered-from","created_at":"2026-07-18T11:12:54Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-kpu","title":"Schedule B: in-run leaf-sharing annealing (ramp grain down, unfold at each step)","description":"Spun out of homemaker-py-yaa, whose investigation is complete. yaa proved Schedule A (two-phase warm-start) works ONLY when shared leaves are unfolded at the sharing-\u003eno-sharing transition: naive warm-start stalls at 8.66e-08/70 fails, but unfold-then-de-share reaches 4.19e-06/15 fails — matching the direct --no-leaf-sharing baseline. operators.unfold_shared_leaves() is built, tested, and proven.\n\nSchedule B is the in-run variant: instead of a manual two-phase chain, anneal leaf_share_factor down within a single driver run (e.g. 4-\u003e3-\u003e2-\u003eoff) at eval thresholds. At each grain transition: (1) rebuild the cached (dir,sharing) evaluator at the new grain, (2) UNFOLD shared leaves that drop below the new grain so the population stays materialised (reuse operators.unfold_shared_leaves), (3) re-evaluate the whole population under the new evaluator, (4) resume local search. Gradual grain ramp = graduated non-convexity: avoids a single fitness cliff, keeps gross topology fixed on the smaller effective problem early, polishes per-room size/proportion/width late.\n\nDriver hooks needed (driver.py): the evaluator is cached per (dir, sharing) at fitness.py:415 and driver caches one per worker; the ramp must rebuild it and re-score the pop at each threshold. Modest change. Compare head-to-head vs (a) direct baseline 5.14e-06 and (b) the manual unfold warm-chain 4.19e-06 from yaa — does a graduated ramp beat a single hard unfold transition?\n\nWants the circulation-aware unfold from homemaker-py-8iv once available.","notes":"A/B DONE — NEGATIVE (2026-07-17). harbor-house 3M (500k/grain x3 + 1.5M polish, workers 4, ~22h): Schedule B = 1.26e-08 / 23 fails (canonical byte-for-byte). Loses decisively to both targets: direct baseline 5.14e-06/15 and warm-chain 4.19e-06/15 (~400x worse, +8 fails). Graduated ramp FALSIFIED: each grain step spikes fails (phase-end 19-\u003e21-\u003e27, final unfold 27-\u003e36); per-phase budget re-polishes partially-materialised states that the next step materialises further, so coarse-grain gains don't carry forward; polish started from a deeper hole (36) than the warm chain's single clean transition and only reached 23. The sharing-phase topology skeleton (yaa) is best cashed in ONCE at full grain, not annealed. Machinery retained (search_annealed, --anneal-grain, unfold above=, seed_pop, max_share override) — correct/tested/honest — but §15 single-transition finish stays the default. DESIGN §16 updated. Closing.","status":"closed","priority":3,"issue_type":"feature","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-07-15T06:48:11Z","created_by":"Bruno Postle","updated_at":"2026-07-17T15:33:01Z","started_at":"2026-07-16T06:47:57Z","closed_at":"2026-07-17T15:33:01Z","close_reason":"Closed","dependencies":[{"issue_id":"homemaker-py-kpu","depends_on_id":"homemaker-py-8iv","type":"blocks","created_at":"2026-07-15T07:48:30Z","created_by":"Bruno Postle","metadata":"{}"},{"issue_id":"homemaker-py-kpu","depends_on_id":"homemaker-py-yaa","type":"blocks","created_at":"2026-07-15T07:48:28Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":2,"dependent_count":0,"comment_count":0} @@ -81,26 +82,26 @@ {"id":"homemaker-py-erc.6","title":"Experiment: inner-loop slack-expansion objective term","description":"Inner-loop counterpart to plot-fill construction. If Diagnostic B shows the inner loop has room to expand leaves into slack but no objective gradient to do so (the scalar rewards hitting target area but not exceeding it where slack exists), add a term/incentive so the ratio optimiser pushes leaf boundaries out to consume neighbouring slack and satisfy size, rather than parking at target.\n\nCONDITIONAL on Diagnostic B: build this only if B localizes the gap to the inner loop (room to expand, no gradient); if B shows construction targets too-small dims, prefer the plot-fill construction sibling. Must preserve the §5.4 inner-loop cliff / §4.9 lexicographic protection — the term sits where it cannot displace the fail-count ordering. A/B vs §12.2 baseline, seeds 0/1/2, 20000 evals, staged, default-OFF. Record DESIGN.md §13.6.","notes":"DEPRIORITISED by Diagnostic B (§13.2). B shows the inner loop CANNOT repair undersize: the slack is depth-driven maldistribution baked into the frozen topology, and the equal-offset ratio DOF cannot shrink a 14x leaf to feed a starved one without trading into shape fails (0.5^n cliff). Wrong DOF and wrong direction — the blocker is slicing POSITION, not a missing expansion reward. Fix belongs upstream in construction/topology (erc.4 re-scoped, erc.3). Keep as a low-priority follow-up only if a depth-balanced construction still leaves a residual size gradient the inner loop could pick up.","status":"closed","priority":4,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-06-22T23:16:24Z","created_by":"Bruno Postle","updated_at":"2026-06-28T13:22:22Z","closed_at":"2026-06-28T13:22:22Z","close_reason":"wont-fix (DESIGN §13.7): Diag B (§13.2) showed the inner loop cannot repair undersize (wrong DOF — slicing position, frozen-topology ratios). Superseded by depth-balanced construction (erc.4). Condition unmet.","dependencies":[{"issue_id":"homemaker-py-erc.6","depends_on_id":"homemaker-py-erc","type":"parent-child","created_at":"2026-06-23T00:16:23Z","created_by":"Bruno Postle","metadata":"{}"},{"issue_id":"homemaker-py-erc.6","depends_on_id":"homemaker-py-erc.2","type":"blocks","created_at":"2026-06-23T00:16:47Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-erc.5","title":"Experiment: compactness-aware cuts (minimize leaf perimeter/area)","description":"Attacks the #1 factor, crinkliness (346) — a per-leaf perimeter/area property DISTINCT from proportion (aspect ratio). Proportion-aware seeding (leu.2) sizes splits but does not bias toward balanced, square-ish subdivision. Add a KD-tree-style 'keep both children compact' cut rule (prefer the cut orientation/position that minimises summed child perimeter/area) in construction.\n\nCONDITIONAL on Diagnostic A: if A shows per-leaf shape-fail is FLAT across densities (floor intrinsic to slicing density), better cuts at the same leaf count will not pay → this should be closed wont-fix in favour of leaf-sharing. Only build if A shows shape-fail RISES with density. A/B vs §12.2 baseline, seeds 0/1/2, 20000 evals, staged, default-OFF. Record DESIGN.md §13.5.","notes":"DEPRIORITISED by erc.1 verdict (§13.1): per-leaf shape-fail flat vs slicing density and cuts already squarest (_size_divisions_from_targets picks squarest rotation) yet still ~1.8 fails/leaf =\u003e little compactness headroom at fixed leaf count. Floor is intrinsic to leaf COUNT, not cut quality. Revisit only if leaf-sharing (erc.3) underdelivers.","status":"closed","priority":4,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-06-22T23:16:21Z","created_by":"Bruno Postle","updated_at":"2026-06-28T13:22:17Z","closed_at":"2026-06-28T13:22:17Z","close_reason":"wont-fix (DESIGN §13.7): Diag A (§13.1) showed the floor is intrinsic to leaf COUNT not cut quality; revisit condition was 'only if leaf-sharing underdelivers' but leaf-sharing OVER-delivered (−32…−39%, §13.3). Condition unmet.","dependencies":[{"issue_id":"homemaker-py-erc.5","depends_on_id":"homemaker-py-erc","type":"parent-child","created_at":"2026-06-23T00:16:21Z","created_by":"Bruno Postle","metadata":"{}"},{"issue_id":"homemaker-py-erc.5","depends_on_id":"homemaker-py-erc.1","type":"blocks","created_at":"2026-06-23T00:16:43Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-2g5","title":"Rebuild occlusion/daylight/sun subsystem in Python (post-Phase-5, after optimisation fully native)","description":"DESIGN.md §6 port scope — a whole subsystem, not a term. quality_daylight (Leaf.pm:281-296) needs Urb::Misc::Sun + Urb::Field::Occlusion (+CIESky); quality_uncrinkliness also takes the occlusion object. Indoor spaces return 1 for daylight; cost is outdoor spaces + crinkliness. Port Sun_horizontal (262980-minute normalisation) and the occlusion wall set from Dom-\u003eWalls.","acceptance_criteria":"Daylight and crinkliness factors match Perl (float tolerance) across the corpus, including multi-storey cases","notes":"Re-scoped 2026-06-12: occlusion disabled in the Urb oracle instead of ported (see homemaker-py-gp2). Native fitness ships with simple crinkliness (illumination factor = 1, in homemaker-py-gnw). This issue is now the eventual Python occlusion rebuild, only after optimisation works entirely in Python. Restores outdoor-daylight and shaded-wall selection pressure.\nReframed 2026-06-17: orthogonal to epic homemaker-py-c4c. This is fitness FIDELITY (restoring daylight + shaded-wall selection pressure to match Perl), not search CAPABILITY — it changes what 'good' means, not the search's ability to find good. It will NOT improve final designs in the sense currently sought. Stays P4, deferred until the topology-search-quality epic lands and optimisation is fully native.","status":"open","priority":4,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-06-11T23:38:25Z","created_by":"Bruno Postle","updated_at":"2026-06-17T19:14:48Z","dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"memory","key":"adjacency-in-binary-slicing-tree-is-structural-not","value":"Adjacency in binary slicing tree is structural, not geometric: the inner-loop NM cannot fix topological adjacency failures. Two paths exist: (1) tree-sibling adjacency — a node is adjacent to its sibling in the tree; (2) cross-zone geometric adjacency — leaves from different subtrees that happen to share a boundary. Staircase/adjacency fails require a topology mutation that changes which nodes are siblings or which zones touch. This was proved empirically on programme-house: staircase fail from rot=0 layout could not be fixed by NM but was fixed by level_retype creating a two-C topology (2026-06-14/15)."} -{"_type":"memory","key":"deceptive-valleys-in-topology-search-when-every-single","value":"Deceptive valleys in topology search: when every single-step mutation from a target state passes through a high-fail intermediary (e.g. level_fix displaces a room into 5+ new fails), a compound operator that atomically applies two coordinated changes can escape. Design compound operators to land on the low-fail state directly, bypassing the deceptive gradient. Programme-house example: level_compound_fix atomically moves the level-constrained room AND re-inserts the displaced room adjacent to C in one step (operators.py, 2026-06-14)."} -{"_type":"memory","key":"warm-x0-initialization-bug-pattern-when-a-topology","value":"warm_x0 initialization bug pattern: when a topology operator explicitly sets division ratios on a newly-created node (e.g. compound_fix sets node.division=[0.25,0.25] for t3), parent.ratios has no entry for that node (it was a leaf). warm_x0 defaults it to 0.5, corrupting the inner loop's starting point and making the operator invisible to lex comparison. Fix: only propagate child ratios for nodes where the parent node was NOT already divided; stale hidden nodes revealed by structural mutations (swap flipping b.below) must NOT contribute their pre-writeback values. See driver.py lines 259-267 (fixed 2026-06-14)."} -{"_type":"memory","key":"strategy-decision-2026-06-12-bruno-occlusion-daylight","value":"Strategy decision 2026-06-12 (Bruno): occlusion/daylight is ORTHOGONAL to building a scalable optimiser. Disable it in Urb (env flag, homemaker-py-gp2) rather than port it; native fitness uses simple crinkliness (illumination factor = 1); rebuild occlusion in Python only after optimisation is fully native (homemaker-py-2g5, now P4). Consequence: all scores change when the flag flips — re-baseline corpus/.score, DESIGN \\$4.5 gains, gate bars at one clean boundary AFTER homemaker-py-1p0 closes; Phase-2 urb-evolve benchmark must run with the same flag."} -{"_type":"memory","key":"user-preference-bruno-this-is-a-fedora-system","value":"User preference (Bruno): this is a Fedora system — NEVER install Python packages via pip without asking first; always ask whether to install the rpm via dnf (e.g. python3-cma) before considering pip. Applies to any dependency additions."} -{"_type":"memory","key":"experiment-harness-gotcha-the-leaf-sharing-relaxed-objective","value":"Experiment harness gotcha: the leaf-sharing RELAXED objective (§13.3) is injected ONLY by monkeypatching fitness.load_config in the parent process (run_staged_search.py / probe scripts). This is parent-process-only and does NOT propagate into ProcessPoolExecutor workers (n_workers\u003e1), which re-import fitness fresh and score under the STRICT on-disk patterns.config -\u003e r.n_fails MISMATCH (worker strict vs parent relaxed re-score). ALL §13.x floor runs were therefore SERIAL. Any future PARALLEL leaf-sharing experiment will silently mis-score until leaf_sharing lives on disk/CLI (tracked: homemaker-py-x3b). The parallel driver itself is correct; both paths score via load_config(programme_dir)."} -{"_type":"memory","key":"island-model-psk-14-is-a-null-priming","value":"Island model (psk, §14) is a NULL: priming a population from N converged independent elites + crossover-heavy migration does not beat best-of-N at equal total budget (maple island 124 vs control 116). The child_probe instrument shows WHY: area-matched crossover across independently-converged elites almost never synthesizes (1-3 of ~64 children beat the better parent, max drop 2-5) because the slicing encoding is non-canonical (9gp), so splices are disruptive not combinatorial. Search-machinery null #3 after graded-objective and niching/restarts; residual stays geometry/shape-bound."} -{"_type":"memory","key":"run-to-run-reproducibility-in-homemaker-layout-serial","value":"Run-to-run reproducibility in homemaker-layout: serial search (workers=1) is byte-for-byte deterministic; parallel (workers\u003e1) is now deterministic too AFTER fixing driver._run_batch to admit futures in submission order (was as_completed/completion order, bug xcy). Reproducibility holds only for a FIXED worker count — serial vs parallel differ because children-per-iteration is 1 vs n_workers (different batch granularity), which is expected, not a bug. The constructive seeder was NEVER nondeterministic: _assign_adjacency_aware has unique idx tiebreaks; comparing topologies with Python builtin hash() of the signature STRING is invalid (PYTHONHASHSEED salts str hashing per process) — use a stable hash (sha1) or genome.signature equality."} -{"_type":"memory","key":"homemaker-py-3l6-fix-leaf-sharing-evolve-runs","value":"homemaker-py-3l6 fix: leaf-sharing evolve runs now auto-finish before write via driver.polish_finish — unfold_shared_leaves() then a warm-started leaf_sharing=False polish search (--polish-budget, default budget//2). Makes the written .dom honest under canonical homemaker-fitness (internal==canonical when leaf_sharing off). Interrupt path forces polish_budget=0 (unfold+rescore only). This is yaa's unfold-then-polish, made automatic; Schedule B annealing is still kpu."} -{"_type":"memory","key":"homemaker-py-pythonpath-set-pythonpath-home-bruno-src","value":"homemaker-layout PYTHONPATH: package installed as 'homemaker-layout' via pip install -e . so 'import homemaker_layout' works from anywhere without PYTHONPATH. For running tests use 'python -m pytest' from project root /home/bruno/src/homemaker-layout (pyproject.toml adds src/ automatically). Never try pip show homemaker — that's the old homemaker-addon conflict."} -{"_type":"memory","key":"never-use-corpus-filenames-candidate-001-dom-candidate","value":"Never use corpus filenames (candidate-001.dom, candidate-002.dom, generated.dom, init.dom, etc.) as --output targets when running experiments. These are test fixtures. Always write experimental outputs to scratch/ or a timestamped path. Lesson from 2026-06-14: warm-start runs overwrote candidate-001/002.dom and broke graph tests."} -{"_type":"memory","key":"urb-fitness-bug-found-fixed-2026-06-12","value":"Urb fitness bug found+fixed 2026-06-12 (patch in /home/bruno/src/urb, uncommitted): ProgrammeDriven.pm ratio_o/ratio_type grepped case-insensitively over the ratios hash and took the FIRST key — nondeterministic (x4.5 score swings) for designs with mixed-case type classes (both 'c' circulation and 'C' covered). Fixed to SUM the class (matches Is_Circulation//Is_Outside semantics); 35/35 corpus scores unchanged. CRITICAL for homemaker-py-3y7/gnw: the native port must implement class-SUM ratios. Building.pm has the same unpatched pattern (site-driven path, not used by our oracle). Also: the memetic search reward-hacked this bug before the fix — search results predating it are noise artifacts."} {"_type":"memory","key":"9o5-multi-use-leaves-is-path-a-superposition","value":"9o5 multi-use leaves is path (a) — superposition as SEARCH RELAXATION that COLLAPSES to specific usage at the end, NOT path (b) loose-fit/no-collapse. Bruno's intent: codes with SIMILAR leaf requirements form an interchangeable equivalence class; during evolution the solver doesn't commit which leaf serves which specific usage (smoother landscape, no fighting over exact leaf usage); at the end the layout is CONDENSED to specific usages by brute-forcing the in-class assignment (3 interchangeable usages over 3 leaves = 3! = 6 combinations to check, pick best). 'Derive automatically' compatibility = requirement-similarity grouping. This reverses the issue's stated 'path b preferred' note."} {"_type":"memory","key":"cli-tool-style-prefer-python-m-homemaker-module","value":"CLI tool style: prefer python -m homemaker.module --parameters pattern, installable via pip install -e . with pyproject.toml entry_points. Not standalone bin/ scripts."} {"_type":"memory","key":"collapse-global-94g-and-any-label-usage-optimisation","value":"collapse_global (94g) and any label/usage optimisation CANNOT fix geometry-intrinsic fails. The harbor-house 15-fail best layout contains long-thin cells that are useless whatever room usage is assigned — their width/proportion/crinkliness fails are shape-bound, not label slack. Two consequences: (1) do not over-claim collapse gains — only ~2-3 of that layout's fails are reclaimable relabel slack, the rest are geometry- or building-level bound; (2) the threshold objective must not be tuned to 'pass' a degenerate cell via a permissive room type — a metric-pass on a physically useless space is gaming, not a fix. Real remedies for these are geometry/topology search (cell shape) and circulation placement, filed separately, not the collapse."} -{"_type":"memory","key":"programme-house-optimisation-result-2026-06-14-15","value":"Programme-house optimisation result (2026-06-14/15): best achievable is 1 fail (l1 wrong level, score ~0.005). 0 fails is geometrically impossible: l1 (min 27m²) must occupy ll (~23m²) at level 0, which eliminates the t3-adj-C provider; dividing ll into lll(l1)+llr(C) gives llr proportion ~6:1 (fails). Python memetic optimizer achieves 1 fail in 50k evals vs Perl optimiser's 2-3 fails. Winning topology: TWO C nodes at level 0 — ll(C) for t3-adj-C via geometric contact, rl(C) for staircase via tree-sibling adjacency to rrr(O). Best .dom: scratch/from-warmstart-fixed.dom and scratch/from-compound3-fixed.dom."} -{"_type":"memory","key":"proportion-aware-constructive-seeding-leu-2-12-2","value":"Proportion-aware constructive seeding (leu.2/§12.2): sizing seed cuts from target AREAS only regresses (thin slivers wreck aspect); you must ALSO pick each cut's rotation for child squareness. It is a convergence ACCELERATOR via a deeper local optimum around the constructed topology: wins where that topology is roughly right and budget is scarce (harbor -13%, maple -10% at 20k evals) but DELAYS small programmes where the seed must be restructured by undivide (programme-house regresses at fixed budget, yet reaches the floor given budget - speed, not asymptote). Default-on. Also: n_storeys must honour storey_minimum, not just level: keys (programme-house storey_minimum:2, all rooms level:0 - was seeded 1 storey short; cq1)."} +{"_type":"memory","key":"experiment-harness-gotcha-the-leaf-sharing-relaxed-objective","value":"Experiment harness gotcha: the leaf-sharing RELAXED objective (§13.3) is injected ONLY by monkeypatching fitness.load_config in the parent process (run_staged_search.py / probe scripts). This is parent-process-only and does NOT propagate into ProcessPoolExecutor workers (n_workers\u003e1), which re-import fitness fresh and score under the STRICT on-disk patterns.config -\u003e r.n_fails MISMATCH (worker strict vs parent relaxed re-score). ALL §13.x floor runs were therefore SERIAL. Any future PARALLEL leaf-sharing experiment will silently mis-score until leaf_sharing lives on disk/CLI (tracked: homemaker-py-x3b). The parallel driver itself is correct; both paths score via load_config(programme_dir)."} {"_type":"memory","key":"experiment-seeding-pitfall-run-search-scaled-py-s","value":"Experiment seeding pitfall: run_search_scaled.py's default PH_SEED (c964…dom) is a FINISHED programme-house design — passing it warm-starts and floors at ~3 fails, NOT a blank-slate topology search. For blank-slate runs comparable to §11.5/§11.6 baselines, seed from examples/programme-house/init.dom (a bare undivided plot; driver bootstrap auto-triggers only on bare plots). Bit the 6zy sweep — first pass used c964 and falsely showed 3-fail floor across the whole grid."} -{"_type":"memory","key":"ld2-13-6-interior-o-seed-diagnostic-all","value":"ld2/§13.6 interior-O seed diagnostic: ALL crinkliness fails in the constructed bal+share seed are UNDER-exposed (crink\u003c0.62, landlocked rooms with no facade + no uncovered-O neighbour) — zero over-exposed sliver fails. So the erc crinkliness residual is genuine under-daylighting, validating the interior light-well premise. Default outside_divisor=6 was too sparse (null: harbor 147-\u003e142, crinkliness even rose). odiv=3 is the seed-optimal joint setting: harbor seed fails 147-\u003e129 (-18), maple 219-\u003e206 (-14), landlocked fails drop, at cost of more leaves (harbor +4, maple +8). Because it ADDS leaves it carries the §13.4 wash-out risk; A/B to convergence pending."} -{"_type":"memory","key":"unfold-strategy-for-shared-leaves-homemaker-py-8iv","value":"Unfold strategy for shared leaves (homemaker-py-8iv, resolved 2026-07-16): use the BALANCED GRID (operators._grow_balanced/_size_subtree_equal), NOT circulation-aware slicing. Slicing a shared leaf perpendicular to its access edge so every child touches the corridor was implemented + A/B-tested and LOST decisively (150k-eval warm-start polish from evolved-3M: slice 41 fails/3.5e-14 vs grid 25 fails/2.4e-09, grid ahead at every milestone). Reason: k rooms all touching one wall are intrinsically thin slices; that geometric debt (proportion/long/width) is unfixable without topology change, while the grid's squarer children let local search re-route access cheaply via level_retype/place_missing/level_fix. Lesson: at the sharing-\u003eno-sharing transition, prioritise squarer children and leave access to local search; do not reintroduce slicing in Schedule B (kpu)."} -{"_type":"memory","key":"urb-oracle-nondeterminism-urb-fitness-pl-output-varies","value":"Urb oracle nondeterminism: urb-fitness.pl output varies run-to-run from Perl hash-order randomisation — .fails line ORDER shuffles (compare sorted, use oracle.Score.fail_lines) and the score float can flip by ~1 ULP (compare with math.isclose rel_tol=1e-12, never ==). Not a batching artifact; affects single runs too. Matters for the Phase 3 native-fitness parity gate (homemaker-py-uxz)."} +{"_type":"memory","key":"island-model-psk-14-is-a-null-priming","value":"Island model (psk, §14) is a NULL: priming a population from N converged independent elites + crossover-heavy migration does not beat best-of-N at equal total budget (maple island 124 vs control 116). The child_probe instrument shows WHY: area-matched crossover across independently-converged elites almost never synthesizes (1-3 of ~64 children beat the better parent, max drop 2-5) because the slicing encoding is non-canonical (9gp), so splices are disruptive not combinatorial. Search-machinery null #3 after graded-objective and niching/restarts; residual stays geometry/shape-bound."} +{"_type":"memory","key":"warm-x0-initialization-bug-pattern-when-a-topology","value":"warm_x0 initialization bug pattern: when a topology operator explicitly sets division ratios on a newly-created node (e.g. compound_fix sets node.division=[0.25,0.25] for t3), parent.ratios has no entry for that node (it was a leaf). warm_x0 defaults it to 0.5, corrupting the inner loop's starting point and making the operator invisible to lex comparison. Fix: only propagate child ratios for nodes where the parent node was NOT already divided; stale hidden nodes revealed by structural mutations (swap flipping b.below) must NOT contribute their pre-writeback values. See driver.py lines 259-267 (fixed 2026-06-14)."} +{"_type":"memory","key":"never-use-corpus-filenames-candidate-001-dom-candidate","value":"Never use corpus filenames (candidate-001.dom, candidate-002.dom, generated.dom, init.dom, etc.) as --output targets when running experiments. These are test fixtures. Always write experimental outputs to scratch/ or a timestamped path. Lesson from 2026-06-14: warm-start runs overwrote candidate-001/002.dom and broke graph tests."} +{"_type":"memory","key":"adjacency-in-binary-slicing-tree-is-structural-not","value":"Adjacency in binary slicing tree is structural, not geometric: the inner-loop NM cannot fix topological adjacency failures. Two paths exist: (1) tree-sibling adjacency — a node is adjacent to its sibling in the tree; (2) cross-zone geometric adjacency — leaves from different subtrees that happen to share a boundary. Staircase/adjacency fails require a topology mutation that changes which nodes are siblings or which zones touch. This was proved empirically on programme-house: staircase fail from rot=0 layout could not be fixed by NM but was fixed by level_retype creating a two-C topology (2026-06-14/15)."} {"_type":"memory","key":"correction-to-urb-fitness-bug-memory-bruno-2026","value":"CORRECTION to urb-fitness-bug memory (Bruno, 2026-06-12): 'C' is NOT a 'covered' type — Is_Covered is a geometric predicate (indoor space above). Urb's generic types are canonically UPPERCASE: C=circulation, O=outside, S=sahn (get_space_types qw/C O S/; corpus is 100% uppercase, never 'c'/'o' leaves). The mixed-case designs that fired the latent ratio_type first-match bug were created by homemaker's own operator type pool emitting lowercase 'c'/'o' — fixed: driver/operators now emit uppercase generics only, and class checks use t[0].lower() in 'cos'. The Urb class-sum patch stays as defensive hardening (zero impact on canonical designs). Native port (3y7/gnw): treat type classes case-insensitively, generics canonically uppercase."} +{"_type":"memory","key":"programme-house-optimisation-result-2026-06-14-15","value":"Programme-house optimisation result (2026-06-14/15): best achievable is 1 fail (l1 wrong level, score ~0.005). 0 fails is geometrically impossible: l1 (min 27m²) must occupy ll (~23m²) at level 0, which eliminates the t3-adj-C provider; dividing ll into lll(l1)+llr(C) gives llr proportion ~6:1 (fails). Python memetic optimizer achieves 1 fail in 50k evals vs Perl optimiser's 2-3 fails. Winning topology: TWO C nodes at level 0 — ll(C) for t3-adj-C via geometric contact, rl(C) for staircase via tree-sibling adjacency to rrr(O). Best .dom: scratch/from-warmstart-fixed.dom and scratch/from-compound3-fixed.dom."} +{"_type":"memory","key":"unfold-strategy-for-shared-leaves-homemaker-py-8iv","value":"Unfold strategy for shared leaves (homemaker-py-8iv, resolved 2026-07-16): use the BALANCED GRID (operators._grow_balanced/_size_subtree_equal), NOT circulation-aware slicing. Slicing a shared leaf perpendicular to its access edge so every child touches the corridor was implemented + A/B-tested and LOST decisively (150k-eval warm-start polish from evolved-3M: slice 41 fails/3.5e-14 vs grid 25 fails/2.4e-09, grid ahead at every milestone). Reason: k rooms all touching one wall are intrinsically thin slices; that geometric debt (proportion/long/width) is unfixable without topology change, while the grid's squarer children let local search re-route access cheaply via level_retype/place_missing/level_fix. Lesson: at the sharing-\u003eno-sharing transition, prioritise squarer children and leave access to local search; do not reintroduce slicing in Schedule B (kpu)."} +{"_type":"memory","key":"urb-fitness-bug-found-fixed-2026-06-12","value":"Urb fitness bug found+fixed 2026-06-12 (patch in /home/bruno/src/urb, uncommitted): ProgrammeDriven.pm ratio_o/ratio_type grepped case-insensitively over the ratios hash and took the FIRST key — nondeterministic (x4.5 score swings) for designs with mixed-case type classes (both 'c' circulation and 'C' covered). Fixed to SUM the class (matches Is_Circulation//Is_Outside semantics); 35/35 corpus scores unchanged. CRITICAL for homemaker-py-3y7/gnw: the native port must implement class-SUM ratios. Building.pm has the same unpatched pattern (site-driven path, not used by our oracle). Also: the memetic search reward-hacked this bug before the fix — search results predating it are noise artifacts."} +{"_type":"memory","key":"deceptive-valleys-in-topology-search-when-every-single","value":"Deceptive valleys in topology search: when every single-step mutation from a target state passes through a high-fail intermediary (e.g. level_fix displaces a room into 5+ new fails), a compound operator that atomically applies two coordinated changes can escape. Design compound operators to land on the low-fail state directly, bypassing the deceptive gradient. Programme-house example: level_compound_fix atomically moves the level-constrained room AND re-inserts the displaced room adjacent to C in one step (operators.py, 2026-06-14)."} +{"_type":"memory","key":"homemaker-py-pythonpath-set-pythonpath-home-bruno-src","value":"homemaker-layout PYTHONPATH: package installed as 'homemaker-layout' via pip install -e . so 'import homemaker_layout' works from anywhere without PYTHONPATH. For running tests use 'python -m pytest' from project root /home/bruno/src/homemaker-layout (pyproject.toml adds src/ automatically). Never try pip show homemaker — that's the old homemaker-addon conflict."} +{"_type":"memory","key":"proportion-aware-constructive-seeding-leu-2-12-2","value":"Proportion-aware constructive seeding (leu.2/§12.2): sizing seed cuts from target AREAS only regresses (thin slivers wreck aspect); you must ALSO pick each cut's rotation for child squareness. It is a convergence ACCELERATOR via a deeper local optimum around the constructed topology: wins where that topology is roughly right and budget is scarce (harbor -13%, maple -10% at 20k evals) but DELAYS small programmes where the seed must be restructured by undivide (programme-house regresses at fixed budget, yet reaches the floor given budget - speed, not asymptote). Default-on. Also: n_storeys must honour storey_minimum, not just level: keys (programme-house storey_minimum:2, all rooms level:0 - was seeded 1 storey short; cq1)."} +{"_type":"memory","key":"urb-oracle-nondeterminism-urb-fitness-pl-output-varies","value":"Urb oracle nondeterminism: urb-fitness.pl output varies run-to-run from Perl hash-order randomisation — .fails line ORDER shuffles (compare sorted, use oracle.Score.fail_lines) and the score float can flip by ~1 ULP (compare with math.isclose rel_tol=1e-12, never ==). Not a batching artifact; affects single runs too. Matters for the Phase 3 native-fitness parity gate (homemaker-py-uxz)."} +{"_type":"memory","key":"homemaker-py-3l6-fix-leaf-sharing-evolve-runs","value":"homemaker-py-3l6 fix: leaf-sharing evolve runs now auto-finish before write via driver.polish_finish — unfold_shared_leaves() then a warm-started leaf_sharing=False polish search (--polish-budget, default budget//2). Makes the written .dom honest under canonical homemaker-fitness (internal==canonical when leaf_sharing off). Interrupt path forces polish_budget=0 (unfold+rescore only). This is yaa's unfold-then-polish, made automatic; Schedule B annealing is still kpu."} +{"_type":"memory","key":"ld2-13-6-interior-o-seed-diagnostic-all","value":"ld2/§13.6 interior-O seed diagnostic: ALL crinkliness fails in the constructed bal+share seed are UNDER-exposed (crink\u003c0.62, landlocked rooms with no facade + no uncovered-O neighbour) — zero over-exposed sliver fails. So the erc crinkliness residual is genuine under-daylighting, validating the interior light-well premise. Default outside_divisor=6 was too sparse (null: harbor 147-\u003e142, crinkliness even rose). odiv=3 is the seed-optimal joint setting: harbor seed fails 147-\u003e129 (-18), maple 219-\u003e206 (-14), landlocked fails drop, at cost of more leaves (harbor +4, maple +8). Because it ADDS leaves it carries the §13.4 wash-out risk; A/B to convergence pending."} {"_type":"memory","key":"multi-storey-staircase-consistency-when-dividing-or-retyping","value":"Multi-storey staircase consistency: when dividing or retyping a circulation (C) leaf at one level, the same structural change should be propagated to the matching leaf on ALL other storeys so the stair core path is maintained. The optimizer cannot fix staircase disruptions through trial-and-error geometry alone — it requires a synchronized multi-level operator that applies the same topology change to every storey simultaneously."} +{"_type":"memory","key":"strategy-decision-2026-06-12-bruno-occlusion-daylight","value":"Strategy decision 2026-06-12 (Bruno): occlusion/daylight is ORTHOGONAL to building a scalable optimiser. Disable it in Urb (env flag, homemaker-py-gp2) rather than port it; native fitness uses simple crinkliness (illumination factor = 1); rebuild occlusion in Python only after optimisation is fully native (homemaker-py-2g5, now P4). Consequence: all scores change when the flag flips — re-baseline corpus/.score, DESIGN \\$4.5 gains, gate bars at one clean boundary AFTER homemaker-py-1p0 closes; Phase-2 urb-evolve benchmark must run with the same flag."} +{"_type":"memory","key":"user-preference-bruno-this-is-a-fedora-system","value":"User preference (Bruno): this is a Fedora system — NEVER install Python packages via pip without asking first; always ask whether to install the rpm via dnf (e.g. python3-cma) before considering pip. Applies to any dependency additions."} +{"_type":"memory","key":"run-to-run-reproducibility-in-homemaker-layout-serial","value":"Run-to-run reproducibility in homemaker-layout: serial search (workers=1) is byte-for-byte deterministic; parallel (workers\u003e1) is now deterministic too AFTER fixing driver._run_batch to admit futures in submission order (was as_completed/completion order, bug xcy). Reproducibility holds only for a FIXED worker count — serial vs parallel differ because children-per-iteration is 1 vs n_workers (different batch granularity), which is expected, not a bug. The constructive seeder was NEVER nondeterministic: _assign_adjacency_aware has unique idx tiebreaks; comparing topologies with Python builtin hash() of the signature STRING is invalid (PYTHONHASHSEED salts str hashing per process) — use a stable hash (sha1) or genome.signature equality."} diff --git a/DESIGN.md b/DESIGN.md index 895d162..5273bf3 100644 --- a/DESIGN.md +++ b/DESIGN.md @@ -2415,7 +2415,7 @@ gated on the 9o5 landscape-flattening risk (§13 / `homemaker-py-xi7`) and its o Tests: `tests/test_collapse_global.py` ×6 (demand-set relabel, level hard constraint, c/o/s exclusion, no-op safety, keep-better/unmerged); 267 pass. -## 18. Graded circulation-connectivity signal (`homemaker-py-qi6`) — in progress +## 18. Graded circulation-connectivity signal (`homemaker-py-qi6`) — DONE (negative) **Motivation — the binary fail is flat.** After the §17 collapse, the residual fails on the harbor-house set are dominated by `level N not connected` (2 of the best layout's 12; also on @@ -2458,13 +2458,39 @@ the gradient toward connected topologies. (env `HOMEMAKER_CONN_GRADE`, default OFF); the grade is read off the optimised tree, one extra native eval per child. -**Status / next.** Signal, fitness wiring, CLI, and 9 tests landed (`tests/test_conn_grade.py`: -pure-graph fraction contract, non-circ cells ignored, monotone under (dis)connection, and the -score/fail-count-invariance of the flag). The A/B — does the gradient actually pull evolve runs -toward connected circulation and clear `not connected` fails — needs full-budget runs and is -pending (short 60-eval smoke run confirms the plumbing only). If the graded key alone is -insufficient, the follow-on is an insert/relocate-circulation mutation operator (mechanism (a), -still `homemaker-py-qi6`) that now has a gradient to climb. 276 tests pass. +**Build.** Signal, fitness wiring, CLI, and 9 tests landed (`tests/test_conn_grade.py`: pure-graph +fraction contract, non-circ cells ignored, monotone under (dis)connection, and the score/fail- +count-invariance of the flag). 276 tests pass. + +**A/B verdict (measured, 2026-07-22, qpk protocol, `experiments/run_qi6_ab.sh`) — NEGATIVE.** +Equal-budget `conn_grade` ON vs OFF, both arms finished with the standard finish-time `--collapse` +(94g), 4 workers, canonical `homemaker-fitness` re-score for the `.fails` breakdown: + +- **harbor-house** (`init.dom`, budget 2500, seeds 1–3): **byte-identical output** in every seed + (dom, fail list, fitness all diff-clean ON vs OFF) — the secondary comparator key never fired, + i.e. the search trajectory never actually hit a tie at fail-count that the grade could break. + This is the programme §18 was motivated on (2 of 15 fails on the best layout are `not + connected`), and the signal moved nothing. +- **programme-house** (`init.dom`, budget 3000, seeds 1–5): 3/5 seeds tie exactly (byte-identical + `.fails`); seeds 1 and 2 diverge to a **different topology** with one fewer total fail (8→7 + each) — but the diff is entirely adjacency/crinkliness/width/access/size fails, not + connectivity. In all 4 seed-arms across both programmes where a `not connected` fail was + actually present (harbor 1&3, programme 3&4), the fail is **unchanged** in both arms — zero + cases of the grade clearing one. +- **Conclusion: the grade does not do what §18 designed it to do.** It occasionally perturbs + tie-breaking among equal-fail-count neighbours (programme-house seeds 1/2), which can + incidentally shift the total fail count, but that perturbation never targets circulation + connectivity specifically — consistent with a comparator key that is either too weak relative + to the primary `(-n_fails, fitness)` keys to steer topology choice, or whose grade values are + rarely distinct enough between the actual neighbours the search compares to break a tie in the + intended direction. + +**Status / next.** Kept default OFF (already was). Mechanism (b) (graded proximity as a tertiary +key) is falsified by this A/B, not just unconfirmed — do not re-attempt without a different +mechanism. The remaining candidate from the original issue is mechanism (a): an explicit +insert/relocate-circulation mutation/repair operator, which does not depend on the search +stumbling onto a fail-count tie to act. Not started; low priority per DISCOVERED-FROM epic +`homemaker-py-94g`'s framing (fitness fidelity, not search capability). ## 19. Geometry/topology repair for shape-intrinsic fails (`homemaker-py-7fm`) — DONE (negative) diff --git a/experiments/run_qi6_ab.sh b/experiments/run_qi6_ab.sh new file mode 100755 index 0000000..a1df4fc --- /dev/null +++ b/experiments/run_qi6_ab.sh @@ -0,0 +1,60 @@ +#!/usr/bin/env bash +# qi6 A/B (DESIGN.md §18): does the graded circulation-connectivity signal +# actually pull full-budget evolve runs toward connected circulation and clear +# 'level N not connected' fails, or is the secondary-comparator gradient too +# weak to matter (mirrors the qpk §20 protocol: equal-budget ON vs OFF, both +# arms finished with the standard finish-time --collapse (94g) so the +# comparison is apples-to-apples on the final collapsed score)? +# +# Authoritative metrics: total fail count and 'not connected' fail count, read +# from the .fails file homemaker-fitness writes (evolve.py itself only prints +# n_fails, not the fail list). Each run appends one TSV row so partial results +# survive an interrupt. +# +# Usage: experiments/run_qi6_ab.sh +set -u +cd "$(dirname "$0")/.." + +WORKERS=4 +OUT=scratch/qi6_ab; mkdir -p "$OUT" +TSV=scratch/qi6_ab_results.tsv +[ -f "$TSV" ] || printf 'programme\tseed\tconn_grade\tbudget\tfails\tnot_connected\tfitness\telapsed_s\n' > "$TSV" + +run() { # programme seed conn_grade(0|1) budget + local prog="$1" seed="$2" cg="$3" budget="$4" + local tag="cg${cg}" + local dom="$OUT/${prog}_${tag}_s${seed}.dom" + local log="$OUT/${prog}_${tag}_s${seed}.log" + local flag="--no-conn-grade"; [ "$cg" = 1 ] && flag="--conn-grade" + echo ">>> $prog seed=$seed conn_grade=$cg budget=$budget" + local t0; t0=$(date +%s) + homemaker-evolve "examples/$prog/init.dom" \ + --budget "$budget" --workers "$WORKERS" --seed "$seed" \ + $flag --output "$dom" > "$log" 2>&1 + local t1; t1=$(date +%s) + local fitness fails notconn + fitness=$(sed -n 's/^best *: \([0-9.e+-]*\) .*/\1/p' "$log") + fails=$(sed -n 's/^best *: [0-9.e+-]* (\([0-9]*\) fails).*/\1/p' "$log") + # re-score with the canonical scorer to get the .fails breakdown (evolve.py + # only prints the count, not the list); dom lives outside examples/$prog so + # pass an absolute path while cd'd there for patterns.config resolution + ( cd "examples/$prog" && homemaker-fitness "$(realpath "../../$dom")" > /dev/null 2>&1 ) + notconn=0 + if [ -f "${dom}.fails" ]; then + notconn=$(grep -c 'not connected' "${dom}.fails") + fi + printf '%s\t%s\t%s\t%s\t%s\t%s\t%s\t%s\n' \ + "$prog" "$seed" "$cg" "$budget" "${fails:-ERR}" "$notconn" "${fitness:-ERR}" "$((t1-t0))" >> "$TSV" + echo " -> ${fails:-ERR} fails (${notconn} not-connected), fitness=${fitness:-ERR}, $((t1-t0))s" +} + +# harbor-house: budget 2500, seeds 1-3 (qpk protocol) +for seed in 1 2 3; do run harbor-house "$seed" 0 2500; done +for seed in 1 2 3; do run harbor-house "$seed" 1 2500; done + +# programme-house: budget 3000, seeds 1-5 (qpk protocol) +for seed in 1 2 3 4 5; do run programme-house "$seed" 0 3000; done +for seed in 1 2 3 4 5; do run programme-house "$seed" 1 3000; done + +echo "=== qi6 conn_grade A/B complete ===" +column -t -s $'\t' "$TSV"