diff --git a/.beads/issues.jsonl b/.beads/issues.jsonl index e191795..4a29031 100644 --- a/.beads/issues.jsonl +++ b/.beads/issues.jsonl @@ -33,7 +33,7 @@ {"id":"homemaker-py-2g7.9","title":"Parallel best-of-N + racing harness (use all cores, kill stragglers early)","description":"§14 measured islands \u003c= best-of-N, and the 3M runs used workers=1-2 on a 4-core box — independent seeds are the proven shape and we are not even using the local machine. Build a harness: launch N independent search_staged seeds across all cores (processes, not threads — mind the cvw id()-keyed cache bug), checkpoint fail-counts periodically, successively halve (hyperband-style: kill runs above median hard-fail count at each rung, reallocate budget to survivors). Fix/respect homemaker-py-b8g (parallel non-determinism) and homemaker-py-cvw first or work around with process isolation. This multiplies whatever eval cost the shape-curve DP issue achieves; on its own it is a free 4x locally and scales to any box. Report best + variance across seeds (the seed-variance in §12-§13 tables is huge — 78 vs 97 same config — so best-of-N is worth several levers combined).","acceptance_criteria":"harness runs N=16 seeds on 4 cores with racing; at equal total native-eval budget beats the single-seed mean on harbor by at least the observed seed spread; deterministic per-seed replay","status":"open","priority":2,"issue_type":"task","owner":"bruno@postle.net","created_at":"2026-08-02T09:15:58Z","created_by":"Bruno Postle","updated_at":"2026-08-02T09:15:58Z","dependencies":[{"issue_id":"homemaker-py-2g7.9","depends_on_id":"homemaker-py-2g7","type":"parent-child","created_at":"2026-08-02T10:15:58Z","created_by":"Bruno Postle","metadata":"{}"},{"issue_id":"homemaker-py-2g7.9","depends_on_id":"homemaker-py-b8g","type":"blocks","created_at":"2026-08-02T10:16:16Z","created_by":"Bruno Postle","metadata":"{}"},{"issue_id":"homemaker-py-2g7.9","depends_on_id":"homemaker-py-cvw","type":"blocks","created_at":"2026-08-02T10:16:15Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":2,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-2g7.7","title":"LLM repair operator at stagnation (dom+fails -\u003e targeted compound edits)","description":"Generalize the §4.10 lesson: deceptive valleys are crossed by COMPOUND edits (move room + re-home displaced room + fix ratios atomically), which we currently hand-code one per valley (mutate_level_compound_fix). Our fail messages are semantically rich and localized ('me1 on wrong level', 'level 1 not connected', '0/rlrlr proportion') and the .dom is readable — ideal LLM input. Loop: on stagnation (no fail-tier improvement for N evals), serialize best individual + .fails + programme summary -\u003e LLM proposes 3-5 multi-step repairs as structured edit scripts (a small DSL over existing operator primitives: swap/divide/retype/rotate with explicit paths — NOT freeform dom text, so proposals are always well-formed) -\u003e apply, inner-loop, lex-accept as usual. Native fitness disposes; a bad proposal costs one child budget. Cost discipline: one LLM call ~ thousands of native evals, so plateau-only, cache by (signature, fails) key. Benchmark: the 3M-run best sat on 'level 0 not connected' + 'me1 on wrong level' for \u003e1M evals — moves a plan-reader fixes in one edit. Use claude via API (see claude-api skill); temperature\u003e0 for diverse proposals. Later extension (separate issue): AlphaEvolve-style operator-code synthesis using our existing A/B harness as the evaluator.","acceptance_criteria":"on the evolved-3M-nols-3 15-fail plateau seed: repair loop reduces hard-fail count where 1M+ blind evals did not, within \u003c=20 LLM calls; edit-DSL rejects malformed proposals; A/B at equal native-eval budget shows strictly better final fails on \u003e=2/3 seeds","status":"open","priority":2,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-08-02T09:15:54Z","created_by":"Bruno Postle","updated_at":"2026-08-02T09:15:54Z","dependencies":[{"issue_id":"homemaker-py-2g7.7","depends_on_id":"homemaker-py-2g7","type":"parent-child","created_at":"2026-08-02T10:15:53Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} {"id":"homemaker-py-2g7.6","title":"Spike: graph-first construction — adjacency-realizing slicing trees / rectangular dualization","description":"Research spike, timeboxed. Literature: rectangular dualization (planar triangulated graph -\u003e rectangular floorplan) and characterizations of slicible adjacency graphs. Our programme already IS an adjacency graph (every room wants c, plus secondary pairs); instead of mutating trees hoping adjacency emerges, construct trees that realize the required adjacency by construction — the direction §11.6/§11.7 crawled toward greedily. Deliverable is a WRITTEN assessment (DESIGN.md section): can harbor's programme graph (16 rooms + spine, 2 storeys with stacking constraint) be dualized into slicing trees, how many, and is enumeration of realizing trees tractable? Prototype only if the answer is clearly yes. Watch for: multi-storey Below-inheritance constrains both floors' trees jointly; circulation spine is a connected dominating set requirement, not a simple adjacency.","acceptance_criteria":"DESIGN.md section with go/no-go verdict, the relevant algorithms named, and complexity estimate for harbor-scale programmes","status":"open","priority":2,"issue_type":"task","owner":"bruno@postle.net","created_at":"2026-08-02T09:15:07Z","created_by":"Bruno Postle","updated_at":"2026-08-02T09:15:07Z","dependencies":[{"issue_id":"homemaker-py-2g7.6","depends_on_id":"homemaker-py-2g7","type":"parent-child","created_at":"2026-08-02T10:15:07Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} -{"id":"homemaker-py-2g7.5","title":"CP-SAT type assignment for a fixed tree (replace swap/retype random walk)","description":"For a FIXED topology, assigning room codes to leaves subject to counts, required levels, adjacency-to-circulation-spine, secondary adjacencies (k1-da1, da1-o...), and share grouping is a small discrete problem (~30-70 leaves, ~16-26 codes) — well within OR-Tools CP-SAT range, solvable optimally in milliseconds. Today swap/retype/level_retype random-walk this space; §11.6/§11.7's greedy constructive assignment was the single biggest fail-count win of Phase 6, and CP-SAT is its exact big brother. Plan: model leaf-graph adjacency (geometry.leaf_graph) as fixed at seed geometry; objective = weighted satisfied adjacencies + level compliance; use as (a) seeder replacing the greedy _assign_adjacency_aware, (b) periodic 'reassign' operator inside search (the assignment analogue of ruin_recreate §23), (c) post-collapse repair. Note the §11.2 lesson: assignment quality at SEED geometry can shift after the inner loop moves ratios — re-run assignment after geometry settles (alternating minimization).","acceptance_criteria":"A/B vs greedy seeder (harbor+maple, 3 seeds, 20k evals): adjacency+access seed fails strictly lower; end-to-end mean fails no worse; reassign operator fires and is accepted at least once per run","status":"open","priority":2,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-08-02T09:15:06Z","created_by":"Bruno Postle","updated_at":"2026-08-02T09:15:06Z","dependencies":[{"issue_id":"homemaker-py-2g7.5","depends_on_id":"homemaker-py-2g7","type":"parent-child","created_at":"2026-08-02T10:15:05Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} +{"id":"homemaker-py-2g7.5","title":"CP-SAT type assignment for a fixed tree (replace swap/retype random walk)","description":"For a FIXED topology, assigning room codes to leaves subject to counts, required levels, adjacency-to-circulation-spine, secondary adjacencies (k1-da1, da1-o...), and share grouping is a small discrete problem (~30-70 leaves, ~16-26 codes) — well within OR-Tools CP-SAT range, solvable optimally in milliseconds. Today swap/retype/level_retype random-walk this space; §11.6/§11.7's greedy constructive assignment was the single biggest fail-count win of Phase 6, and CP-SAT is its exact big brother. Plan: model leaf-graph adjacency (geometry.leaf_graph) as fixed at seed geometry; objective = weighted satisfied adjacencies + level compliance; use as (a) seeder replacing the greedy _assign_adjacency_aware, (b) periodic 'reassign' operator inside search (the assignment analogue of ruin_recreate §23), (c) post-collapse repair. Note the §11.2 lesson: assignment quality at SEED geometry can shift after the inner loop moves ratios — re-run assignment after geometry settles (alternating minimization).","acceptance_criteria":"A/B vs greedy seeder (harbor+maple, 3 seeds, 20k evals): adjacency+access seed fails strictly lower; end-to-end mean fails no worse; reassign operator fires and is accepted at least once per run","notes":"2026-08-04: Shipped items (a) seeder + (b) reassign operator; item (c)\ndeferred to new child bead homemaker-py-5bv. Full details: DESIGN.md §37.7.\n\nWhat shipped: src/homemaker_layout/cpsat.py (new module) -- exact room-code\nlabelling via OR-Tools CP-SAT (pyproject.toml: ortools\u003e=9.10). Wired into\noperators._assign_adjacency_aware/constructive_topology/lift_base_to_storeys\nas assign_solver=\"greedy\"|\"cpsat\" (EXPERIMENTAL, default \"greedy\" = prior\nbehaviour byte-identical), and driver.search's same-named passthrough. New\noperators.mutate_reassign + MUTATIONS entry, gated by driver.search's new\nenable_reassign=False default (mirrors enable_ruin_recreate's pattern).\n\nTwo real bugs found and fixed during integration (not just the isolated\nsolver): (1) proportion-aware target-size resizing right after assignment\ncan shrink a wall segment below the door-width adjacency threshold,\ninvalidating an edge CP-SAT's tighter solve relied on more than greedy's\nconservative placement -- fixed with operators._cpsat_relabel_settled, a\nsecond exact-solve pass against the now-settled geometry (the bead's own\n\"re-run assignment after geometry settles\" note, applied literally).\n(2) CP-SAT symmetry blowup: several codes sharing an identical unreferenced\nadjacency signature (e.g. 4 same-need bedroom instances) are fully\ninterchangeable, causing multi-second non-deterministic stalls on ~15-slot\nmodels via wasted branch-and-bound proof time -- fixed with an explicit\nsymmetry-breaking grouping constraint (canonical slot-index ordering within\ninterchangeable-code groups). A naive lexicographic-tiebreak-in-objective\nfirst attempt made this WORSE, reverted.\n\nMeasured: seeder-level (constructive_topology alone) A/B on harbor-house is\na solid, low-noise positive -- 10 seeds, cpsat 13 wins/4 ties/3 losses vs\ngreedy on real fitness-scored secondary-adjacency fails, ~13% fewer total\n(tests/test_operators.py::test_assign_cpsat_matches_or_beats_greedy_secondary_adjacency).\n\nNOT done / acceptance criteria NOT fully met (same reason 6xh stayed\nin_progress rather than closing): the bead's own acceptance protocol is\nharbor+maple, 3 seeds, 20k-budget driver.search A/B -- wall-clock budget\nthis session only stretched to a pilot (harbor-house only, budget=3000,\n3 seeds, experiments/ab_cpsat_assign.py). That pilot is INCONCLUSIVE at\nfull-search level: cpsat arm worse on mean hard fails (24.0 vs greedy's\n21.0) but better on soft (36.7 vs 39.0) and ~5.7x mean fitness; reassign\narm never actually fired in any of the 3 seeds (mean_reassign_fired=0.0 --\nbudget=3000 gives the uniform-weight op too few draws, ~15-20 children\ntotal), so its arm's difference from cpsat's is RNG noise, not a measured\nreassign effect. mutate_reassign firing+multiset-preservation IS confirmed\nindependently at the operator level (20/20 direct trials,\ntest_reassign_fires_and_preserves_room_multiset) -- just not yet inside a\nreal full-budget driver.search run.\n\nBoth assign_solver=\"cpsat\" and enable_reassign stay opt-in EXPERIMENTAL,\ndefault off, regardless -- matches every other flag in this codebase\n(c94-style precedent), independent of the pilot's inconclusive full-search\nresult.\n\nRemaining work, tracked here (not a new bead, since it's literally this\nbead's own unmet acceptance criterion): run the real harbor+maple/3-seed/\n20k-budget A/B per the bead's original acceptance criteria and decide\ndefault-on based on that, not the pilot. homemaker-py-5bv (child of\n2g7/2g7.5) tracks item (c), the CP-SAT post-collapse repair inside\nFitness.collapse_global -- deliberately not attempted here since\ncollapse_global runs inside every in-search eval by default\n(collapse_insearch=True) and is far more delicate than the seeder/reassign\nwork.\n\nTests: tests/test_cpsat.py (5, new), tests/test_operators.py (+5),\ntests/test_driver.py (+2). Full suite 409 passed.","status":"in_progress","priority":2,"issue_type":"feature","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-08-02T09:15:06Z","created_by":"Bruno Postle","updated_at":"2026-08-04T08:17:58Z","started_at":"2026-08-03T22:44:01Z","dependencies":[{"issue_id":"homemaker-py-2g7.5","depends_on_id":"homemaker-py-2g7","type":"parent-child","created_at":"2026-08-02T10:15:05Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-cvw","title":"Parallel staged runs: substrate_readiness reads stale id()-keyed geometry cache in the parent process","description":"Found by the homemaker-py-zrx expert review. geometry._cache is keyed by (id(node), idx) and relies on every reader being preceded by clear_cache(). In driver.search_staged stage 1 with n_workers\u003e1 that contract breaks: the PARENT process never runs score_with_fails (children are scored in the pool workers), so its cache is never cleared, yet _rank_fitness -\u003e rank_bonus_fn -\u003e graph.substrate_readiness(ind.root) reads geometry.area()/coordinate() in the parent on every tournament/admit comparison. Evicted individuals are eventually gc'd (Node trees are parent\u003c-\u003echild reference cycles, freed by the cycle collector) while their cache entries linger; freshly unpickled worker results reuse those addresses, and substrate_readiness then serves another (dead) tree's coordinates.\n\nVerified with a probe simulating the parent's allocation pattern (unpickle jittered harbor-house trees, pop-16 eviction churn, periodic gc.collect): 24/300 readiness computations returned a corrupted value, worst absolute error 0.999 on the [0,1] readiness scale (i.e. completely wrong), and the parent cache grew without bound (38k entries after 300 children — it is never cleared for the whole run). Serial staged runs are safe (every in-process score_with_fails clears the cache between children, and live/dead id coexistence prevents collisions).\n\nImpact: silently biases stage-1 substrate selection in every parallel staged run (run_staged_search.py with WORKERS\u003e1 — the default experimental harness), and makes the bias address-dependent, i.e. NON-DETERMINISTIC across byte-identical re-runs. This is a concrete, static-read-visible candidate mechanism for part of homemaker-py-b8g's irreproducibility (it is not BLAS): it perturbs the stage-1 trajectory, not a single fixed-genome score. Reported fitness numbers are unaffected (the bonus only reorders the comparator).\n\nRecommended fix: geometry.clear_cache() at substrate_readiness entry (cheap: the readiness read is a handful of areas on the base level; serial-mode behaviour is unchanged because the cache there is already cold at that point). The durable fix for the whole bug class — also covering the (unobserved but real) gc-timing hazard in collapse_finish's cand deepcopy, probed 0/6 today only because cyclic trees outlive the deepcopy window — is to cache on the Node object itself (as Urb does, per geometry.py's own comment) or key by a per-tree epoch, so a recycled address can never alias. Also add a defensive geometry.clear_cache() at collapse_global entry (one line, zero practical cost: finish-time it is one-shot, in-search the cache was just cleared by _evaluate_full).","status":"closed","priority":2,"issue_type":"bug","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-08-02T08:19:18Z","created_by":"Bruno Postle","updated_at":"2026-08-02T09:52:49Z","started_at":"2026-08-02T09:52:09Z","closed_at":"2026-08-02T09:52:49Z","close_reason":"Fixed: geometry.clear_cache() added at substrate_readiness (graph.py) and collapse_global (fitness.py) entry points; commit 2f26f46. Full test suite (338 tests) passes.","dependency_count":0,"dependent_count":1,"comment_count":0} {"id":"homemaker-py-r5a","title":"Stale leaf-share stamp resurrects when collapse_global commits a leaf back to its stamped code","description":"Found by the homemaker-py-zrx expert review. The homemaker-py-iio fix stops _collapse_value/_usage_quality PROBES from seeing a stale share (share\u003e1, share_type != type), but the COMMIT path still resurrects it: when collapse_global's assignment (or a 2-opt swap in _two_opt_adjacency_polish) relabels a leaf back to its stale share_type, leaf.type == share_type again, graph.leaf_share goes live, and the leaf immediately counts as k rooms with a k*target size centre — a credit the Hungarian matrix valued at 1x (the iio guard cleared the stamp for exactly that probe). The resurrected stamp then SERIALISES (dom._emit's guard passes once type == share_type), so it persists in the output.\n\nConsequences: (1) in-process eval vs dump/reload eval of the SAME tree diverge again — the exact 91f/iio divergence class, reopened through the commit door. Repro (verified today): 12x8 two-leaf tree, left leaf typed b1 carrying stale share=3/share_type=n, programme n(count 3, size 24+-5, w 3+-0.8, p 2+-0.6) + b1(count 1, size 24+-5, w 12+-0.5, p 2+-0.6), leaf_sharing+collapse_insearch on: live eval = 12 fails / score 7.43e-08; dump+reload twin = 19 fails / 7.29e-11 (twin gains '0/l size', 'missing required space n#1' + critical + 3 would-need lines; live instead has 'too many spaces: n (found 4, expected 3)'). (2) The Jacobi valuation (1x, post-iio) and the committed reality (kx) disagree, so assignments are made under one objective and scored under another; the 2-opt reward() sees the kx credit during trial swaps while the Jacobi matrix never did — the two phases of the same optimiser price the same relabel differently. (3) An ordinary retype mutation that happens to restore a leaf's old code resurrects the stamp the same way (no collapse needed), with the same live-vs-reloaded divergence.\n\nRecommended fix: canonicalise stale stamps instead of guarding readers one by one — at _evaluate_full entry (or minimally at collapse_global entry over the supply set), drop share/share_type whenever share_type is set and != type, exactly mirroring dom._emit's serialisation guard, so the in-memory tree can never disagree with its canonical dumped form. Add a dump/reload-agreement regression test in the style of test_collapse_global_dump_reload_agree_with_stale_share but driving the COMMIT (use the repro above: assignment must relabel the stamped leaf back to its stamped code). Note this slightly changes search dynamics (accidental resurrection credit disappears), so re-run a quick harbor-house sanity A/B when landing. Feeds homemaker-py-d86 (historical re-verification should use the post-fix semantics).","status":"closed","priority":2,"issue_type":"bug","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-08-02T08:18:36Z","created_by":"Bruno Postle","updated_at":"2026-08-02T09:44:06Z","started_at":"2026-08-02T09:25:44Z","closed_at":"2026-08-02T09:44:06Z","close_reason":"Fixed: dom.canonicalize_shares() drops share/share_type whenever share_type != type, called at the top of collapse_global and _evaluate_full so a leaf relabelled back to its stale share_type (collapse commit, collapse_superposition, or a retype mutation) can no longer resurrect a multiplicity credit. Added regression test test_collapse_global_commit_does_not_resurrect_stale_share; confirmed via monkeypatch that it fails without the fix. Full suite (338 tests) passes; harbor-house A/B (evolved-3M/-nols/-anneal) shows identical scores pre/post-fix (no stale stamps on those files).","dependency_count":0,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-zrx","title":"Targeted expert review of core numeric/scoring path (fitness/solver/collapse) for silent correctness bugs","description":"Run a deep, expensive-model code review scoped to the core numeric/scoring\nlogic: fitness.py, solver.py, collapse_cmd.py, and the collapse_insearch\npath in innerloop.py/driver.py. Motivated by homemaker-py-iio: a stale\nleaf-share leak in collapse_global's probe valuation silently corrupted\nscores for an unknown period before being caught by manual diagnostic\nreview (see DESIGN.md §35 for the retroactive-impact writeup). That bug\nclass - subtle numeric/state bugs that don't crash, just quietly bias\nscores - is exactly what a careful full-context review with a stronger\nmodel is suited to catching, and exactly what a quick pass would miss.\n\nScope: read fitness.py, solver.py, collapse_cmd.py, and the\ncollapse_insearch code path end-to-end looking for:\n- other stale-state/leak bugs analogous to iio (shared mutable state\n reused across probes/leaves without proper reset)\n- valuation/accounting mismatches between search-time scoring and\n finish-time collapse scoring (the class of bug behind 7ua)\n- non-determinism sources under n_workers\u003e1 (b8g) if visible from a\n static read\n- anything else that would bias .score output without raising an\n exception or failing a test\n\nOut of scope: CLI wrappers, dom.py parsing, genome/operators (topology\nsearch), occlusion/daylight (2g5) - not on the numeric-correctness path.\n\nRelated: d86 (re-verify qpk/1ph historical numbers against the iio fix)\nand 7ua (false MISMATCH bug) are follow-ups from the same root cause\nclass this review is meant to catch earlier next time. This review\nshould probably run before/alongside d86 so any new findings feed into\nthe historical re-verification rather than requiring a second pass.","notes":"REVIEW COMPLETE (2026-08-02). Files read end-to-end: fitness.py, solver.py, collapse_cmd.py, innerloop.py, driver.py (collapse_insearch path), plus geometry.py/graph.py/dom.py support and evolve.py plumbing. Three confirmed bugs and one hygiene task filed:\n\n- homemaker-py-r5a (P2, CONFIRMED by minimal repro): stale leaf-share stamps resurrect when collapse_global's COMMIT relabels a leaf back to its stamped code — the iio fix guarded the probes but not the commit; live vs dump/reload evals of the same tree diverge again (12 vs 19 fails, score 7.4e-08 vs 7.3e-11 in the repro), and the resurrected stamp serialises and persists. Recommend canonicalising stale stamps at _evaluate_full (or collapse_global) entry, mirroring dom._emit.\n- homemaker-py-cvw (P2, CONFIRMED by probe): n_workers\u003e1 search_staged stage 1 — parent process never clears geometry._cache but substrate_readiness reads geometry there every ranking comparison; dead individuals' id()-keyed entries alias freshly unpickled children (24/300 readiness values corrupted, worst error ~1.0). Address-dependent selection bias; candidate mechanism for part of b8g. Serial runs safe.\n- homemaker-py-sd3 (P3, CONFIRMED on 5 evolved files): driver.collapse_best builds its evaluator with collapse_insearch=True baked in (no way to thread the run flag); the 94g keep-better guard is vacuous (base==coll 5/5, e.g. logs 12-\u003e12 where canonical shows 15-\u003e12) and a canonically fail-increasing collapse would be silently applied. Same gap in search_annealed's final rescore.\n- homemaker-py-pek (P3): fitness.py has two process_storey definitions; the first (~line 1146) is dead code silently shadowed by the second (~1452).\n\nReviewed clean (no defect found): solver.py (experiments-only, not on the scoring path; its residuals ignore share/co_type but nothing in search calls it); gaussian/truncated-e and clipped-gaussian ports; _gaussian_product combination; check_space_counts coverage arithmetic and missing-id suppression; collapse_global's pin/slot accounting, forbid handling, Jacobi synchronous update and the xcy submission-order determinism fix; merge_divided (merges only o/s leaves, so no share-stamp loss); NativeEvaluator deepcopy hygiene (per-eval clear_cache at _evaluate_full entry protects the whole in-eval path including collapse_insearch); collapse_finish's cand-deepcopy id-reuse hazard probed 0/6 (cyclic trees outlive the deepcopy window) — defensive clear recommended in cvw. Cross-links added to b8g and d86.","status":"closed","priority":2,"issue_type":"task","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-08-02T07:46:56Z","created_by":"Bruno Postle","updated_at":"2026-08-02T08:20:42Z","started_at":"2026-08-02T07:58:03Z","closed_at":"2026-08-02T08:20:42Z","close_reason":"Closed","dependency_count":0,"dependent_count":0,"comment_count":0} @@ -77,6 +77,7 @@ {"id":"homemaker-py-nyb","title":"High-locality topology operators (mutation + subtree crossover)","description":"DESIGN.md §5, §7 Phase 2, §8.4. Mutation moves: divide/undivide leaf, swap children, rotate cut, retype leaf, per-floor delta edits, storey add/delete (cf. Urb Mutate.pm — but geometry sliding belongs to the inner loop, not the operator set). Crossover: area-matched subtree exchange (a subtree = a contiguous region, so crossover is meaningful — Crossover.pm). Operators must be high-locality: small genome change =\u003e small phenotype change, so warm-started inner loops stay cheap.","acceptance_criteria":"Each operator produces valid genomes (oracle scores them without error); locality measured (mean fitness/geometry perturbation per operator)","status":"closed","priority":2,"issue_type":"feature","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-06-11T23:37:27Z","created_by":"Bruno Postle","updated_at":"2026-06-12T13:07:37Z","started_at":"2026-06-12T12:54:23Z","closed_at":"2026-06-12T13:07:37Z","close_reason":"operators.py lands: 7 mutations + area-matched crossover, valid-by-construction via genome.encode repair. 115/115 oracle-valid children; locality measured: geom-pert 0.07-0.33 per op, fitness-pert 0.68-0.99 (0.5^n cliff flags raw moves — warm restart + penalty reshaping confirmed load-bearing). Also fixed dom._link stale below-links on structural mutation.","dependencies":[{"issue_id":"homemaker-py-nyb","depends_on_id":"homemaker-py-k2g","type":"blocks","created_at":"2026-06-12T00:39:36Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":1,"comment_count":0} {"id":"homemaker-py-k2g","title":"Topology genome: base-floor tree + per-floor deltas + type assignment","description":"DESIGN.md §5.2, §7 Phase 2. Genome = base-floor slicing topology (primary) + per-leaf type assignment + per-floor divide/undivide deltas (Below-inheritance as regulariser; cut owned by lowest storey where its path is divided — §10). Must round-trip to/from dom.py Node trees so the oracle and inner loop consume it directly. Includes storey count and per-floor type overrides.","acceptance_criteria":"Genome \u003c-\u003e .dom round-trip on all 35 corpus files preserves fitness; multi-storey wall stacking preserved","status":"closed","priority":2,"issue_type":"feature","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-06-11T23:37:26Z","created_by":"Bruno Postle","updated_at":"2026-06-12T12:52:34Z","started_at":"2026-06-12T10:55:21Z","closed_at":"2026-06-12T12:52:34Z","close_reason":"genome.py encode/decode lands. 35/35 oracle fitness parity after round-trip (flag-on); genome fixed-point + owned-projection tests. Dead-field discovery: corpus upper storeys carry drifted dead divisions (97) and rotations (187) — canonicalised by decode, validated fitness-neutral.","dependency_count":0,"dependent_count":1,"comment_count":0} {"id":"homemaker-py-d0s","title":"Experiment: inner-loop optimiser bake-off at equal oracle budgets","description":"DESIGN.md §7 Phase 1, §8.3. DOF is only ~rooms-1 (6–7 on corpus). Compare Nelder-Mead vs CMA-ES vs batched multi-start pattern search at equal oracle-call budgets, measuring fitness gained per oracle call and wall-clock (batch-friendliness matters — §4.6). Measure, don't commit blind.","acceptance_criteria":"Table of fitness-per-budget across \u003e=3 candidates; one optimiser chosen and recorded in DESIGN.md","status":"closed","priority":2,"issue_type":"task","assignee":"Bruno Postle","owner":"bruno@postle.net","created_at":"2026-06-11T23:36:59Z","created_by":"Bruno Postle","updated_at":"2026-06-13T08:48:13Z","started_at":"2026-06-12T21:22:15Z","closed_at":"2026-06-13T08:48:13Z","close_reason":"Bake-off complete: CMA-ES confirmed as Phase 1/2 optimiser. NM wins quality per eval but sequential architecture incompatible with batching (§4.6). Compass stalls on narrow valleys. Results in DESIGN.md §8.3 and experiments/bakeoff_innerloop.*","dependencies":[{"issue_id":"homemaker-py-d0s","depends_on_id":"homemaker-py-1p0","type":"blocks","created_at":"2026-06-12T00:39:35Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} +{"id":"homemaker-py-5bv","title":"CP-SAT post-collapse repair (Fitness.collapse_global's Jacobi+2-opt QAP relaxation)","description":"homemaker-py-2g7.5 item (c), deferred (DESIGN.md §37.7). Fitness.collapse_global (fitness.py:655-847) approximates a finish-time cell\u003c-\u003eroom relabelling QAP with a Jacobi-style fixpoint iteration (_best_assignment, linear_sum_assignment warm-started each round from the previous round's neighbour labels) plus _two_opt_adjacency_polish to escape 2-cycle plateaus. DESIGN.md §25 explicitly considered and rejected OR-Tools for this exact problem 'because the project has no ortools' -- 2g7.5 has now added that dependency (for a simpler, different problem: assignment on a FIXED topology, not this finish-time relabel). This bead: replace or augment collapse_global's Jacobi+2-opt loop with an exact CP-SAT solve of the same cell\u003c-\u003eroom assignment (reusing _collapse_value's per-(leaf,code) value function so both stay consistent), verified safe against the 94g keep-better guard. Riskier than 2g7.5's seeder/reassign work since collapse_global is delicate, heavily tested, and runs inside every in-search eval when collapse_insearch=True (driver.py default) -- correctness and wall-clock regressions would be felt everywhere, not just in an opt-in flag.","status":"open","priority":3,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-08-04T07:43:33Z","created_by":"Bruno Postle","updated_at":"2026-08-04T07:43:33Z","dependencies":[{"issue_id":"homemaker-py-5bv","depends_on_id":"homemaker-py-2g7","type":"parent-child","created_at":"2026-08-04T08:44:25Z","created_by":"Bruno Postle","metadata":"{}"},{"issue_id":"homemaker-py-5bv","depends_on_id":"homemaker-py-2g7.5","type":"parent-child","created_at":"2026-08-04T08:44:03Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-v4s","title":"driver.search A/B for shapecurve warm-start/prune on real multi-storey programmes","description":"Follow-up from homemaker-py-koo (DESIGN.md §37.6): koo generalised shapecurve.py's DP to handle below-inherited multi-storey trees and validated it (DP-vs-NM agreement/false-negative bar, 200 topologies on the real examples/harbor-house, 99.5% agreement, 0 false negatives, 117.7x speedup, DESIGN.md §37.6). Not measured: the search-level payoff of shapecurve_warmstart/shapecurve_prune on a real multi-storey programme at the 6xh/wkh A/B protocol (budget=2000, seeds 0-4, driver.search mean hard/soft/fitness fails, off vs on). Best sized as a single A/B once homemaker-py-tym (leaf_sharing/co_type modelling in shapecurve.leaf_constraints) also lands, since leaf_sharing defaults True in driver.search and both programme-house and harbor-house require it by default -- measuring the combined win (multi-storey + leaf_sharing) in one pass avoids two partial A/Bs that each only apply with a flag most real runs don't use.","notes":"Depends on homemaker-py-tym landing first for the combined measurement to be meaningful.","status":"open","priority":3,"issue_type":"task","owner":"bruno@postle.net","created_at":"2026-08-03T22:23:50Z","created_by":"Bruno Postle","updated_at":"2026-08-03T22:24:36Z","dependencies":[{"issue_id":"homemaker-py-v4s","depends_on_id":"homemaker-py-tym","type":"blocks","created_at":"2026-08-03T23:24:37Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-ekc","title":"True skew-quad polygon algebra for the shape-curve DP leaf region (remove ~7-12% rectangle approximation error)","description":"homemaker-py-6xh item (DESIGN.md §37.2, 'Remaining approximation error, root-caused'). src/homemaker_layout/shapecurve.py approximates every quad (leaf or internal) as a rectangle with edge-length-derived (w,h) = ((edge0+edge2)/2, (edge1+edge3)/2) -- exact only for a true rectangle/parallelogram. DESIGN.md §37.2's 200-topology harbor-house-l0 validation root-caused both measured false positives to this approximation specifically (not to global rotation or to the rotation-parity composition rule, both already fixed/verified exact): the DP's own realised point had a leaf whose edge-length-approximated area was comfortably inside its feasible bound but whose true geometry.area (a real, slightly non-parallelogram quad) fell just below the true lower bound -- an ~8-12% gap, the same magnitude as harbor-house-l0's own plot-level residual skew. Needs: either (a) replace the rectangle approximation with true skew-quad polygon algebra (a harder closed-form derivation, or a numerically-solved per-leaf feasible region), or (b) at minimum re-characterise the error's magnitude on a LESS rectangular plot than harbor-house-l0's near-rectangular trapezoid (§37.2 flagged this as untested and likely worse elsewhere) so shapecurve_warmstart's real-world false-positive rate is known before wider rollout.","status":"open","priority":3,"issue_type":"task","owner":"bruno@postle.net","created_at":"2026-08-03T17:30:23Z","created_by":"Bruno Postle","updated_at":"2026-08-03T17:30:23Z","dependencies":[{"issue_id":"homemaker-py-ekc","depends_on_id":"homemaker-py-2g7","type":"parent-child","created_at":"2026-08-03T18:31:51Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":0,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-tym","title":"leaf_sharing/co_type target-adjustment modelling in shapecurve.leaf_constraints","description":"homemaker-py-6xh item 4 (DESIGN.md §37.2/§37.4). src/homemaker_layout/shapecurve.py's leaf_constraints() uses each leaf's own type's base (target, sigma) params only -- it does not model the leaf-sharing/co_type k-scaling (target*=k, sigma adjustment) that fitness.py's quality_size applies for shared/multi-use leaves. shapecurve.eligible() currently guards this by excluding any run with leaf_sharing/superpose/max_share/multi_use on, so the DP warm-start never fires for those runs -- but leaf_sharing defaults to True in driver.search(), so most real runs are excluded today. Needs: read fitness.py's actual k-scaling formula (quality_size's leaf-sharing branch) and mirror it in leaf_constraints so (amin, amax) reflects a shared leaf's k-multiplied target, then relax shapecurve.eligible's leaf_sharing/max_share guards accordingly (superpose/multi_use may need separate analysis -- check whether either changes the per-leaf target formula the same way share does, or a different one).","status":"open","priority":3,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-08-03T17:30:01Z","created_by":"Bruno Postle","updated_at":"2026-08-03T17:30:01Z","dependencies":[{"issue_id":"homemaker-py-tym","depends_on_id":"homemaker-py-2g7","type":"parent-child","created_at":"2026-08-03T18:31:49Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":0,"dependent_count":1,"comment_count":0} @@ -128,27 +129,27 @@ {"id":"homemaker-py-erc.6","title":"Experiment: inner-loop slack-expansion objective term","description":"Inner-loop counterpart to plot-fill construction. If Diagnostic B shows the inner loop has room to expand leaves into slack but no objective gradient to do so (the scalar rewards hitting target area but not exceeding it where slack exists), add a term/incentive so the ratio optimiser pushes leaf boundaries out to consume neighbouring slack and satisfy size, rather than parking at target.\n\nCONDITIONAL on Diagnostic B: build this only if B localizes the gap to the inner loop (room to expand, no gradient); if B shows construction targets too-small dims, prefer the plot-fill construction sibling. Must preserve the §5.4 inner-loop cliff / §4.9 lexicographic protection — the term sits where it cannot displace the fail-count ordering. A/B vs §12.2 baseline, seeds 0/1/2, 20000 evals, staged, default-OFF. Record DESIGN.md §13.6.","notes":"DEPRIORITISED by Diagnostic B (§13.2). B shows the inner loop CANNOT repair undersize: the slack is depth-driven maldistribution baked into the frozen topology, and the equal-offset ratio DOF cannot shrink a 14x leaf to feed a starved one without trading into shape fails (0.5^n cliff). Wrong DOF and wrong direction — the blocker is slicing POSITION, not a missing expansion reward. Fix belongs upstream in construction/topology (erc.4 re-scoped, erc.3). Keep as a low-priority follow-up only if a depth-balanced construction still leaves a residual size gradient the inner loop could pick up.","status":"closed","priority":4,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-06-22T23:16:24Z","created_by":"Bruno Postle","updated_at":"2026-06-28T13:22:22Z","closed_at":"2026-06-28T13:22:22Z","close_reason":"wont-fix (DESIGN §13.7): Diag B (§13.2) showed the inner loop cannot repair undersize (wrong DOF — slicing position, frozen-topology ratios). Superseded by depth-balanced construction (erc.4). Condition unmet.","dependencies":[{"issue_id":"homemaker-py-erc.6","depends_on_id":"homemaker-py-erc","type":"parent-child","created_at":"2026-06-23T00:16:23Z","created_by":"Bruno Postle","metadata":"{}"},{"issue_id":"homemaker-py-erc.6","depends_on_id":"homemaker-py-erc.2","type":"blocks","created_at":"2026-06-23T00:16:47Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-erc.5","title":"Experiment: compactness-aware cuts (minimize leaf perimeter/area)","description":"Attacks the #1 factor, crinkliness (346) — a per-leaf perimeter/area property DISTINCT from proportion (aspect ratio). Proportion-aware seeding (leu.2) sizes splits but does not bias toward balanced, square-ish subdivision. Add a KD-tree-style 'keep both children compact' cut rule (prefer the cut orientation/position that minimises summed child perimeter/area) in construction.\n\nCONDITIONAL on Diagnostic A: if A shows per-leaf shape-fail is FLAT across densities (floor intrinsic to slicing density), better cuts at the same leaf count will not pay → this should be closed wont-fix in favour of leaf-sharing. Only build if A shows shape-fail RISES with density. A/B vs §12.2 baseline, seeds 0/1/2, 20000 evals, staged, default-OFF. Record DESIGN.md §13.5.","notes":"DEPRIORITISED by erc.1 verdict (§13.1): per-leaf shape-fail flat vs slicing density and cuts already squarest (_size_divisions_from_targets picks squarest rotation) yet still ~1.8 fails/leaf =\u003e little compactness headroom at fixed leaf count. Floor is intrinsic to leaf COUNT, not cut quality. Revisit only if leaf-sharing (erc.3) underdelivers.","status":"closed","priority":4,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-06-22T23:16:21Z","created_by":"Bruno Postle","updated_at":"2026-06-28T13:22:17Z","closed_at":"2026-06-28T13:22:17Z","close_reason":"wont-fix (DESIGN §13.7): Diag A (§13.1) showed the floor is intrinsic to leaf COUNT not cut quality; revisit condition was 'only if leaf-sharing underdelivers' but leaf-sharing OVER-delivered (−32…−39%, §13.3). Condition unmet.","dependencies":[{"issue_id":"homemaker-py-erc.5","depends_on_id":"homemaker-py-erc","type":"parent-child","created_at":"2026-06-23T00:16:21Z","created_by":"Bruno Postle","metadata":"{}"},{"issue_id":"homemaker-py-erc.5","depends_on_id":"homemaker-py-erc.1","type":"blocks","created_at":"2026-06-23T00:16:43Z","created_by":"Bruno Postle","metadata":"{}"}],"dependency_count":1,"dependent_count":0,"comment_count":0} {"id":"homemaker-py-2g5","title":"Rebuild occlusion/daylight/sun subsystem in Python (post-Phase-5, after optimisation fully native)","description":"DESIGN.md §6 port scope — a whole subsystem, not a term. quality_daylight (Leaf.pm:281-296) needs Urb::Misc::Sun + Urb::Field::Occlusion (+CIESky); quality_uncrinkliness also takes the occlusion object. Indoor spaces return 1 for daylight; cost is outdoor spaces + crinkliness. Port Sun_horizontal (262980-minute normalisation) and the occlusion wall set from Dom-\u003eWalls.","acceptance_criteria":"Daylight and crinkliness factors match Perl (float tolerance) across the corpus, including multi-storey cases","notes":"Re-scoped 2026-06-12: occlusion disabled in the Urb oracle instead of ported (see homemaker-py-gp2). Native fitness ships with simple crinkliness (illumination factor = 1, in homemaker-py-gnw). This issue is now the eventual Python occlusion rebuild, only after optimisation works entirely in Python. Restores outdoor-daylight and shaded-wall selection pressure.\nReframed 2026-06-17: orthogonal to epic homemaker-py-c4c. This is fitness FIDELITY (restoring daylight + shaded-wall selection pressure to match Perl), not search CAPABILITY — it changes what 'good' means, not the search's ability to find good. It will NOT improve final designs in the sense currently sought. Stays P4, deferred until the topology-search-quality epic lands and optimisation is fully native.","status":"open","priority":4,"issue_type":"feature","owner":"bruno@postle.net","created_at":"2026-06-11T23:38:25Z","created_by":"Bruno Postle","updated_at":"2026-06-17T19:14:48Z","dependency_count":0,"dependent_count":0,"comment_count":0} -{"_type":"memory","key":"urb-oracle-nondeterminism-urb-fitness-pl-output-varies","value":"Urb oracle nondeterminism: urb-fitness.pl output varies run-to-run from Perl hash-order randomisation — .fails line ORDER shuffles (compare sorted, use oracle.Score.fail_lines) and the score float can flip by ~1 ULP (compare with math.isclose rel_tol=1e-12, never ==). Not a batching artifact; affects single runs too. Matters for the Phase 3 native-fitness parity gate (homemaker-py-uxz)."} -{"_type":"memory","key":"collapse-global-s-jacobi-adjacency-relaxation-homemaker-py","value":"collapse_global's Jacobi adjacency relaxation (homemaker-py-94g) is a synchronous per-round linear-assignment re-solve, which can 2-cycle indefinitely between two labellings that each satisfy ZERO adjacency requirements even though a permutation satisfying ALL of them exists -- proven on a minimal 4-cell chain (p1-q1-p2-q2, two disjoint adjacency pairs p1\u003c-\u003ep2/q1\u003c-\u003eq2) in test_two_opt_polish_escapes_jacobi_plateau. homemaker-py-9wi added Fitness._two_opt_adjacency_polish: a same-level pairwise-swap local search run after the Jacobi fixpoint, gated behind collapse_global(local_search=True) (default off, exposed as homemaker-collapse --local-search). Monotone by construction (a swap is kept only if it strictly increases total reward). Empirically on the 11 harbor-house evolved-*.dom/3m.dom/materialised-3M.dom layouts: 10 matched Jacobi-only exactly, 0 regressed, and evolved-anneal-3M.dom improved 21-\u003e19 fails (fixed a genuine mutual da1\u003c-\u003ek1 adjacency miss the Jacobi loop couldn't reach)."} -{"_type":"memory","key":"never-use-corpus-filenames-candidate-001-dom-candidate","value":"Never use corpus filenames (candidate-001.dom, candidate-002.dom, generated.dom, init.dom, etc.) as --output targets when running experiments. These are test fixtures. Always write experimental outputs to scratch/ or a timestamped path. Lesson from 2026-06-14: warm-start runs overwrote candidate-001/002.dom and broke graph tests."} -{"_type":"memory","key":"programme-house-optimisation-result-2026-06-14-15","value":"Programme-house optimisation result (2026-06-14/15): best achievable is 1 fail (l1 wrong level, score ~0.005). 0 fails is geometrically impossible: l1 (min 27m²) must occupy ll (~23m²) at level 0, which eliminates the t3-adj-C provider; dividing ll into lll(l1)+llr(C) gives llr proportion ~6:1 (fails). Python memetic optimizer achieves 1 fail in 50k evals vs Perl optimiser's 2-3 fails. Winning topology: TWO C nodes at level 0 — ll(C) for t3-adj-C via geometric contact, rl(C) for staircase via tree-sibling adjacency to rrr(O). Best .dom: scratch/from-warmstart-fixed.dom and scratch/from-compound3-fixed.dom."} -{"_type":"memory","key":"proportion-aware-constructive-seeding-leu-2-12-2","value":"Proportion-aware constructive seeding (leu.2/§12.2): sizing seed cuts from target AREAS only regresses (thin slivers wreck aspect); you must ALSO pick each cut's rotation for child squareness. It is a convergence ACCELERATOR via a deeper local optimum around the constructed topology: wins where that topology is roughly right and budget is scarce (harbor -13%, maple -10% at 20k evals) but DELAYS small programmes where the seed must be restructured by undivide (programme-house regresses at fixed budget, yet reaches the floor given budget - speed, not asymptote). Default-on. Also: n_storeys must honour storey_minimum, not just level: keys (programme-house storey_minimum:2, all rooms level:0 - was seeded 1 storey short; cq1)."} -{"_type":"memory","key":"homemaker-py-3l6-fix-leaf-sharing-evolve-runs","value":"homemaker-py-3l6 fix: leaf-sharing evolve runs now auto-finish before write via driver.polish_finish — unfold_shared_leaves() then a warm-started leaf_sharing=False polish search (--polish-budget, default budget//2). Makes the written .dom honest under canonical homemaker-fitness (internal==canonical when leaf_sharing off). Interrupt path forces polish_budget=0 (unfold+rescore only). This is yaa's unfold-then-polish, made automatic; Schedule B annealing is still kpu."} +{"_type":"memory","key":"cli-tool-style-prefer-python-m-homemaker-module","value":"CLI tool style: prefer python -m homemaker.module --parameters pattern, installable via pip install -e . with pyproject.toml entry_points. Not standalone bin/ scripts."} {"_type":"memory","key":"island-model-psk-14-is-a-null-priming","value":"Island model (psk, §14) is a NULL: priming a population from N converged independent elites + crossover-heavy migration does not beat best-of-N at equal total budget (maple island 124 vs control 116). The child_probe instrument shows WHY: area-matched crossover across independently-converged elites almost never synthesizes (1-3 of ~64 children beat the better parent, max drop 2-5) because the slicing encoding is non-canonical (9gp), so splices are disruptive not combinatorial. Search-machinery null #3 after graded-objective and niching/restarts; residual stays geometry/shape-bound."} -{"_type":"memory","key":"experiment-seeding-pitfall-run-search-scaled-py-s","value":"Experiment seeding pitfall: run_search_scaled.py's default PH_SEED (c964…dom) is a FINISHED programme-house design — passing it warm-starts and floors at ~3 fails, NOT a blank-slate topology search. For blank-slate runs comparable to §11.5/§11.6 baselines, seed from examples/programme-house/init.dom (a bare undivided plot; driver bootstrap auto-triggers only on bare plots). Bit the 6zy sweep — first pass used c964 and falsely showed 3-fail floor across the whole grid."} -{"_type":"memory","key":"unfold-strategy-for-shared-leaves-homemaker-py-8iv","value":"Unfold strategy for shared leaves (homemaker-py-8iv, resolved 2026-07-16): use the BALANCED GRID (operators._grow_balanced/_size_subtree_equal), NOT circulation-aware slicing. Slicing a shared leaf perpendicular to its access edge so every child touches the corridor was implemented + A/B-tested and LOST decisively (150k-eval warm-start polish from evolved-3M: slice 41 fails/3.5e-14 vs grid 25 fails/2.4e-09, grid ahead at every milestone). Reason: k rooms all touching one wall are intrinsically thin slices; that geometric debt (proportion/long/width) is unfixable without topology change, while the grid's squarer children let local search re-route access cheaply via level_retype/place_missing/level_fix. Lesson: at the sharing-\u003eno-sharing transition, prioritise squarer children and leave access to local search; do not reintroduce slicing in Schedule B (kpu)."} -{"_type":"memory","key":"user-preference-bruno-this-is-a-fedora-system","value":"User preference (Bruno): this is a Fedora system — NEVER install Python packages via pip without asking first; always ask whether to install the rpm via dnf (e.g. python3-cma) before considering pip. Applies to any dependency additions."} -{"_type":"memory","key":"warm-x0-initialization-bug-pattern-when-a-topology","value":"warm_x0 initialization bug pattern: when a topology operator explicitly sets division ratios on a newly-created node (e.g. compound_fix sets node.division=[0.25,0.25] for t3), parent.ratios has no entry for that node (it was a leaf). warm_x0 defaults it to 0.5, corrupting the inner loop's starting point and making the operator invisible to lex comparison. Fix: only propagate child ratios for nodes where the parent node was NOT already divided; stale hidden nodes revealed by structural mutations (swap flipping b.below) must NOT contribute their pre-writeback values. See driver.py lines 259-267 (fixed 2026-06-14)."} -{"_type":"memory","key":"9o5-multi-use-leaves-is-path-a-superposition","value":"9o5 multi-use leaves is path (a) — superposition as SEARCH RELAXATION that COLLAPSES to specific usage at the end, NOT path (b) loose-fit/no-collapse. Bruno's intent: codes with SIMILAR leaf requirements form an interchangeable equivalence class; during evolution the solver doesn't commit which leaf serves which specific usage (smoother landscape, no fighting over exact leaf usage); at the end the layout is CONDENSED to specific usages by brute-forcing the in-class assignment (3 interchangeable usages over 3 leaves = 3! = 6 combinations to check, pick best). 'Derive automatically' compatibility = requirement-similarity grouping. This reverses the issue's stated 'path b preferred' note."} +{"_type":"memory","key":"urb-oracle-nondeterminism-urb-fitness-pl-output-varies","value":"Urb oracle nondeterminism: urb-fitness.pl output varies run-to-run from Perl hash-order randomisation — .fails line ORDER shuffles (compare sorted, use oracle.Score.fail_lines) and the score float can flip by ~1 ULP (compare with math.isclose rel_tol=1e-12, never ==). Not a batching artifact; affects single runs too. Matters for the Phase 3 native-fitness parity gate (homemaker-py-uxz)."} +{"_type":"memory","key":"ld2-13-6-interior-o-seed-diagnostic-all","value":"ld2/§13.6 interior-O seed diagnostic: ALL crinkliness fails in the constructed bal+share seed are UNDER-exposed (crink\u003c0.62, landlocked rooms with no facade + no uncovered-O neighbour) — zero over-exposed sliver fails. So the erc crinkliness residual is genuine under-daylighting, validating the interior light-well premise. Default outside_divisor=6 was too sparse (null: harbor 147-\u003e142, crinkliness even rose). odiv=3 is the seed-optimal joint setting: harbor seed fails 147-\u003e129 (-18), maple 219-\u003e206 (-14), landlocked fails drop, at cost of more leaves (harbor +4, maple +8). Because it ADDS leaves it carries the §13.4 wash-out risk; A/B to convergence pending."} {"_type":"memory","key":"experiment-harness-gotcha-the-leaf-sharing-relaxed-objective","value":"Experiment harness gotcha: the leaf-sharing RELAXED objective (§13.3) is injected ONLY by monkeypatching fitness.load_config in the parent process (run_staged_search.py / probe scripts). This is parent-process-only and does NOT propagate into ProcessPoolExecutor workers (n_workers\u003e1), which re-import fitness fresh and score under the STRICT on-disk patterns.config -\u003e r.n_fails MISMATCH (worker strict vs parent relaxed re-score). ALL §13.x floor runs were therefore SERIAL. Any future PARALLEL leaf-sharing experiment will silently mis-score until leaf_sharing lives on disk/CLI (tracked: homemaker-py-x3b). The parallel driver itself is correct; both paths score via load_config(programme_dir)."} -{"_type":"memory","key":"strategy-decision-2026-06-12-bruno-occlusion-daylight","value":"Strategy decision 2026-06-12 (Bruno): occlusion/daylight is ORTHOGONAL to building a scalable optimiser. Disable it in Urb (env flag, homemaker-py-gp2) rather than port it; native fitness uses simple crinkliness (illumination factor = 1); rebuild occlusion in Python only after optimisation is fully native (homemaker-py-2g5, now P4). Consequence: all scores change when the flag flips — re-baseline corpus/.score, DESIGN \\$4.5 gains, gate bars at one clean boundary AFTER homemaker-py-1p0 closes; Phase-2 urb-evolve benchmark must run with the same flag."} -{"_type":"memory","key":"correction-to-urb-fitness-bug-memory-bruno-2026","value":"CORRECTION to urb-fitness-bug memory (Bruno, 2026-06-12): 'C' is NOT a 'covered' type — Is_Covered is a geometric predicate (indoor space above). Urb's generic types are canonically UPPERCASE: C=circulation, O=outside, S=sahn (get_space_types qw/C O S/; corpus is 100% uppercase, never 'c'/'o' leaves). The mixed-case designs that fired the latent ratio_type first-match bug were created by homemaker's own operator type pool emitting lowercase 'c'/'o' — fixed: driver/operators now emit uppercase generics only, and class checks use t[0].lower() in 'cos'. The Urb class-sum patch stays as defensive hardening (zero impact on canonical designs). Native port (3y7/gnw): treat type classes case-insensitively, generics canonically uppercase."} +{"_type":"memory","key":"experiment-seeding-pitfall-run-search-scaled-py-s","value":"Experiment seeding pitfall: run_search_scaled.py's default PH_SEED (c964…dom) is a FINISHED programme-house design — passing it warm-starts and floors at ~3 fails, NOT a blank-slate topology search. For blank-slate runs comparable to §11.5/§11.6 baselines, seed from examples/programme-house/init.dom (a bare undivided plot; driver bootstrap auto-triggers only on bare plots). Bit the 6zy sweep — first pass used c964 and falsely showed 3-fail floor across the whole grid."} +{"_type":"memory","key":"warm-x0-initialization-bug-pattern-when-a-topology","value":"warm_x0 initialization bug pattern: when a topology operator explicitly sets division ratios on a newly-created node (e.g. compound_fix sets node.division=[0.25,0.25] for t3), parent.ratios has no entry for that node (it was a leaf). warm_x0 defaults it to 0.5, corrupting the inner loop's starting point and making the operator invisible to lex comparison. Fix: only propagate child ratios for nodes where the parent node was NOT already divided; stale hidden nodes revealed by structural mutations (swap flipping b.below) must NOT contribute their pre-writeback values. See driver.py lines 259-267 (fixed 2026-06-14)."} +{"_type":"memory","key":"programme-house-optimisation-result-2026-06-14-15","value":"Programme-house optimisation result (2026-06-14/15): best achievable is 1 fail (l1 wrong level, score ~0.005). 0 fails is geometrically impossible: l1 (min 27m²) must occupy ll (~23m²) at level 0, which eliminates the t3-adj-C provider; dividing ll into lll(l1)+llr(C) gives llr proportion ~6:1 (fails). Python memetic optimizer achieves 1 fail in 50k evals vs Perl optimiser's 2-3 fails. Winning topology: TWO C nodes at level 0 — ll(C) for t3-adj-C via geometric contact, rl(C) for staircase via tree-sibling adjacency to rrr(O). Best .dom: scratch/from-warmstart-fixed.dom and scratch/from-compound3-fixed.dom."} +{"_type":"memory","key":"user-preference-bruno-this-is-a-fedora-system","value":"User preference (Bruno): this is a Fedora system — NEVER install Python packages via pip without asking first; always ask whether to install the rpm via dnf (e.g. python3-cma) before considering pip. Applies to any dependency additions."} {"_type":"memory","key":"adjacency-in-binary-slicing-tree-is-structural-not","value":"Adjacency in binary slicing tree is structural, not geometric: the inner-loop NM cannot fix topological adjacency failures. Two paths exist: (1) tree-sibling adjacency — a node is adjacent to its sibling in the tree; (2) cross-zone geometric adjacency — leaves from different subtrees that happen to share a boundary. Staircase/adjacency fails require a topology mutation that changes which nodes are siblings or which zones touch. This was proved empirically on programme-house: staircase fail from rot=0 layout could not be fixed by NM but was fixed by level_retype creating a two-C topology (2026-06-14/15)."} {"_type":"memory","key":"collapse-global-94g-and-any-label-usage-optimisation","value":"collapse_global (94g) and any label/usage optimisation CANNOT fix geometry-intrinsic fails. The harbor-house 15-fail best layout contains long-thin cells that are useless whatever room usage is assigned — their width/proportion/crinkliness fails are shape-bound, not label slack. Two consequences: (1) do not over-claim collapse gains — only ~2-3 of that layout's fails are reclaimable relabel slack, the rest are geometry- or building-level bound; (2) the threshold objective must not be tuned to 'pass' a degenerate cell via a permissive room type — a metric-pass on a physically useless space is gaming, not a fix. Real remedies for these are geometry/topology search (cell shape) and circulation placement, filed separately, not the collapse."} -{"_type":"memory","key":"ld2-13-6-interior-o-seed-diagnostic-all","value":"ld2/§13.6 interior-O seed diagnostic: ALL crinkliness fails in the constructed bal+share seed are UNDER-exposed (crink\u003c0.62, landlocked rooms with no facade + no uncovered-O neighbour) — zero over-exposed sliver fails. So the erc crinkliness residual is genuine under-daylighting, validating the interior light-well premise. Default outside_divisor=6 was too sparse (null: harbor 147-\u003e142, crinkliness even rose). odiv=3 is the seed-optimal joint setting: harbor seed fails 147-\u003e129 (-18), maple 219-\u003e206 (-14), landlocked fails drop, at cost of more leaves (harbor +4, maple +8). Because it ADDS leaves it carries the §13.4 wash-out risk; A/B to convergence pending."} -{"_type":"memory","key":"multi-storey-staircase-consistency-when-dividing-or-retyping","value":"Multi-storey staircase consistency: when dividing or retyping a circulation (C) leaf at one level, the same structural change should be propagated to the matching leaf on ALL other storeys so the stair core path is maintained. The optimizer cannot fix staircase disruptions through trial-and-error geometry alone — it requires a synchronized multi-level operator that applies the same topology change to every storey simultaneously."} -{"_type":"memory","key":"cli-tool-style-prefer-python-m-homemaker-module","value":"CLI tool style: prefer python -m homemaker.module --parameters pattern, installable via pip install -e . with pyproject.toml entry_points. Not standalone bin/ scripts."} -{"_type":"memory","key":"deceptive-valleys-in-topology-search-when-every-single","value":"Deceptive valleys in topology search: when every single-step mutation from a target state passes through a high-fail intermediary (e.g. level_fix displaces a room into 5+ new fails), a compound operator that atomically applies two coordinated changes can escape. Design compound operators to land on the low-fail state directly, bypassing the deceptive gradient. Programme-house example: level_compound_fix atomically moves the level-constrained room AND re-inserts the displaced room adjacent to C in one step (operators.py, 2026-06-14)."} -{"_type":"memory","key":"homemaker-py-pythonpath-set-pythonpath-home-bruno-src","value":"homemaker-layout PYTHONPATH: package installed as 'homemaker-layout' via pip install -e . so 'import homemaker_layout' works from anywhere without PYTHONPATH. For running tests use 'python -m pytest' from project root /home/bruno/src/homemaker-layout (pyproject.toml adds src/ automatically). Never try pip show homemaker — that's the old homemaker-addon conflict."} -{"_type":"memory","key":"run-to-run-reproducibility-in-homemaker-layout-serial","value":"Run-to-run reproducibility in homemaker-layout: serial search (workers=1) is byte-for-byte deterministic; parallel (workers\u003e1) is now deterministic too AFTER fixing driver._run_batch to admit futures in submission order (was as_completed/completion order, bug xcy). Reproducibility holds only for a FIXED worker count — serial vs parallel differ because children-per-iteration is 1 vs n_workers (different batch granularity), which is expected, not a bug. The constructive seeder was NEVER nondeterministic: _assign_adjacency_aware has unique idx tiebreaks; comparing topologies with Python builtin hash() of the signature STRING is invalid (PYTHONHASHSEED salts str hashing per process) — use a stable hash (sha1) or genome.signature equality."} +{"_type":"memory","key":"collapse-global-s-jacobi-adjacency-relaxation-homemaker-py","value":"collapse_global's Jacobi adjacency relaxation (homemaker-py-94g) is a synchronous per-round linear-assignment re-solve, which can 2-cycle indefinitely between two labellings that each satisfy ZERO adjacency requirements even though a permutation satisfying ALL of them exists -- proven on a minimal 4-cell chain (p1-q1-p2-q2, two disjoint adjacency pairs p1\u003c-\u003ep2/q1\u003c-\u003eq2) in test_two_opt_polish_escapes_jacobi_plateau. homemaker-py-9wi added Fitness._two_opt_adjacency_polish: a same-level pairwise-swap local search run after the Jacobi fixpoint, gated behind collapse_global(local_search=True) (default off, exposed as homemaker-collapse --local-search). Monotone by construction (a swap is kept only if it strictly increases total reward). Empirically on the 11 harbor-house evolved-*.dom/3m.dom/materialised-3M.dom layouts: 10 matched Jacobi-only exactly, 0 regressed, and evolved-anneal-3M.dom improved 21-\u003e19 fails (fixed a genuine mutual da1\u003c-\u003ek1 adjacency miss the Jacobi loop couldn't reach)."} +{"_type":"memory","key":"correction-to-urb-fitness-bug-memory-bruno-2026","value":"CORRECTION to urb-fitness-bug memory (Bruno, 2026-06-12): 'C' is NOT a 'covered' type — Is_Covered is a geometric predicate (indoor space above). Urb's generic types are canonically UPPERCASE: C=circulation, O=outside, S=sahn (get_space_types qw/C O S/; corpus is 100% uppercase, never 'c'/'o' leaves). The mixed-case designs that fired the latent ratio_type first-match bug were created by homemaker's own operator type pool emitting lowercase 'c'/'o' — fixed: driver/operators now emit uppercase generics only, and class checks use t[0].lower() in 'cos'. The Urb class-sum patch stays as defensive hardening (zero impact on canonical designs). Native port (3y7/gnw): treat type classes case-insensitively, generics canonically uppercase."} +{"_type":"memory","key":"9o5-multi-use-leaves-is-path-a-superposition","value":"9o5 multi-use leaves is path (a) — superposition as SEARCH RELAXATION that COLLAPSES to specific usage at the end, NOT path (b) loose-fit/no-collapse. Bruno's intent: codes with SIMILAR leaf requirements form an interchangeable equivalence class; during evolution the solver doesn't commit which leaf serves which specific usage (smoother landscape, no fighting over exact leaf usage); at the end the layout is CONDENSED to specific usages by brute-forcing the in-class assignment (3 interchangeable usages over 3 leaves = 3! = 6 combinations to check, pick best). 'Derive automatically' compatibility = requirement-similarity grouping. This reverses the issue's stated 'path b preferred' note."} +{"_type":"memory","key":"homemaker-py-3l6-fix-leaf-sharing-evolve-runs","value":"homemaker-py-3l6 fix: leaf-sharing evolve runs now auto-finish before write via driver.polish_finish — unfold_shared_leaves() then a warm-started leaf_sharing=False polish search (--polish-budget, default budget//2). Makes the written .dom honest under canonical homemaker-fitness (internal==canonical when leaf_sharing off). Interrupt path forces polish_budget=0 (unfold+rescore only). This is yaa's unfold-then-polish, made automatic; Schedule B annealing is still kpu."} +{"_type":"memory","key":"never-use-corpus-filenames-candidate-001-dom-candidate","value":"Never use corpus filenames (candidate-001.dom, candidate-002.dom, generated.dom, init.dom, etc.) as --output targets when running experiments. These are test fixtures. Always write experimental outputs to scratch/ or a timestamped path. Lesson from 2026-06-14: warm-start runs overwrote candidate-001/002.dom and broke graph tests."} {"_type":"memory","key":"urb-fitness-bug-found-fixed-2026-06-12","value":"Urb fitness bug found+fixed 2026-06-12 (patch in /home/bruno/src/urb, uncommitted): ProgrammeDriven.pm ratio_o/ratio_type grepped case-insensitively over the ratios hash and took the FIRST key — nondeterministic (x4.5 score swings) for designs with mixed-case type classes (both 'c' circulation and 'C' covered). Fixed to SUM the class (matches Is_Circulation//Is_Outside semantics); 35/35 corpus scores unchanged. CRITICAL for homemaker-py-3y7/gnw: the native port must implement class-SUM ratios. Building.pm has the same unpatched pattern (site-driven path, not used by our oracle). Also: the memetic search reward-hacked this bug before the fix — search results predating it are noise artifacts."} +{"_type":"memory","key":"deceptive-valleys-in-topology-search-when-every-single","value":"Deceptive valleys in topology search: when every single-step mutation from a target state passes through a high-fail intermediary (e.g. level_fix displaces a room into 5+ new fails), a compound operator that atomically applies two coordinated changes can escape. Design compound operators to land on the low-fail state directly, bypassing the deceptive gradient. Programme-house example: level_compound_fix atomically moves the level-constrained room AND re-inserts the displaced room adjacent to C in one step (operators.py, 2026-06-14)."} +{"_type":"memory","key":"multi-storey-staircase-consistency-when-dividing-or-retyping","value":"Multi-storey staircase consistency: when dividing or retyping a circulation (C) leaf at one level, the same structural change should be propagated to the matching leaf on ALL other storeys so the stair core path is maintained. The optimizer cannot fix staircase disruptions through trial-and-error geometry alone — it requires a synchronized multi-level operator that applies the same topology change to every storey simultaneously."} +{"_type":"memory","key":"proportion-aware-constructive-seeding-leu-2-12-2","value":"Proportion-aware constructive seeding (leu.2/§12.2): sizing seed cuts from target AREAS only regresses (thin slivers wreck aspect); you must ALSO pick each cut's rotation for child squareness. It is a convergence ACCELERATOR via a deeper local optimum around the constructed topology: wins where that topology is roughly right and budget is scarce (harbor -13%, maple -10% at 20k evals) but DELAYS small programmes where the seed must be restructured by undivide (programme-house regresses at fixed budget, yet reaches the floor given budget - speed, not asymptote). Default-on. Also: n_storeys must honour storey_minimum, not just level: keys (programme-house storey_minimum:2, all rooms level:0 - was seeded 1 storey short; cq1)."} +{"_type":"memory","key":"run-to-run-reproducibility-in-homemaker-layout-serial","value":"Run-to-run reproducibility in homemaker-layout: serial search (workers=1) is byte-for-byte deterministic; parallel (workers\u003e1) is now deterministic too AFTER fixing driver._run_batch to admit futures in submission order (was as_completed/completion order, bug xcy). Reproducibility holds only for a FIXED worker count — serial vs parallel differ because children-per-iteration is 1 vs n_workers (different batch granularity), which is expected, not a bug. The constructive seeder was NEVER nondeterministic: _assign_adjacency_aware has unique idx tiebreaks; comparing topologies with Python builtin hash() of the signature STRING is invalid (PYTHONHASHSEED salts str hashing per process) — use a stable hash (sha1) or genome.signature equality."} +{"_type":"memory","key":"strategy-decision-2026-06-12-bruno-occlusion-daylight","value":"Strategy decision 2026-06-12 (Bruno): occlusion/daylight is ORTHOGONAL to building a scalable optimiser. Disable it in Urb (env flag, homemaker-py-gp2) rather than port it; native fitness uses simple crinkliness (illumination factor = 1); rebuild occlusion in Python only after optimisation is fully native (homemaker-py-2g5, now P4). Consequence: all scores change when the flag flips — re-baseline corpus/.score, DESIGN \\$4.5 gains, gate bars at one clean boundary AFTER homemaker-py-1p0 closes; Phase-2 urb-evolve benchmark must run with the same flag."} +{"_type":"memory","key":"unfold-strategy-for-shared-leaves-homemaker-py-8iv","value":"Unfold strategy for shared leaves (homemaker-py-8iv, resolved 2026-07-16): use the BALANCED GRID (operators._grow_balanced/_size_subtree_equal), NOT circulation-aware slicing. Slicing a shared leaf perpendicular to its access edge so every child touches the corridor was implemented + A/B-tested and LOST decisively (150k-eval warm-start polish from evolved-3M: slice 41 fails/3.5e-14 vs grid 25 fails/2.4e-09, grid ahead at every milestone). Reason: k rooms all touching one wall are intrinsically thin slices; that geometric debt (proportion/long/width) is unfixable without topology change, while the grid's squarer children let local search re-route access cheaply via level_retype/place_missing/level_fix. Lesson: at the sharing-\u003eno-sharing transition, prioritise squarer children and leave access to local search; do not reintroduce slicing in Schedule B (kpu)."} +{"_type":"memory","key":"homemaker-py-pythonpath-set-pythonpath-home-bruno-src","value":"homemaker-layout PYTHONPATH: package installed as 'homemaker-layout' via pip install -e . so 'import homemaker_layout' works from anywhere without PYTHONPATH. For running tests use 'python -m pytest' from project root /home/bruno/src/homemaker-layout (pyproject.toml adds src/ automatically). Never try pip show homemaker — that's the old homemaker-addon conflict."} diff --git a/CLAUDE.md b/CLAUDE.md index 701a7b9..8ad528e 100644 --- a/CLAUDE.md +++ b/CLAUDE.md @@ -71,6 +71,7 @@ Key modules: - `programme.py` — parse `patterns.config` space requirements - `solver.py` — bottom-up ratio solve (scipy) - `shapecurve.py` — Otten/Stockmeyer shape-curve DP: exact size/width/proportion feasibility for a frozen topology, any storey count (DESIGN.md §37.2/§37.4-§37.6); used as `driver._evaluate`'s NM warm-start/hard pre-filter +- `cpsat.py` — exact room-code-to-leaf labelling via OR-Tools CP-SAT for a fixed topology (DESIGN.md §37.7); replaces `operators._assign_adjacency_aware`'s greedy/beam room placement behind `assign_solver="cpsat"`, and powers the `operators.mutate_reassign` in-search repair operator - `fitness.py` — native Python fitness evaluator (replaces Perl oracle) - `fitness_cmd.py` — `homemaker-fitness` CLI entry point - `collapse_cmd.py` — `homemaker-collapse` CLI: finish-time global cell→room collapse (94g) diff --git a/DESIGN.md b/DESIGN.md index ce6fa93..1e7829f 100644 --- a/DESIGN.md +++ b/DESIGN.md @@ -4581,3 +4581,138 @@ multi-storey fixture. `tests/test_driver.py`: the multi-storey warm-start test renamed and inverted (`test_shapecurve_warmstart_handles_multistorey` now asserts `shapecurve.solve` **is** called on a multi-storey child, where it previously asserted the opposite). Full suite: 397 passed. + +## 37.7 CP-SAT type assignment for a fixed tree (`homemaker-py-2g7.5`) — PARTIAL, seeder-level positive, driver-level INCONCLUSIVE at pilot scale + +`§11.6`/`§11.7`'s greedy connected-dominating-set + hardest-constrained- +code-first room placement (`operators._assign_adjacency_aware`) was Phase +6's single biggest fail-count win, but it is a one-shot heuristic: each +room code is placed onto the locally-best open slot and never revisited. +The bead's premise: for a FIXED topology, room-code-to-leaf assignment +(~30-70 leaves, ~16-26 codes) is small enough for exact solve. DESIGN.md +§25 (line ~3089) explicitly rejected adding OR-Tools for a harder, different +problem (`Fitness.collapse_global`'s finish-time relabel) "because the +project has no ortools" — that gap is now closed (`pyproject.toml` +`ortools>=9.10`), but only for this bead's simpler fixed-topology labelling +problem; `collapse_global` itself is untouched (deferred, see below). + +**What shipped.** `src/homemaker_layout/cpsat.py`: a single pure function +`solve_room_labels(slots, codes, reqs, neighbors, context_types)` — a +boolean assignment ILP (`x[i,s]`, one code per slot) with a `sat[i,s,adj]` +reified-AND term per (code, adjacency-requirement, slot), maximising total +satisfied requirements. Matches `graph.check_adjacency`'s REAL semantics +(full-code case-insensitive prefix match) rather than the existing greedy +heuristic's first-character-only local approximation — a strictly closer +proxy for what `homemaker-fitness` actually scores. Wired in as: + +- (a) **seeder**: `_assign_adjacency_aware`/`constructive_topology`/ + `lift_base_to_storeys` gain `assign_solver: str = "greedy"|"cpsat"` + (EXPERIMENTAL, default `"greedy"` — byte-identical to before). Circulation/ + outside placement (the dominating-set step, a graph-connectivity problem, + not this bead's ~30-70-leaf combinatorial one) is unchanged either way; + only the room-code-to-slot step is replaced. Falls through to the + greedy/beam path on any solver failure (unavailable/infeasible/timeout). +- (b) **`operators.mutate_reassign`** (new `MUTATIONS` entry, default weight + 0 unless `driver.search(..., enable_reassign=True)`): the "assignment + analogue of ruin_recreate" (§23) the bead's own plan named — picks the + same kind of wing `mutate_ruin_recreate` does, but does NOT un-divide or + regrow it; only re-solves which leaf gets which code, preserving the + wing's exact room-code multiset and topology. +- (c) **post-collapse repair** — deferred, filed as `homemaker-py-5bv` + (child of `2g7`/`2g7.5`): replacing/augmenting `Fitness.collapse_global`'s + Jacobi+2-opt QAP relaxation is a materially separate, riskier change to a + delicate routine that runs inside every in-search eval by default + (`collapse_insearch=True`) — correctness/wall-clock regressions there + would be felt everywhere, not just behind an opt-in flag. + +**Two bugs found and fixed en route, both worth recording.** + +1. *Resize fragility.* First measurement (seeder-only, `constructive_topology`, + harbor-house, 6 seeds): with `proportion_aware=True` (the real default — + target-size-based ratio resizing right after assignment), `cpsat` was + WORSE than greedy on real fitness-scored secondary-adjacency fails (104 + vs 92 total) despite tying/slightly-beating it with `proportion_aware=False` + (86 vs 87). Root cause: resizing can shrink a shared-wall segment below + the door-width adjacency threshold, silently invalidating an edge the + exact solve specifically relied on — it packs satisfaction tightly + against the PRE-resize graph, leaving less slack than the greedy path's + more conservative, degree-biased placement. Fix: `_cpsat_relabel_settled` + re-runs the exact solve once more against the now-settled geometry, + right after `_size_divisions_from_targets` — cheap (same small model), + never worse (can only improve on wherever resizing left it). This is the + bead's own "§11.2 lesson … re-run assignment after geometry settles + (alternating minimization)" applied literally. After the fix: 82 vs 92 + (cpsat now ahead) at the same protocol; a wider 10-seed re-check + (`tests/test_operators.py::test_assign_cpsat_matches_or_beats_greedy_secondary_adjacency`) + holds: 13/20 seed-pairs cpsat-better, 4 ties, 3 cpsat-worse, net ~13% + fewer total real fails. +2. *Symmetry blowup.* Isolated solves on real harbor-house models (~15 + slots — trivial by variable count) occasionally stalled for multiple + seconds against a 2s `time_limit_s`, non-deterministically (system-load + dependent, since a timeout returns whatever CP-SAT's branch-and-bound + had reached). Cause: several codes sharing an identical, unreferenced + adjacency signature (e.g. four "t" bedroom instances all needing only + "c") are fully interchangeable — CP-SAT's branch-and-bound was proving + optimality across their entire permutation space. Fix: group codes that + share their own adjacency-requirement set AND are never themselves a + match target for any other code's requirement; force a canonical + slot-index ordering within each group (never removes an achievable + objective value, only the redundant permutations of it). All previously- + slow captured instances now solve in <200ms. A naive first attempt at a + fix (a lexicographic tie-break term folded into the objective) made + things WORSE (more instances timed out) by widening the objective's + coefficient range — reverted in favour of the explicit grouping + constraint above. + +**`driver.search`-level A/B: INCONCLUSIVE at pilot scale, NOT the bead's own +20k-budget/harbor+maple/3-seed acceptance protocol.** Wall-clock budget for +this session did not stretch to the bead's own acceptance criteria (~28min +per arm at budget=20000 on real harbor-house, ×3 arms ×3 seeds ×2 +programmes ≈ 8+ hours). `experiments/ab_cpsat_assign.py`, harbor-house only, +budget=3000, 3 seeds (greedy / cpsat-seed-only / cpsat+`enable_reassign`): + +| arm | mean hard | mean soft | mean fitness | mean wall | +|---|---|---|---|---| +| greedy | 21.000 | 39.000 | 9.308e-19 | 237.2s | +| cpsat | 24.000 | 36.667 | 5.313e-18 | 245.6s | +| reassign | 19.667 | 44.667 | 3.930e-19 | 240.0s | + +Mixed: `cpsat` is worse on mean HARD fails but better on soft fails and +~5.7x the mean fitness; `reassign` has the best mean hard fails but the +worst soft fails. The `reassign` arm's own mechanism was **never observed +to fire** in any of the 3 seeds (`mean_reassign_fired=0.0`) — at +budget=3000 the loop only generates ~15-20 children total, and `reassign` +carries the implicit uniform mutation weight (no boost was added — DESIGN.md +§22's `bridge_circulation` precedent measured that an un-A/B-tested weight +boost can backfire, so none was assumed here without evidence) — so the +`reassign` arm's difference from `cpsat` at this budget is attributable to +RNG/exploration noise from the changed weight-normalisation denominator, +not to the operator's own effect. `mutate_reassign` firing-and-being- +accepted IS independently confirmed at the operator level +(`tests/test_operators.py::test_reassign_fires_and_preserves_room_multiset`, +20/20 direct trials on a real seeded harbor-house design) — the pilot budget +was simply too small to give it enough draws in the full search loop. + +**Verdict: ship as opt-in EXPERIMENTAL, both default off** (matching every +other flag in this codebase) — the seeder-level win (item (a)) is real and +measured with low noise; the full-search-level payoff (matching the bead's +own acceptance criteria) is unconfirmed at pilot scale and needs a proper +larger-budget/larger-N run, tracked as follow-up work on `2g7.5` itself +(left `in_progress`, not closed — same pattern `6xh` used when its own +acceptance bar wasn't fully met). `homemaker-py-5bv` (child of `2g7`/`2g7.5`) +tracks the deferred item (c). + +**Verification.** `tests/test_cpsat.py` (5 tests): a hand-built +counter-example graph (hub + one non-hub edge) where the beam/greedy +heuristic (`operators._beam_place_rooms`) provably strands two codes that +need each other while CP-SAT finds the assignment satisfying all of them; +fixed-context credit without a decision-neighbour; over-capacity code +dropping (least-constrained first); determinism; empty-input degeneracy. +`tests/test_operators.py` (+5): CP-SAT seed satisfies +`graph.check_space_counts`/stays canonical; the 10-seed secondary-adjacency +A/B above; `mutate_reassign` no-ops without `reqs`; fires-and-preserves- +multiset over 20 trials. `tests/test_driver.py` (+2): `assign_solver` +default and `enable_reassign` default both reproduce prior runs +byte-for-byte (`sig`/`n_topologies`/`n_evals` equality), the same clean +single-variable-toggle control every other experimental flag in `driver. +search` uses. Full suite: 409 passed. diff --git a/experiments/ab_cpsat_assign.py b/experiments/ab_cpsat_assign.py new file mode 100644 index 0000000..fa65910 --- /dev/null +++ b/experiments/ab_cpsat_assign.py @@ -0,0 +1,84 @@ +"""A/B: does CP-SAT exact room-code labelling beat today's greedy/beam +heuristic seeder on the real ``driver.search`` loop (homemaker-py-2g7.5, +DESIGN.md §37.7)? + +Three arms per programme/seed: + - baseline: assign_solver="greedy" (today's default) + - cpsat: assign_solver="cpsat" (seeder only, item (a)) + - reassign: assign_solver="cpsat" + enable_reassign=True (adds item (b), + the periodic in-search re-labelling operator) + +Metric: mean (n_hard, n_soft, fitness) of ``driver.search``'s best individual +at a FIXED budget across several seeds, same format as the shapecurve A/Bs +(``experiments/ab_shapecurve_warmstart.py``) -- plus a count of how many +runs the ``reassign`` operator actually fired+was-accepted in, per the +bead's acceptance criterion. + +Usage: python experiments/ab_cpsat_assign.py [budget] [n_seeds] [programme] +""" + +from __future__ import annotations + +import sys +import time +from pathlib import Path + +from homemaker_layout import dom, driver + +EXAMPLES = Path(__file__).parent.parent / "examples" +ARMS = { + "greedy": {"assign_solver": "greedy", "enable_reassign": False}, + "cpsat": {"assign_solver": "cpsat", "enable_reassign": False}, + "reassign": {"assign_solver": "cpsat", "enable_reassign": True}, +} + + +def run_arm(seed_root: dom.Node, programme_dir: Path, budget: int, seed: int, + arm_kw: dict): + t0 = time.perf_counter() + r = driver.search( + seed_root, programme_dir, budget=budget, pop_size=8, child_budget=80, + seed_budget=200, seed=seed, **arm_kw, + ) + elapsed = time.perf_counter() - t0 + # "fired and accepted": a reassign-descended child survived tournament + # replacement into the FINAL population (lineage is per-generation, not + # cumulative -- see Individual/driver._evaluate -- so this only counts + # children born directly from a non-noop reassign, not their descendants). + fired = sum(1 for ind in r.population + if ind.lineage.startswith("reassign") and "noop" not in ind.lineage) + return r, elapsed, fired + + +def main() -> None: + budget = int(sys.argv[1]) if len(sys.argv) > 1 else 4000 + n_seeds = int(sys.argv[2]) if len(sys.argv) > 2 else 3 + programme_name = sys.argv[3] if len(sys.argv) > 3 else "harbor-house" + + programme_dir = EXAMPLES / programme_name + seed_root = dom.load(str(programme_dir / "init.dom")) + + rows: dict[str, list[tuple]] = {name: [] for name in ARMS} + for seed in range(n_seeds): + line = [f"seed {seed}:"] + for name, kw in ARMS.items(): + r, elapsed, fired = run_arm(seed_root, programme_dir, budget, seed, kw) + rows[name].append((r.best.n_hard, r.best.n_soft, r.best.fitness, elapsed, fired)) + line.append(f"{name} hard={r.best.n_hard} soft={r.best.n_soft} " + f"fit={r.best.fitness:.4g} {elapsed:.1f}s" + + (f" reassign_fired={fired}" if name == "reassign" else "")) + print(" | ".join(line), flush=True) + + print() + print(f"budget={budget} n_seeds={n_seeds} programme={programme_dir.name}") + def mean(data: list[tuple], idx: int) -> float: + return sum(d[idx] for d in data) / len(data) + + for name, data in rows.items(): + extra = f" mean_reassign_fired={mean(data, 4):.1f}" if name == "reassign" else "" + print(f"{name:9s}: mean hard={mean(data, 0):.3f} soft={mean(data, 1):.3f} " + f"fitness={mean(data, 2):.6g} wall={mean(data, 3):.1f}s{extra}") + + +if __name__ == "__main__": + main() diff --git a/pyproject.toml b/pyproject.toml index 5315682..40a587b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -10,6 +10,7 @@ dependencies = [ "shapely>=2.0", "networkx>=3.0", "cma>=3.0", + "ortools>=9.10", ] [project.scripts] diff --git a/src/homemaker_layout/cpsat.py b/src/homemaker_layout/cpsat.py new file mode 100644 index 0000000..48c5f2a --- /dev/null +++ b/src/homemaker_layout/cpsat.py @@ -0,0 +1,188 @@ +"""Exact room-code-to-leaf labelling via OR-Tools CP-SAT (homemaker-py-2g7.5). + +For a FIXED topology (and a fixed circulation/outside placement — that part +stays on ``operators._assign_adjacency_aware``'s existing connected- +dominating-set heuristic, a graph-connectivity problem, not this module's +concern), assigning the remaining room codes to the remaining leaf slots so +that secondary adjacency requirements (``k1<->da1``, ``da1<->o``, ...) are +satisfied is a small discrete optimisation: ~30-70 leaves, ~16-26 codes, +well within CP-SAT's exact-solve range in milliseconds. This replaces the +one-shot greedy/beam heuristic (``operators._assign_adjacency_aware``'s +hardest-constrained-code-first placement, ``_beam_place_rooms``) with an +exact solve of the same decision, usable as (a) a seeder, (b) the periodic +in-search ``mutate_reassign`` repair operator (``operators.py``). + +DESIGN.md §25 (line ~3089) rejected adding OR-Tools for a different, +harder QAP-relaxation problem (``Fitness.collapse_global``'s finish-time +relabelling) specifically because the project had no such dependency; that +gap is now closed, but only for this module's simpler fixed-topology +labelling problem — ``collapse_global`` itself is untouched (tracked as a +separate follow-up, DESIGN.md §37.7). + +Matches ``graph.check_adjacency``'s real semantics exactly (full-code, +case-insensitive PREFIX match, ``graph._codes_match_prefix``/ +``has_adjacency``) rather than the existing greedy heuristic's +first-character-only approximation (``_assign_adjacency_aware``'s local +``_sat``), so the objective this module maximises is a strictly closer proxy +for what ``homemaker-fitness`` actually scores. +""" + +from __future__ import annotations + +from collections.abc import Hashable +from itertools import pairwise +from typing import Any + + +def _n_secondary_adjacency(reqs: dict, code: str) -> int: + r = reqs.get(code) + return len(r.adjacency) if r else 0 + + +def solve_room_labels( + slots: list[Hashable], + codes: list[str], + reqs: dict, + neighbors: dict[Hashable, set], + context_types: dict[Hashable, set[str]], + time_limit_s: float = 2.0, +) -> dict[Hashable, str] | None: + """Assign each of ``codes`` to one of ``slots``, maximising satisfied + secondary adjacency requirements. + + ``slots``: the room slots to label (any hashable key — a ``dom.Node``, + a plain string in tests, ...). ``codes``: one entry per required room + instance; if longer than ``slots`` the least-constrained (fewest + ``reqs[code].adjacency`` entries) codes are dropped first; if shorter, + the excess slots are simply left unassigned in the returned dict (the + caller's existing leftover-handling applies, e.g. typing them ``"O"``). + ``reqs``: ``dict[code, SpaceReq]``. ``neighbors``: adjacency among + ``slots`` themselves (the room-slot subgraph). ``context_types``: for + each slot, the set of (lowercase-comparable) type strings of any FIXED + neighbour outside ``slots`` (e.g. circulation ``"C"``, outside ``"O"``, + or — for :func:`operators.mutate_reassign`'s scoped re-solve — room + codes just outside the re-solved wing). + + Returns ``{slot: code}`` (covering ``min(len(slots), len(codes))`` + slots) or ``None`` if OR-Tools is unavailable, the model is infeasible, + or no solution is found within ``time_limit_s`` — callers must always + have a defined fallback (the existing greedy/beam path) for ``None``. + """ + if not slots or not codes: + return {} + try: + from ortools.sat.python import cp_model + except ImportError: + return None + + reqs = reqs or {} + n = len(slots) + if len(codes) > n: + codes = sorted(codes, key=lambda c: -_n_secondary_adjacency(reqs, c))[:n] + k = len(codes) + idx = {slot: i for i, slot in enumerate(slots)} + + model = cp_model.CpModel() + x = {(i, s): model.NewBoolVar(f"x_{i}_{s}") for i in range(k) for s in range(n)} + for i in range(k): + model.AddExactlyOne(x[i, s] for s in range(n)) + for s in range(n): + model.Add(sum(x[i, s] for i in range(k)) <= 1) + + def _matches(code: str, prefix: str) -> bool: + return code.lower().startswith(prefix.lower()) + + # Symmetry breaking (homemaker-py-2g7.5, measured necessary on + # harbor-house: several unrelated same-requirement codes, e.g. four "t" + # bedroom instances all needing only "c", turn any permutation among + # them into an equally-optimal solution — CP-SAT's branch-and-bound can + # spend seconds proving optimality across that permutation space on an + # otherwise ~15-variable model). Two code instances are provably + # interchangeable iff they share the same OWN adjacency requirement set + # AND neither is ever required as a match target by any code (including + # each other) — group those and force a canonical slot-index ordering + # within each group; this never removes an achievable objective value, + # only the redundant permutations of it. + referenced = {a.lower() for c in codes for a in (reqs.get(c).adjacency if reqs.get(c) else [])} + + def _is_referenced(code: str) -> bool: + cl = code.lower() + return any(cl.startswith(r) for r in referenced) + + groups: dict[tuple, list[int]] = {} + for i, code in enumerate(codes): + req = reqs.get(code) + own = frozenset(a.lower() for a in (req.adjacency if req else [])) + key = ("unique", i) if _is_referenced(code) else ("group", own) + groups.setdefault(key, []).append(i) + + slot_index: dict[int, Any] = {} + for key, ids in groups.items(): + if key[0] != "group" or len(ids) < 2: + continue + for i in ids: + if i not in slot_index: + slot_index[i] = model.NewIntVar(0, n - 1, f"slotidx_{i}") + model.Add(slot_index[i] == sum(s * x[i, s] for s in range(n))) + for a, b in pairwise(ids): + model.Add(slot_index[a] <= slot_index[b]) + + neighbor_ok_cache: dict[tuple[int, str], Any] = {} + + def _neighbor_ok(s: int, adj_lower: str): + key = (s, adj_lower) + if key in neighbor_ok_cache: + return neighbor_ok_cache[key] + slot = slots[s] + fixed = context_types.get(slot, ()) + if any(_matches(t, adj_lower) for t in fixed): + neighbor_ok_cache[key] = 1 + return 1 + nbr_idxs = [idx[nb] for nb in neighbors.get(slot, ()) if nb in idx] + matches = [x[j, ns] for ns in nbr_idxs + for j, code in enumerate(codes) if _matches(code, adj_lower)] + if not matches: + neighbor_ok_cache[key] = 0 + return 0 + var = model.NewBoolVar(f"nbr_ok_{s}_{adj_lower}") + model.AddMaxEquality(var, matches) + neighbor_ok_cache[key] = var + return var + + sat_vars = [] + for i, code in enumerate(codes): + req = reqs.get(code) + if not req or not req.adjacency: + continue + seen_adj: set[str] = set() + for adj_code in req.adjacency: + adj_lower = adj_code.lower() + if adj_lower in seen_adj: + continue + seen_adj.add(adj_lower) + for s in range(n): + ok = _neighbor_ok(s, adj_lower) + if isinstance(ok, int) and ok == 0: + continue # provably unsatisfiable here — no var needed + sat = model.NewBoolVar(f"sat_{i}_{s}_{adj_lower}") + model.Add(sat <= x[i, s]) + model.Add(sat <= ok) + sat_vars.append(sat) + + if sat_vars: + model.Maximize(sum(sat_vars)) + + solver = cp_model.CpSolver() + solver.parameters.max_time_in_seconds = time_limit_s + solver.parameters.num_search_workers = 1 # determinism (same inputs -> same result) + status = solver.Solve(model) + if status not in (cp_model.OPTIMAL, cp_model.FEASIBLE): + return None + + result: dict[Hashable, str] = {} + for i, code in enumerate(codes): + for s in range(n): + if solver.Value(x[i, s]) == 1: + result[slots[s]] = code + break + return result diff --git a/src/homemaker_layout/driver.py b/src/homemaker_layout/driver.py index 1fc1288..e966b7d 100644 --- a/src/homemaker_layout/driver.py +++ b/src/homemaker_layout/driver.py @@ -314,6 +314,8 @@ def search( collapse_insearch: bool = True, shapecurve_warmstart: bool = False, shapecurve_prune: bool = False, + assign_solver: str = "greedy", + enable_reassign: bool = False, ) -> SearchResult: """Run the memetic loop from ``seed_root`` until ``budget`` oracle evaluations are consumed. Returns the best individual found; its ``root`` @@ -401,6 +403,19 @@ def search( a width-K beam search over which leaf a room lands on during construction, instead of one irrevocable greedy pass. ``1`` (default) reproduces the prior greedy seeding exactly. + + ``assign_solver`` (homemaker-py-2g7.5, EXPERIMENTAL, default "greedy") + forwarded to ``operators.constructive_topology``/``lift_base_to_storeys``'s + same-named parameter: ``"cpsat"`` replaces the greedy/beam room-code + placement with an exact OR-Tools CP-SAT solve (DESIGN.md §37.7), + falling through to the greedy/beam path on any solver failure. + ``"greedy"`` (default) reproduces prior seeding exactly. + + ``enable_reassign`` (homemaker-py-2g7.5, EXPERIMENTAL, default off) + un-mutes ``operators.mutate_reassign``: the CP-SAT analogue of + ``enable_ruin_recreate`` — re-solves one wing's room-code labelling + exactly instead of un-dividing and regrowing it. Gated the same way as + ``ruin_recreate`` (zero mutation weight unless enabled). """ from .oracle import DEFAULT_URB_ROOT @@ -417,6 +432,8 @@ def search( mutation_weights["bridge_circulation"] = 0.0 if not enable_ruin_recreate: mutation_weights["ruin_recreate"] = 0.0 + if not enable_reassign: + mutation_weights["reassign"] = 0.0 # homemaker-py-161: shape_rotate/deslim are gated by operators.mutate itself # (fit_ops go to zero probability when fit=None) — only build the Fitness # instance, and thus only let them fire, when explicitly enabled. @@ -613,7 +630,7 @@ def search( depth_balanced=depth_balanced, interior_outside=interior_outside, outside_divisor=outside_divisor, construction_beam_width=construction_beam_width, - multi_use=multi_use) + multi_use=multi_use, assign_solver=assign_solver) return (topo, None, child_budget, {}, f"construct/{tag}") n = int(rng.integers(max(1, n_target - 1), n_target + 2)) return (random_topology(seed_root, n, rng, types), None, child_budget, @@ -1049,6 +1066,8 @@ def search_staged( interior_outside: bool = True, outside_divisor: int = 3, construction_beam_width: int = 1, + assign_solver: str = "greedy", + enable_reassign: bool = False, ) -> SearchResult: """Staged per-floor topology search (DESIGN.md §11.3, ``homemaker-py-c4c.3``). @@ -1108,7 +1127,9 @@ def search_staged( depth_balanced=depth_balanced, interior_outside=interior_outside, outside_divisor=outside_divisor, - construction_beam_width=construction_beam_width) + construction_beam_width=construction_beam_width, + assign_solver=assign_solver, + enable_reassign=enable_reassign) if types is None: types = sorted(reqs) + ["C", "O"] @@ -1150,6 +1171,8 @@ def search_staged( interior_outside=interior_outside, outside_divisor=outside_divisor, construction_beam_width=construction_beam_width, + assign_solver=assign_solver, + enable_reassign=enable_reassign, ) best_base = r1.best.root _log(f"[staged] stage 1 done: base {r1.best.fitness:.6g} " @@ -1172,7 +1195,7 @@ def search_staged( depth_balanced=depth_balanced, interior_outside=interior_outside, outside_divisor=outside_divisor, construction_beam_width=construction_beam_width, - multi_use=multi_use) + multi_use=multi_use, assign_solver=assign_solver) _log(f"[staged] stage 2: upper floors as deltas, budget {b2}, base_p {base_p}") r2 = search( @@ -1202,6 +1225,8 @@ def search_staged( interior_outside=interior_outside, outside_divisor=outside_divisor, construction_beam_width=construction_beam_width, + assign_solver=assign_solver, + enable_reassign=enable_reassign, ) # Stitch the two stages into one accounting (total evals, tagged history). diff --git a/src/homemaker_layout/operators.py b/src/homemaker_layout/operators.py index 8cb7476..543694d 100644 --- a/src/homemaker_layout/operators.py +++ b/src/homemaker_layout/operators.py @@ -914,7 +914,8 @@ def _assign_adjacency_aware(lvl: dom.Node, room_codes: list[str], reqs, interior_outside: bool = False, n_outside: int = 1, scope: "set[dom.Node] | None" = None, - beam_width: int = 1) -> None: + beam_width: int = 1, + assign_solver: str = "greedy") -> None: """Assign leaf types so rooms cluster around a connected circulation spine. s44 (DESIGN.md §11.2 follow-up): random type assignment leaves rooms stranded @@ -956,6 +957,15 @@ def _assign_adjacency_aware(lvl: dom.Node, room_codes: list[str], reqs, since circulation/outside are already fixed and shared across every beam branch), so a room whose best slot is later needed by a harder-to-place room is no longer locked in by one irrevocable greedy step. + + ``assign_solver`` (homemaker-py-2g7.5, EXPERIMENTAL, default "greedy"): + "cpsat" replaces the room-placement pass above (both the plain-greedy + and beam variants) with an exact solve of the same decision via + :func:`cpsat.solve_room_labels` — see DESIGN.md §37.7. Circulation and + outside placement (the connected-dominating-set step above) is + unaffected either way. Falls through to the greedy/beam path on any + solver failure (OR-Tools unavailable, infeasible, or timeout), so + behaviour is always defined. """ from . import geometry @@ -1051,7 +1061,24 @@ def _assign_adjacency_aware(lvl: dom.Node, room_codes: list[str], reqs, codes.sort(key=_n_secondary, reverse=True) - if beam_width <= 1: + placed = None + if assign_solver == "cpsat": + # homemaker-py-2g7.5 (DESIGN.md §37.7): exact assignment via + # OR-Tools CP-SAT, in place of this block's greedy/beam heuristics + # below. Falls through to them on any solver failure (unavailable/ + # infeasible/timeout) — always a defined outcome. + from . import cpsat + room_set = set(room_slots) + neighbors = {L: {nb for nb in _nbrs(L) if nb in room_set} for L in room_slots} + context_types = {L: {nb.type for nb in _nbrs(L) if nb.type and nb not in room_set} + for L in room_slots} + placed = cpsat.solve_room_labels(room_slots, codes, reqs, neighbors, context_types) + + if placed is not None: + for leaf, code in placed.items(): + leaf.type = code + leftover = [L for L in room_slots if L not in placed] + elif beam_width <= 1: open_slots = sorted(room_slots, key=lambda L: (L in dominated, deg.get(L, 0), -idx[L]), reverse=True) @@ -1069,7 +1096,7 @@ def _assign_adjacency_aware(lvl: dom.Node, room_codes: list[str], reqs, deg.get(L, 0), -idx[L])) best.type = code open_slots.remove(best) - placed, leftover = None, open_slots + leftover = open_slots else: placed = _beam_place_rooms(codes, room_slots, dominated, deg, idx, _nbrs, reqs, beam_width) @@ -1132,6 +1159,43 @@ def _beam_place_rooms(codes: list[str], slots: list, dominated: set, return max(beam, key=lambda c: c[0])[1] if beam else {} +def _cpsat_relabel_settled(lvl: dom.Node, reqs) -> None: + """Re-solve room-code labelling against the storey's SETTLED geometry + (homemaker-py-2g7.5, DESIGN.md §37.7's "alternating minimization" + follow-up, §11.2's lesson applied at seed time). + + ``assign_solver="cpsat"``'s exact solve is measured (A/B, harbor-house) + to be MORE fragile than the greedy/beam heuristic to the proportion- + aware target-size resizing that runs right after assignment + (``_size_divisions_from_targets``): resizing can shrink a shared-wall + segment below the door-width adjacency threshold, silently invalidating + an edge the exact solve specifically relied on (it packs adjacency + satisfaction tightly against the pre-resize graph, leaving less slack + than the greedy path's more conservative, degree-biased placement). + Re-running the exact solve once more against the now-settled geometry + recovers this — cheap (same small model) and never worse (it can only + IMPROVE satisfied-adjacency count from wherever resizing left it). + """ + from . import cpsat, geometry + + geometry.clear_cache() + G = geometry.leaf_graph(lvl) + room_slots = [lf for lf in lvl.leaves() if lf.type in reqs] + if not room_slots: + return + codes = [lf.type for lf in room_slots] + room_set = set(room_slots) + neighbors = {L: {nb for nb in G.neighbors(L) if nb in room_set} + for L in room_slots if G.has_node(L)} + context_types = {L: {nb.type for nb in G.neighbors(L) + if nb.type and nb not in room_set} + for L in room_slots if G.has_node(L)} + result = cpsat.solve_room_labels(room_slots, codes, reqs, neighbors, context_types) + if result: + for leaf, code in result.items(): + leaf.type = code + + def constructive_topology(seed_root: dom.Node, reqs, rng: np.random.Generator, types: list[str], min_storeys: int = 1, adjacency_aware: bool = True, @@ -1143,7 +1207,8 @@ def constructive_topology(seed_root: dom.Node, reqs, rng: np.random.Generator, interior_outside: bool = True, outside_divisor: int = 3, construction_beam_width: int = 1, - multi_use: bool = False) -> dom.Node: + multi_use: bool = False, + assign_solver: str = "greedy") -> dom.Node: """Build a seed that instantiates every required space by construction. The §11.0 diagnosis: random divide+retype chains leave required programme @@ -1159,6 +1224,11 @@ def constructive_topology(seed_root: dom.Node, reqs, rng: np.random.Generator, forwarded to ``_assign_adjacency_aware``'s ``beam_width`` — see there. ``1`` reproduces the prior greedy room placement exactly. + ``assign_solver`` (homemaker-py-2g7.5, EXPERIMENTAL, default "greedy"): + forwarded to ``_assign_adjacency_aware``'s same-named parameter — see + there. ``"greedy"`` reproduces the prior placement exactly regardless + of ``construction_beam_width``. + Returns a finalised deep copy; ``seed_root`` is unchanged. """ from . import genome as _g @@ -1223,7 +1293,8 @@ def constructive_topology(seed_root: dom.Node, reqs, rng: np.random.Generator, dom.link(child) _assign_adjacency_aware(lvl, rooms, reqs, rng, interior_outside=interior_outside, n_outside=n_o, - beam_width=construction_beam_width) + beam_width=construction_beam_width, + assign_solver=assign_solver) else: assign = rooms + ["C", "O"] # +core circulation, +outside _grow_leaves(lvl, len(assign), rng, balance=depth_balanced) @@ -1245,6 +1316,8 @@ def constructive_topology(seed_root: dom.Node, reqs, rng: np.random.Generator, _size_divisions_from_targets( lvl, reqs, leaf_mult=_leaf_mult_from_plan(lvl, share_plan), leaf_extra=leaf_extra) + if adjacency_aware and assign_solver == "cpsat": + _cpsat_relabel_settled(lvl, reqs) return _finalise(child) @@ -1260,7 +1333,8 @@ def lift_base_to_storeys(base_root: dom.Node, upper_buckets: list[dict[str, int] interior_outside: bool = True, outside_divisor: int = 3, construction_beam_width: int = 1, - multi_use: bool = False) -> dom.Node: + multi_use: bool = False, + assign_solver: str = "greedy") -> dom.Node: """Stack upper storeys onto an evolved single-storey base (DESIGN.md §11.3). Stage 2 seeder: the Stage-1 base is the credible ground floor and is left @@ -1272,6 +1346,10 @@ def lift_base_to_storeys(base_root: dom.Node, upper_buckets: list[dict[str, int] splits/assignment keep a bootstrap batch diverse; ``mutate_place_missing`` repairs any residual gaps during the loop. + ``assign_solver`` (homemaker-py-2g7.5, EXPERIMENTAL, default "greedy"): + forwarded to ``_assign_adjacency_aware``'s same-named parameter — see + there. + Returns a finalised deep copy; ``base_root`` is unchanged. """ from . import genome as _g, geometry as _geo @@ -1336,7 +1414,8 @@ def lift_base_to_storeys(base_root: dom.Node, upper_buckets: list[dict[str, int] dup, rooms, reqs, rng, fixed_circ=[core_node] if core_node is not None else None, interior_outside=interior_outside, n_outside=n_o, - beam_width=construction_beam_width) + beam_width=construction_beam_width, + assign_solver=assign_solver) else: assign = rooms + ["O"] # courtyard / outside on the upper floor if core_node is None: @@ -1375,6 +1454,8 @@ def lift_base_to_storeys(base_root: dom.Node, upper_buckets: list[dict[str, int] _size_divisions_from_targets( dup, reqs, leaf_mult=_leaf_mult_from_plan(dup, share_plan), leaf_extra=leaf_extra) + if adjacency_aware and assign_solver == "cpsat": + _cpsat_relabel_settled(dup, reqs) prev = dup @@ -1453,6 +1534,65 @@ def mutate_ruin_recreate(root: dom.Node, rng: np.random.Generator, f"({len(rooms)} rooms, {len(border_circ)} anchors)") +def mutate_reassign(root: dom.Node, rng: np.random.Generator, + types: list[str], reqs=None) -> tuple[dom.Node, str]: + """CP-SAT re-labelling of one wing's room codes (homemaker-py-2g7.5). + + The "assignment analogue of ruin_recreate" (§23/DESIGN.md §37.7): picks + the same kind of wing ``mutate_ruin_recreate`` does (a divided, + live-cut subtree of one storey holding a genuine partial neighbourhood, + at least 2 leaves, at most half the storey), but does NOT un-divide or + regrow it — the topology and the wing's room-code multiset are both + preserved exactly. Only which leaf gets which code is re-decided, via + an exact :func:`cpsat.solve_room_labels` solve over the wing's room + leaves (circulation/outside leaves inside or bordering the wing stay + fixed, contributing as context for adjacency-to-``c`` credit). Where + ``mutate_ruin_recreate`` repairs a badly-grown wing structurally, this + move repairs a badly-LABELLED wing whose leaf structure is already + fine — the seeder's own one-shot greedy assignment can be beaten by + revisiting it once the wing's geometry (and therefore its adjacency + graph) has settled during search, not just once at construction time. + + No ``reqs``, no eligible wing, no room leaves in the chosen wing, or + solver failure (unavailable/infeasible/no improvement) all no-op. + """ + if not reqs: + return _finalise(copy.deepcopy(root)), "reassign noop" + from . import cpsat, geometry + + child = copy.deepcopy(root) + _finalise(child) + lvls = dom.levels(child) + totals = {li: len(lvl.leaves()) for li, lvl in enumerate(lvls)} + cands = [(li, n) for li, n in _owned_branches(child) + if totals[li] >= 4 and 2 <= len(n.leaves()) <= max(2, totals[li] // 2)] + if not cands: + return _finalise(child), "reassign noop" + li, wing = _pick(rng, cands) + lvl = lvls[li] + + room_slots = [lf for lf in wing.leaves() if lf.type in reqs] + if not room_slots: + return _finalise(child), "reassign noop" + codes = [lf.type for lf in room_slots] + + G = geometry.leaf_graph(lvl) + room_set = set(room_slots) + neighbors = {L: {nb for nb in G.neighbors(L) if nb in room_set} + for L in room_slots if G.has_node(L)} + context_types = {L: {nb.type for nb in G.neighbors(L) + if nb.type and nb not in room_set} + for L in room_slots if G.has_node(L)} + result = cpsat.solve_room_labels(room_slots, codes, reqs, neighbors, context_types) + if not result or all(leaf.type == code for leaf, code in result.items()): + return _finalise(child), "reassign noop" + + for leaf, code in result.items(): + leaf.type = code + + return _finalise(child), f"reassign {li}/{wing.id or 'root'} ({len(room_slots)} rooms)" + + def mutate_reassociate(root: dom.Node, rng: np.random.Generator, types: list[str]) -> tuple[dom.Node, str]: """Wong-Liu M3 associativity move: ``(a|b)|c <-> a|(b|c)`` on parallel cuts. @@ -1676,6 +1816,7 @@ MUTATIONS = { "shape_rotate": mutate_shape_rotate, "deslim": mutate_deslim, "ruin_recreate": mutate_ruin_recreate, + "reassign": mutate_reassign, } @@ -1693,7 +1834,8 @@ def mutate(root: dom.Node, rng: np.random.Generator, types: list[str], names = sorted(MUTATIONS) p = np.array([(weights or {}).get(n, 1.0) for n in names], dtype=float) # these operators need programme reqs; disable them when not available - reqs_ops = ("level_fix", "level_compound_fix", "place_missing", "ruin_recreate") + reqs_ops = ("level_fix", "level_compound_fix", "place_missing", "ruin_recreate", + "reassign") # also takes reqs (to avoid displacing a required room) but works without # it — never zero-weighted, unlike reqs_ops above reqs_optional_ops = ("bridge_circulation",) diff --git a/tests/test_cpsat.py b/tests/test_cpsat.py new file mode 100644 index 0000000..4c3aa24 --- /dev/null +++ b/tests/test_cpsat.py @@ -0,0 +1,103 @@ +"""Tests for the exact CP-SAT room-code labelling solver (homemaker-py-2g7.5). + +``cpsat.solve_room_labels`` is dom/geometry-independent (same decoupled- +testability convention as ``operators._beam_place_rooms``, exercised in +``test_operators.py::test_beam_place_rooms_is_deterministic_given_inputs``), +so these tests use plain hashable keys except where a direct comparison +against the existing beam/greedy heuristic requires real ``dom.Node`` +objects (``_beam_place_rooms`` reads a neighbour's ``.type`` attribute for +already-fixed context). +""" + +from homemaker_layout import cpsat, dom, operators + + +class _Req: + def __init__(self, adjacency): + self.adjacency = adjacency + + +def test_empty_inputs_return_empty_dict(): + assert cpsat.solve_room_labels([], [], {}, {}, {}) == {} + assert cpsat.solve_room_labels(["s1"], [], {}, {}, {}) == {} + assert cpsat.solve_room_labels([], ["a"], {}, {}, {}) == {} + + +def test_determinism(): + slots = ["s1", "s2", "s3"] + codes = ["a", "b", "c"] + reqs = {"a": _Req(["b"]), "b": _Req(["a"]), "c": _Req([])} + neighbors = {"s1": {"s2"}, "s2": {"s1", "s3"}, "s3": {"s2"}} + r1 = cpsat.solve_room_labels(slots, codes, reqs, neighbors, {}) + r2 = cpsat.solve_room_labels(slots, codes, reqs, neighbors, {}) + assert r1 == r2 + + +def test_fixed_context_credits_adjacency_without_a_decision_neighbour(): + # a single slot with no room-slot neighbours at all, but a fixed + # (already-typed) circulation neighbour "c" — the requirement must be + # creditable purely from context_types, no decision variable involved. + reqs = {"k1": _Req(["c"])} + result = cpsat.solve_room_labels( + ["s1"], ["k1"], reqs, {"s1": set()}, {"s1": {"c"}}) + assert result == {"s1": "k1"} + + +def test_drops_least_constrained_code_when_over_capacity(): + # more codes than slots: the code with a real adjacency requirement is + # kept over the unconstrained one, same priority the greedy path's + # hardest-first ordering uses (_n_secondary). + reqs = {"a": _Req(["b"]), "b": _Req([])} + result = cpsat.solve_room_labels(["s1"], ["b", "a"], reqs, {"s1": set()}, {}) + assert result == {"s1": "a"} + + +def test_finds_globally_optimal_labelling_beam_search_misses(): + """Hand-built counter-example (same "adversarial hand-built graph" + convention as test_collapse_global.py's + test_two_opt_polish_escapes_jacobi_plateau): a hub H (already typed + "r") connects to four leaves L1-L4; L1-L2 also has its own direct + edge. "s" and "t" each need only a "r" neighbour — satisfiable from + ANY leaf, since every leaf touches the hub. "p" and "q" need EACH + OTHER as a neighbour — only satisfiable via the one non-hub edge, + L1-L2. + + All four codes tie at exactly one secondary-adjacency requirement, so + the beam/greedy heuristic (``operators._beam_place_rooms``, + beam_width=1 reproduces the plain greedy pass) processes them in + whatever order the caller's shuffle produced. Given the order + s, t, p, q, the degree/id tie-break greedily claims the special L1-L2 + edge for s and t (who don't need it — they're satisfiable everywhere), + stranding p and q on L3/L4 with no edge between them: 2 of their 4 + combined requirements met. CP-SAT reasons globally and finds the + assignment that satisfies all 4/4, regardless of processing order. + """ + H = dom.Node(type="r") + L1, L2, L3, L4 = (dom.Node(type=None) for _ in range(4)) + slots = [L1, L2, L3, L4] + nbrs = {H: {L1, L2, L3, L4}, L1: {H, L2}, L2: {H, L1}, L3: {H}, L4: {H}} + deg = {n: len(ns) for n, ns in nbrs.items()} + idx = {L1: 0, L2: 1, L3: 2, L4: 3} + dominated = set(slots) + reqs = {"r": _Req(["s", "t"]), "s": _Req(["r"]), "t": _Req(["r"]), + "p": _Req(["q"]), "q": _Req(["p"])} + codes = ["s", "t", "p", "q"] + + placed = operators._beam_place_rooms( + codes, slots, dominated, deg, idx, lambda s: nbrs[s], reqs, + beam_width=1) + for leaf, code in placed.items(): + leaf.type = code + p_leaf = next(leaf for leaf, code in placed.items() if code == "p") + q_leaf = next(leaf for leaf, code in placed.items() if code == "q") + assert q_leaf not in nbrs[p_leaf], ( + "expected the greedy heuristic to strand p/q apart in this setup") + + neighbors_among_slots = {L1: {L2}, L2: {L1}, L3: set(), L4: set()} + context_types = {s: {"r"} for s in slots} # every leaf touches the hub + result = cpsat.solve_room_labels( + slots, codes, reqs, neighbors_among_slots, context_types) + p_slot = next(s for s, c in result.items() if c == "p") + q_slot = next(s for s, c in result.items() if c == "q") + assert q_slot in neighbors_among_slots[p_slot], ( + "CP-SAT should place p/q on the one edge that satisfies both") diff --git a/tests/test_driver.py b/tests/test_driver.py index f0f2b22..a49b49f 100644 --- a/tests/test_driver.py +++ b/tests/test_driver.py @@ -388,6 +388,36 @@ def test_shapecurve_prune_off_matches_baseline(fake_inner): assert off.n_evals == base.n_evals +def test_assign_solver_default_matches_greedy(fake_inner): + """homemaker-py-2g7.5: with assign_solver left at its default ("greedy"), + the run is identical to one that passes it explicitly — the same clean + A/B control as shapecurve_prune's.""" + init_root = dom.load(str(INIT_FILE)) + base = driver.search(init_root, CORPUS, budget=600, pop_size=4, + child_budget=60, seed_budget=100, seed=9) + explicit = driver.search(init_root, CORPUS, budget=600, pop_size=4, + child_budget=60, seed_budget=100, seed=9, + assign_solver="greedy") + assert explicit.best.sig == base.best.sig + assert explicit.n_topologies == base.n_topologies + assert explicit.n_evals == base.n_evals + + +def test_enable_reassign_default_off_matches_baseline(fake_inner): + """homemaker-py-2g7.5: with enable_reassign left at its default (off), + the run is identical to one that passes it explicitly False — the + reassign operator never fires (zero mutation weight).""" + init_root = dom.load(str(INIT_FILE)) + base = driver.search(init_root, CORPUS, budget=600, pop_size=4, + child_budget=60, seed_budget=100, seed=9) + off = driver.search(init_root, CORPUS, budget=600, pop_size=4, + child_budget=60, seed_budget=100, seed=9, + enable_reassign=False) + assert off.best.sig == base.best.sig + assert off.n_topologies == base.n_topologies + assert off.n_evals == base.n_evals + + def test_shapecurve_prune_vetoes_heuristic_when_dp_feasible(monkeypatch): """homemaker-py-wkh (DESIGN.md §37.5): a DP-feasible verdict is a real certificate that some ratio point clears every leaf's shape threshold, so diff --git a/tests/test_operators.py b/tests/test_operators.py index 9ff3092..7f3c2a3 100644 --- a/tests/test_operators.py +++ b/tests/test_operators.py @@ -494,6 +494,97 @@ def test_beam_place_rooms_is_deterministic_given_inputs(): assert r1 == r2 +@pytest.mark.skipif(not HARBOR.is_dir(), reason="harbor-house not available") +def test_construction_assign_cpsat_yields_valid_seed(): + # homemaker-py-2g7.5: the CP-SAT room-labelling path must satisfy the same + # construction invariants as the greedy path — every required space + # present, canonical genome. + from homemaker_layout import graph, programme + + reqs = programme.load_programme_dir(str(HARBOR)) + types = sorted(reqs) + ["C", "O"] + seed = dom.load(str(HARBOR / "init.dom")) + for trial in range(5): + root = operators.constructive_topology( + seed, reqs, np.random.default_rng(trial), types, + assign_solver="cpsat") + _, missing = graph.check_space_counts(root, reqs) + assert missing == [], f"trial {trial} left {missing}" + canonical(root) + + +@pytest.mark.skipif(not HARBOR.is_dir(), reason="harbor-house not available") +def test_assign_cpsat_matches_or_beats_greedy_secondary_adjacency(): + # homemaker-py-2g7.5: CP-SAT solves the same room-labelling decision the + # greedy/beam heuristic approximates exactly — its secondary-adjacency + # (not the access/adjacent-to-c fails the dominating-set step already + # solves) fail count must be strictly lower in aggregate. + import copy + + from homemaker_layout import fitness, programme + + reqs = programme.load_programme_dir(str(HARBOR)) + conf, cost = fitness.load_config(str(HARBOR)) + fit = fitness.Fitness(conf, cost) + types = sorted(reqs) + ["C", "O"] + seed = dom.load(str(HARBOR / "init.dom")) + + def secondary_fails(solver: str) -> list[int]: + counts = [] + for trial in range(10): + root = operators.constructive_topology( + seed, reqs, np.random.default_rng(trial), types, + assign_solver=solver) + _, fails = fit.score_with_fails(copy.deepcopy(root)) + counts.append(sum(1 for f in fails if "not adjacent to" in f)) + return counts + + # Per-seed outcomes are noisy (both solvers depend on the same random + # room-order shuffle before falling into their own placement logic), so + # the comparison is on the aggregate over several seeds, not every seed + # individually — measured on harbor-house (10 seeds): cpsat wins on + # most, ties on a few, loses on rare ones, net ~13% fewer total fails. + greedy = secondary_fails("greedy") + cpsat = secondary_fails("cpsat") + assert sum(cpsat) < sum(greedy) + + +def test_reassign_noop_without_reqs(): + root = genome.decode(genome.encode(dom.load(str(CORPUS / FILES[0])))) + child, desc = operators.mutate_reassign(root, np.random.default_rng(0), TYPES) + assert desc == "reassign noop" + canonical(child) + + +@pytest.mark.skipif(not HARBOR.is_dir(), reason="harbor-house not available") +def test_reassign_fires_and_preserves_room_multiset(): + # homemaker-py-2g7.5: the reassign operator must fire (find at least one + # wing to re-label) on a real seeded design, and it must preserve the + # wing's exact room-code multiset — only the leaf<->code labelling + # changes, never the topology or which codes are present. + from collections import Counter + + from homemaker_layout import programme + + reqs = programme.load_programme_dir(str(HARBOR)) + types = sorted(reqs) + ["C", "O"] + seed = dom.load(str(HARBOR / "init.dom")) + root = operators.constructive_topology( + seed, reqs, np.random.default_rng(0), types) + before = Counter(lf.type for lf in root.leaves()) + + fired = False + for trial in range(20): + child, desc = operators.mutate_reassign( + root, np.random.default_rng(trial), types, reqs=reqs) + canonical(child) + after = Counter(lf.type for lf in child.leaves()) + assert after == before, f"trial {trial}: room multiset changed ({desc})" + if not desc.endswith("noop"): + fired = True + assert fired, "reassign never fired across 20 trials on a real seeded design" + + @pytest.mark.skipif(not HARBOR.is_dir(), reason="harbor-house not available") def test_place_missing_repairs_deficient_tree(): # §11.2 repair: iterating mutate_place_missing drives a deficient design's