fix(genetic): keep fitness-cache memory within pymalloc and release arena after each run (#1353)

* fix(genetic): keep fitness-cache memory in pymalloc and release arena after each run

At fine time resolution (interval_sec=900, ~192 control slots over a multi-day
horizon) the fitness-cache keys are ~1.6 KB int tuples, above CPython's 512-byte
pymalloc threshold, so they are served by glibc malloc in the optimization worker
thread's arena and are not returned to the OS on `self._fitness_cache.clear()`.
With re-optimization every 15 min, RSS stair-steps up to the memory limit within
about a day (OOM / forced restart). At hourly resolution the tuples stay < 512 B,
so pymalloc reclaims them and the effect is negligible. See #1352.

- Store the cache key/genome compactly as bytes (1 byte per gene, 8-byte fallback
  for larger state spaces) so entries stay within pymalloc regardless of resolution.
- After each optimization run, gc.collect() + malloc_trim(0) (guarded, glibc-only)
  to return freed arena pages to the OS.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>

* fix(genetic): pack fitness-cache genome as bounded chunks

Storing the genome as a single bytes object still exceeds pymalloc's
512-byte threshold once the genome grows: at 15-min resolution over a
60 h horizon with EV genes the key is ~480 genes, so even the one-byte
encoding is 481 bytes (514 with the object header) and any value >255
switches the whole genome to 8 bytes per gene. Such keys land in glibc
malloc, which does not reliably return the pages (malloc_trim is
glibc-only, absent on musl) — the platform-independent guarantee did
not actually hold for supported settings.

Encode the genome (key and FitnessCacheEntry.genome) as a tuple of
bounded byte chunks instead — 256 one-byte genes or 32 signed-64-bit
genes per chunk, 256 bytes each — built per chunk so the encoder never
materialises an oversized temporary. Every object then stays inside
pymalloc regardless of horizon, on every platform. malloc_trim after a
run is kept as a secondary release for the rest of the run's heap.

Tests: round-trips incl. 480-gene narrow/wide and negative genes, and
sys.getsizeof for the key AND every chunk at 480 genes in both
encodings.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>

---------

Co-authored-by: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
Christin
2026-09-26 12:39:29 +02:00
committed by GitHub
co-authored by Claude Opus 4.8
parent 986b4eb0d4
commit efab8cd0a0
2 changed files with 138 additions and 13 deletions
@@ -1,8 +1,13 @@
"""Genetic algorithm."""
import ctypes
import ctypes.util
import gc
import math
import random
import sys
import time
from array import array
from collections import defaultdict
from dataclasses import dataclass, field
from functools import lru_cache
@@ -92,11 +97,76 @@ class ApplianceGeneLayout:
)
# A packed genome is a tuple of bounded byte chunks: a leading tag byte
# (b"\x00" narrow / b"\x01" wide) followed by one bytes object per chunk.
PackedGenes = tuple[bytes, ...]
# Chunk widths chosen so every chunk stays well under pymalloc's 512-byte
# threshold: 256 one-byte genes and 32 signed-64-bit genes are both 256 bytes.
_NARROW_CHUNK_GENES = 256
_WIDE_CHUNK_GENES = 32
def _pack_genes(values: list[int]) -> PackedGenes:
"""Encode genes as a compact, hashable key held entirely within pymalloc.
A fitness cache holds ~100k entries per run. A single ``bytes`` object for a
long genome exceeds pymalloc's 512-byte threshold once the genome grows
(e.g. ~480 genes at 15-min resolution over a 60 h horizon with EV genes: the
one-byte encoding alone is 481 bytes, over 512 with the object header) and
then lands in glibc malloc, which does not reliably return those pages to
the OS - ``malloc_trim`` is glibc-only and absent on musl. Splitting the
genome into bounded chunks keeps every object small regardless of horizon,
so the cache stays inside pymalloc on every platform. Gene values are small
indices (states, charge rates, appliance start offsets), so one byte per
gene fits nearly always; any value outside 0-255 switches the whole genome
to 8 bytes per gene. The leading tag byte keeps both encodings apart.
"""
narrow = all(0 <= value <= 255 for value in values)
width = _NARROW_CHUNK_GENES if narrow else _WIDE_CHUNK_GENES
chunks: list[bytes] = []
for start in range(0, len(values), width):
# Slice per chunk so the encoder never materialises an oversized
# temporary bytes object for the whole genome.
window = values[start : start + width]
chunk = bytes(window) if narrow else array("q", window).tobytes()
chunks.append(chunk)
return (b"\x00" if narrow else b"\x01", *chunks)
def _unpack_genes(packed: PackedGenes) -> list[int]:
"""Decode genes encoded by ``_pack_genes``."""
if packed[0] == b"\x00":
return [value for chunk in packed[1:] for value in chunk]
return [value for chunk in packed[1:] for value in array("q", chunk)]
_LIBC: Optional[ctypes.CDLL] = None
if sys.platform.startswith("linux"):
try:
_LIBC = ctypes.CDLL(ctypes.util.find_library("c") or "libc.so.6")
_LIBC.malloc_trim # noqa: B018 - glibc only; musl has no malloc_trim
except (OSError, AttributeError):
_LIBC = None
def _release_freed_memory() -> None:
"""Hand memory freed by an optimization run back to the operating system.
glibc does not return freed heap pages of worker-thread arenas on its own,
so a long-running EOS server would keep each run's peak forever and grow
whenever a run lands in another arena.
"""
gc.collect()
if _LIBC is not None:
_LIBC.malloc_trim(0)
@dataclass(frozen=True)
class FitnessCacheEntry:
"""One canonical, successful fitness evaluation within an optimization run."""
genome: tuple[int, ...]
genome: PackedGenes # _pack_genes() of the canonical individual
fitness: tuple[float]
extra_data: tuple[float, float, float]
@@ -745,7 +815,7 @@ class GeneticOptimization(OptimizationBase):
# never shared across runs because forecasts, prices and device state may
# have changed even when the genome is identical.
self._fitness_cache_enabled = False
self._fitness_cache: dict[tuple[int, ...], FitnessCacheEntry] = {}
self._fitness_cache: dict[PackedGenes, FitnessCacheEntry] = {}
self._fitness_cache_hits = 0
self._fitness_cache_misses = 0
@@ -2392,7 +2462,7 @@ class GeneticOptimization(OptimizationBase):
def _best_unique(self, population: list[Any], count: int) -> list[Any]:
"""Return the best fitness-relevant unique candidates."""
selected: list[Any] = []
seen: set[tuple[int, ...]] = set()
seen: set[PackedGenes] = set()
for candidate in tools.selBest(population, len(population)):
key = self._fitness_key(candidate)
if key in seen:
@@ -2407,8 +2477,8 @@ class GeneticOptimization(OptimizationBase):
self,
candidates: list[Any],
selected: list[Any],
selected_keys: list[tuple[int, ...]],
best_key: tuple[int, ...],
selected_keys: list[PackedGenes],
best_key: PackedGenes,
) -> bool:
"""Carry still-protected immigrants into ``selected`` in place.
@@ -2485,7 +2555,7 @@ class GeneticOptimization(OptimizationBase):
count,
max(1, int(count * self.SELECTION_DIVERSITY_FLOOR + 0.999999)),
)
key_counts: dict[tuple[int, ...], int] = defaultdict(int)
key_counts: dict[PackedGenes, int] = defaultdict(int)
for key in selected_keys:
key_counts[key] += 1
if len(key_counts) >= target_unique:
@@ -2825,7 +2895,7 @@ class GeneticOptimization(OptimizationBase):
original_key = self._fitness_key(individual)
cached = self._fitness_cache.get(original_key)
if cached is not None:
individual[:] = cached.genome
individual[:] = _unpack_genes(cached.genome)
individual.extra_data = cached.extra_data # type: ignore[attr-defined]
self._fitness_cache_hits += 1
return cached.fitness
@@ -2842,7 +2912,7 @@ class GeneticOptimization(OptimizationBase):
canonical_key = self._fitness_key(individual)
extra_value1, extra_value2, extra_value3 = extra_data
entry = FitnessCacheEntry(
genome=tuple(int(value) for value in individual),
genome=_pack_genes([int(value) for value in individual]),
fitness=fitness,
extra_data=(
float(extra_value1),
@@ -2854,7 +2924,7 @@ class GeneticOptimization(OptimizationBase):
self._fitness_cache[canonical_key] = entry
return fitness
def _fitness_key(self, individual: list[int]) -> tuple[int, ...]:
def _fitness_key(self, individual: list[int]) -> PackedGenes:
"""Return the fitness-relevant genome, excluding elapsed control slots."""
start_slot = self._control_start_slot()
relevant = list(individual[start_slot : self.control_end_slot])
@@ -2864,7 +2934,7 @@ class GeneticOptimization(OptimizationBase):
n_appliance_genes = self.appliance_layout.n_genes
if n_appliance_genes > 0:
relevant.extend(individual[-n_appliance_genes:])
return tuple(int(value) for value in relevant)
return _pack_genes([int(value) for value in relevant])
def _ev_soc_at_deadline(self, simulation_result: dict[str, Any], start_slot: int) -> float:
"""EV state of charge the target is checked against [%].
@@ -3271,6 +3341,7 @@ class GeneticOptimization(OptimizationBase):
)
except Exception:
self._fitness_cache.clear()
_release_freed_memory()
raise
finally:
self._fitness_cache_enabled = False
@@ -3329,9 +3400,10 @@ class GeneticOptimization(OptimizationBase):
member["verluste"].append(extra_value2)
member["nebenbedingung"].append(extra_value3)
# Avoid retaining large genome tuples in a long-lived API process until
# cyclic garbage collection happens. Cache statistics above are scalar.
# Avoid retaining the cache in a long-lived API process. Cache statistics
# above are scalar.
self._fitness_cache.clear()
_release_freed_memory()
return best_solution, member
def optimize_ems(
+54 -1
View File
@@ -1,3 +1,4 @@
import sys
from types import SimpleNamespace
from unittest.mock import patch
@@ -7,7 +8,12 @@ from deap import creator, tools
from akkudoktoreos.config.config import ConfigEOS
from akkudoktoreos.core.coreabc import get_ems
from akkudoktoreos.optimization.genetic.genetic import GeneticOptimization
from akkudoktoreos.optimization.genetic.genetic import (
GeneticOptimization,
_pack_genes,
_release_freed_memory,
_unpack_genes,
)
from akkudoktoreos.utils.datetimeutil import to_datetime
@@ -56,6 +62,53 @@ def test_ev_repair_is_resimulated_before_fitness_assignment(config_eos: ConfigEO
assert individual[opt.control_slots :] == [0] * opt.control_slots
@pytest.mark.parametrize(
"genes",
[
[],
[0, 1, 255],
[3] * 192,
[0, 256, 1],
[70000, 2, 0],
[7] * 480, # 60 h at 15 min with EV genes, all one-byte
[7] * 479 + [256], # same length, one value forces the 8-byte encoding
[-1, 0, 300], # negatives also take the signed 8-byte path
],
)
def test_pack_genes_roundtrip(genes: list[int]):
assert _unpack_genes(_pack_genes(genes)) == genes
@pytest.mark.parametrize(
"genes",
[
[7] * 480, # narrow: ~60 h at 15 min with EV genes, all one-byte
[7] * 479 + [256], # wide: one value forces 8 bytes for the whole genome
],
)
def test_pack_genes_key_and_chunks_stay_below_pymalloc_limit(genes: list[int]):
# A 60 h/15 min horizon with EV genes reaches ~480 genes. The packed key and
# every chunk must stay small objects (<= 512 B), otherwise ~100k cache
# entries per run go to glibc malloc and are never given back to the OS.
packed = _pack_genes(genes)
assert sys.getsizeof(packed) <= 512
for chunk in packed:
assert sys.getsizeof(chunk) <= 512
assert _unpack_genes(packed) == genes
def test_pack_genes_encodings_do_not_collide():
small = _pack_genes([1, 0, 0, 0, 0, 0, 0, 0])
wide = _pack_genes([1, 256])
assert _pack_genes([1]) != _pack_genes([1, 0])
assert small != wide
assert _unpack_genes(wide) == [1, 256]
def test_release_freed_memory_is_safe_to_call():
_release_freed_memory()
def test_fitness_cache_restores_canonical_ev_genome(config_eos: ConfigEOS):
_configure_hourly_grid(config_eos)
opt = GeneticOptimization(fixed_seed=42)