diff --git a/.github/workflows/loadsim-measure.yml b/.github/workflows/loadsim-measure.yml
new file mode 100644
index 0000000..39a16a0
--- /dev/null
+++ b/.github/workflows/loadsim-measure.yml
@@ -0,0 +1,47 @@
+name: Loadsim measure
+
+# Throughput and memory per Cloud SQL-like profile and writer count, run by
+# hand. Shared runners are noisy: the numbers are for comparison, never a gate.
+on:
+ workflow_dispatch:
+ inputs:
+ profiles:
+ description: Profiles to measure (comma-separated)
+ default: cloudsql-micro,cloudsql-2vcpu
+
+permissions:
+ contents: read
+
+jobs:
+ measure:
+ runs-on: ubuntu-latest
+ timeout-minutes: 60
+ steps:
+ - uses: actions/checkout@v4
+
+ - name: Setup Go
+ uses: actions/setup-go@v5
+ with:
+ go-version: '1.25'
+ cache: true
+
+ - name: Measure
+ env:
+ SEEDSTORM_LOADSIM_MEASURE: ${{ github.workspace }}/loadsim-measure.json
+ SEEDSTORM_LOADSIM_PROFILES: ${{ inputs.profiles || 'cloudsql-micro,cloudsql-2vcpu' }}
+ run: cd integration && go test -v -tags "integration loadsim" -count=1 -run TestLoadsim_MeasureWriters ./... -timeout 3000s
+
+ - name: Summary
+ if: always()
+ run: |
+ if [ -f loadsim-measure.json ]; then
+ { echo "### Loadsim measure"; echo '```json'; cat loadsim-measure.json; echo '```'; } >> "$GITHUB_STEP_SUMMARY"
+ fi
+
+ - uses: actions/upload-artifact@v4
+ if: always()
+ with:
+ name: loadsim-measure
+ path: loadsim-measure.json
+ if-no-files-found: ignore
+ retention-days: 30
diff --git a/.github/workflows/pr.yml b/.github/workflows/pr.yml
index 03364e0..89ef585 100644
--- a/.github/workflows/pr.yml
+++ b/.github/workflows/pr.yml
@@ -145,6 +145,80 @@ jobs:
- name: Run integration tests
run: cd integration && go test -race -v -tags integration -count=1 ./... -timeout 1500s
+ loadsim:
+ runs-on: ubuntu-latest
+ timeout-minutes: 40
+ name: loadsim (cloud sql profiles)
+ services:
+ postgres:
+ image: postgres:17-alpine
+ env:
+ POSTGRES_USER: seedstorm
+ POSTGRES_PASSWORD: seedstorm
+ POSTGRES_DB: testdb
+ ports:
+ - 5432:5432
+ options: >-
+ --health-cmd pg_isready
+ --health-interval 5s
+ --health-timeout 5s
+ --health-retries 10
+
+ mysql:
+ image: mysql:8.4
+ env:
+ MYSQL_ROOT_PASSWORD: root
+ MYSQL_USER: seedstorm
+ MYSQL_PASSWORD: seedstorm
+ MYSQL_DATABASE: testdb
+ ports:
+ - 3306:3306
+ options: >-
+ --health-cmd "mysqladmin ping -h localhost -u root -proot"
+ --health-interval 5s
+ --health-timeout 5s
+ --health-retries 10
+
+ steps:
+ - uses: actions/checkout@v4
+
+ - name: Setup Go
+ uses: actions/setup-go@v5
+ with:
+ go-version: '1.25'
+ cache: true
+
+ # Runner size is not assumed: tests skip profiles that do not fit and say so.
+ - name: Probe runner
+ run: |
+ {
+ echo "### Runner"
+ echo '```'
+ echo "cpus: $(nproc)"
+ free -m
+ df -h / /var/lib/docker 2>/dev/null || df -h /
+ echo "cgroup: $(stat -fc %T /sys/fs/cgroup)"
+ echo "docker data: $(findmnt -no SOURCE,FSTYPE -T /var/lib/docker || true)"
+ echo '```'
+ } >> "$GITHUB_STEP_SUMMARY"
+
+ - name: Pull database images
+ run: docker pull -q postgres:17-alpine && docker pull -q mysql:8.4 && docker pull -q alpine
+
+ - name: Run resource-limited evals
+ shell: bash # pipefail: tee must not hide a failure
+ run: cd integration && go test -v -tags "integration loadsim" -count=1 -run TestLoadsim ./... -timeout 1800s 2>&1 | tee loadsim.log
+
+ - name: Summarize skips
+ if: always()
+ run: |
+ {
+ echo "### Loadsim results"
+ echo '```'
+ grep -E -- "^(--- (PASS|FAIL|SKIP)|\s+[a-z_]+_test.go:[0-9]+: (only|skip))" integration/loadsim.log || true
+ echo '```'
+ } >> "$GITHUB_STEP_SUMMARY"
+
e2e:
runs-on: ubuntu-latest
timeout-minutes: 25
diff --git a/CLAUDE.md b/CLAUDE.md
index 0751fdc..bb402ec 100644
--- a/CLAUDE.md
+++ b/CLAUDE.md
@@ -29,6 +29,9 @@ seedstorm/
│ │ ├── compare.go # compare command
│ │ ├── mirror.go # mirror command
│ │ ├── snapshot.go # snapshot command (row counts to JSON/YAML)
+│ │ ├── tune.go # tune command, --workers auto
+│ │ ├── production.go # --production / --allow-production write guard
+│ │ ├── relationships.go # --relationships scan flags, --shape-rows, achieved-shape logs
│ │ ├── progress.go # --workers flag, progress log lines, profile ignore
│ │ ├── profile.go # profile command + shared --profile flag/loading
│ │ ├── endpoints.go # --source-*/--target-* flags and connection opening
@@ -38,7 +41,12 @@ seedstorm/
│ │ ├── postgres.go # PostgreSQL schema introspection + constraint parsing
│ │ ├── mysql.go # MySQL schema introspection + constraint parsing
│ │ ├── truncate.go # Truncate helper (FK-safe order)
-│ │ ├── counts.go # GetTableRowCounts helper (used by gaps)
+│ │ ├── counts.go # CountTables / CountTablesWithin: per-table outcomes, unknown never 0
+│ │ ├── read_scope.go # ReadOnce: read-only tx, statement/lock timeouts, MySQL KILL QUERY on cancel
+│ │ ├── relations.go # DegreeHistogram (CASE buckets), LeadingIndexed, EstimateDegrees
+│ │ ├── server_info.go # DetectServer (capacity, replica, server id), ConnectionUsage
+│ │ ├── explain.go # Explain: disk full / connection lost errors in plain words
+│ │ ├── partitions.go # Postgres partitioned tables (bounds, leaves)
│ │ ├── stats.go # Table sizes, estimated counts, column lists, DB identity
│ │ ├── copy.go # CopyRows: Postgres COPY for a chunk of rows
│ │ ├── sequences.go # SyncSequences: move Postgres sequences past inserted ids
@@ -56,6 +64,9 @@ seedstorm/
│ │ ├── stream.go # Stream: generation state kept across chunks (NewStream, Generate)
│ │ ├── keys.go # keySet: exact map, then scalable Bloom filter
│ │ ├── pools.go # PK pool reservoir sampling and capping
+│ │ ├── shapes.go # Relationship shapes: degree dealer (Fenwick slots), DeriveShapedRows
+│ │ ├── references.go # Pools for FKs to non-key columns ("table.column")
+│ │ ├── partitions.go # Partition-key generators, CheckSeedable
│ │ └── *_test.go # Unit tests alongside production files
│ ├── graph/
│ │ ├── graph.go # Dependency graph (Build, TopologicalSort, RenderPlan)
@@ -63,14 +74,21 @@ seedstorm/
│ │ └── graph_test.go # Unit tests
│ ├── rules/ # Seed profile rules: model, templates, resolve/validate/compile
│ ├── profiles/ # Saved profile store (profiles.yaml) + Resolve(file|name)
-│ ├── compare/ # Snapshots, Diff, PlanMirror, snapshot files (Encode/ParseSnapshot), renderers (never writes)
+│ ├── compare/ # Snapshots, Diff, PlanMirror, snapshot files v1/v2 (Encode/ParseSnapshot), DiffShapes, renderers (never writes)
+│ ├── relations/ # Relationship shapes: Scan (index gate, per-edge timeout, cancel keeps finished), Shape
+│ ├── tuning/ # Recommend, ClampWriters, DetectHost (cgroup CPU/memory)
+│ ├── safego/ # Run/Recover: a panic becomes an error with an id
+│ ├── runerr/ # Located errors: side · phase · table
+│ ├── faultinject/ # SEEDSTORM_FAULT points (build tag faultinject only)
│ ├── dataio/ # Streaming data documents: writers (yaml/json/sql/csv), ReadTables
│ ├── seeder/ # Seed (strict chunked seed/gaps), Fill (resilient mirror inserts), MirrorJob, Preview,
│ │ # writer.go (FK-gated concurrent writes), meter.go (rate/ETA),
-│ │ # seed.go generateTables (parallel generation) + poolReleases
+│ │ # seed.go generateTables (parallel generation) + poolReleases + clampToServer,
+│ │ # relationships.go (Endpoint.Shapes, CompareShapes, MeasureShapes), servers.go (RelateServers)
│ ├── fsutil/ # WriteFileAtomic for on-disk stores
│ ├── tui/ # Bubble Tea flows (seed, gaps, generate, clone, mirror)
-│ ├── web/ # serve: handlers_*.go per area, templates/, static/ (page.js + page.css per page)
+│ ├── web/ # serve: handlers_*.go per area, templates/, static/ (page.js + page.css per page),
+│ │ # jobs.go (list, reattach, eviction), production.go, preflight.go, shapes.go (session shape cache)
│ ├── ai/ai.go # Gemini enrichment (prompt building, response parsing)
│ ├── schema/schema.go # Schema YAML types and loader
│ ├── build/info.go # Version info injected at build time
@@ -80,7 +98,9 @@ seedstorm/
│ ├── binary_test.go # Builds the binary; scratch DB helpers per engine
│ ├── *_test.go # Scenario evals: reseed, mirror, seeder, keycloak, compare_web, parallel_seed,
│ │ # wide_schema (150 generated tables), access, snapshot, clone_objects,
-│ │ # web_jobs (+helpers), reproducible, seed_scale (memory bounds)
+│ │ # web_jobs (+helpers), reproducible, seed_scale (memory bounds), read_scope, production,
+│ │ # reliability (faultinject), relationships, shaped_seed, partitions, tune, servers
+│ ├── loadsim_*_test.go # Resource-limited evals in Cloud SQL-shaped containers (tags: integration loadsim)
│ ├── fixtures/ # Keycloak schema dumps (real-world 87-table stress schema)
│ ├── schema_postgres.sql # 28-table schema for Postgres integration tests
│ └── schema_mysql.sql # 28-table schema for MySQL integration tests
@@ -88,12 +108,13 @@ seedstorm/
│ ├── playwright.config.ts # chromium, 1 worker; global setup builds the binary and starts serve
│ ├── support/ # global.setup/teardown, databases.fixture (ss_e2e_* DBs), wide-schema.fixture,
│ │ # db.helpers (SQL assertions), test.fixture, selectors.ts (every data-testid)
-│ └── tests/.spec.ts # workspace, seed, clone, compare, access, profiles, mobile
+│ └── tests/.spec.ts # workspace, seed, clone, compare, access, profiles, mobile, memory, production, tuning, snapshot, relationships
├── docs/benchmarks.md # Measured throughput, memory and layout numbers behind the defaults
├── README.md # User-facing documentation (keep in sync with code)
├── Makefile # Build, test, lint, dev-up/down targets
-├── compose.yaml # Local MySQL + PostgreSQL via Docker Compose
-└── .github/workflows/pr.yml # CI: title, review, structlint, unit tests, lint, integration (-race), e2e
+├── docker-compose.yaml # Local MySQL + PostgreSQL via Docker Compose
+└── .github/workflows/ # pr.yml: title, review, gauntlet (structlint, unit -race), lint, integration, loadsim, e2e;
+ # loadsim-measure.yml: manual throughput measurement
```
---
@@ -205,12 +226,12 @@ After standard row generation, `topUpEnumCoverage` guarantees every enum value (
| Job | What it checks |
|-----|---------------|
-| `title` | Conventional Commits format |
+| `pr-title` | Conventional Commits format |
| `review` | AI code review via reviewforge (Gemini) |
-| `validate` | Directory/file structure via structlint |
-| `test` | `go test ./...` + `make build` |
+| `gauntlet` | structlint, dupehound, unit tests with `-race` |
| `lint` | `golangci-lint` |
| `integration` | Full suite + scenario evals on Postgres 13/15/17 × MySQL 5.7/8.0/8.4, with the race detector |
+| `loadsim` | Resource-limited evals (Cloud SQL-shaped containers); outcomes only, never timings |
| `e2e` | Playwright journeys (`e2e/`) against a built binary on Postgres 15 + MySQL 8.0 |
The integration job in CI uses `-race -timeout 1500s` (the race detector roughly doubles the ~5 minute suite). Use the same locally.
@@ -279,6 +300,24 @@ The integration job in CI uses `-race -timeout 1500s` (the race detector roughly
- Don't rely on server defaults (PG < 15 `CREATE` on `public`, MySQL 5.7 CHECK/roles).
- Do give scratch databases unique `ss__*` names.
+**Shared state a run borrows**
+- Do restore a pool setting a run changes (`SetMaxOpenConns`): the web seeds on the pool its pages query.
+- Don't judge a replica by MySQL `read_only`: a user with SUPER writes anyway, only `super_read_only` blocks.
+- Do buffer every SSE event past the subscriber's channel: a fast job drops events, the handler refills from `job.Events()`.
+
+**Database truths that break guards**
+- Do round datetime keys to the second: MySQL rounds fractional seconds on insert, so adjacent values collide.
+- Don't read Postgres row statistics right after inserting: they land asynchronously, a partial count is a valid estimate.
+
+**Reads and failures**
+- Do route new reads through `db.ReadOnce` (read-only, lock timeout, real cancel); never report an unknown count as 0.
+- Do wrap goroutines in `safego.Run` and errors in `runerr.At`/`OnSide` so failures name side, phase and table.
+- Don't trust the MySQL driver to stop a cancelled query: it keeps running without `KILL QUERY`.
+
+**Relationship shapes**
+- Do keep the unshaped path free of extra random draws (`--seed` hashes must not change).
+- Do return dealt slots when a row is retried or dropped; never loop to fit a shape, adjust and warn.
+
**Web UI**
- Do check 320–1920 widths with the interaction active (search, dialogs, runs).
- Do test graph changes on `wideSchemaDDL(150)`, not only small schemas.
diff --git a/Makefile b/Makefile
index 47bcc3a..9727e7f 100644
--- a/Makefile
+++ b/Makefile
@@ -64,6 +64,13 @@ test-integration: dev-up
done
cd integration && go test -race -v -tags integration -count=1 ./... -timeout 1500s
+.PHONY: test-loadsim
+# Resource-limited evals: throwaway database containers shaped like managed
+# Cloud SQL instances (integration/loadsim_*_test.go). Needs Docker; skips when
+# the machine has too little free memory. Pass ARGS to filter, e.g. ARGS=-run=TestLoadsim_Read
+test-loadsim:
+ cd integration && go test -race -v -tags "integration loadsim" -count=1 -run TestLoadsim $(ARGS) ./... -timeout 1800s
+
.PHONY: test-e2e
# Playwright journeys against a freshly built `seedstorm serve` and the compose
# databases (make dev-up). Pass ARGS to filter, e.g. ARGS=tests/compare.spec.ts
diff --git a/README.md b/README.md
index 032a00f..506f9cd 100644
--- a/README.md
+++ b/README.md
@@ -115,6 +115,13 @@ seedstorm mirror --source-dsn "$PROD" --target-dsn "$STAGE" --scale 5 --max-rows
No live access to production at mirror time? Save its counts once with `seedstorm snapshot --out prod-counts.yaml` (or **Export counts** in the web UI) and pass `--source-snapshot prod-counts.yaml`.
+Row counts are not the whole picture: `--relationships` on `compare` and `snapshot` measures children per parent for every foreign key (min, avg, p95, max, parents with none), read-only and one key at a time on a timeout, and `mirror --shape-like-source` seeds the target with the same skew instead of an even spread:
+
+```bash
+seedstorm compare --source-dsn "$PROD" --target-dsn "$STAGE" --relationships
+seedstorm mirror --source-dsn "$PROD" --target-dsn "$STAGE" --shape-like-source
+```
+
Engines can differ (MySQL volumes onto Postgres works). Only the target is written;
the command refuses when source and target are the same database. Rows the database
rejects are regenerated, and a table that cannot be filled is reported without
@@ -166,13 +173,17 @@ live example values and sample rows. See [docs/profiles.md](docs/profiles.md).
- **Mirror volumes** — `mirror` seeds a target to a source's row counts at any scale (top-up or reset), with a reviewable plan, sample rows, FK-aware parents, and resilient inserts that report what could not be filled
- **Seed profiles** — value rules (templates with `{{auto}}`/`{{seq}}`/`{{run}}`, fixed values, lists, NULL) applied by column pattern or per column; saved from the web UI and reused with `--profile` in the CLI and TUI
- **Parallel writes with live progress** — unrelated tables (and pieces of a large table) write on `--workers` connections while generation streams in memory-bounded chunks; runs report rows written, rate and ETA (174k rows into MySQL: 41s → 14.5s). `--gen-workers` generates tables on several cores (1.79M rows/s on 8) for databases that can keep up — see [benchmarks](docs/benchmarks.md)
+- **Relationship shapes** — measure children per parent for every foreign key (exact with an index gate and per-key timeouts, or planner estimates), compare them between databases, save them in counts files, and seed with them (`relationships:` in a profile, `mirror --shape-like-source`): no parent above max, real zero and NULL shares
+- **Safe on production** — every read runs in a read-only transaction with a lock timeout and server-side cancel; unknown counts are never 0; connections marked production need an explicit confirmation to write, and relationship scans there read estimates unless confirmed
+- **Tuning advice** — `tune` (and **Recommend** in the web UI) picks writers and generators from the database's free connections, vCPU, memory and disk, and checks the rows fit; every run lowers its writers when the server has fewer free connections
+- **Failures say what failed** — errors name the side, phase and table and list what was written; a panic in one web job never stops the server or other jobs; exit codes 1 / 70 / 130
- **Reproducible** — `--seed` writes byte-identical data on every run
- **Counts snapshots** — `snapshot` saves row counts to JSON/YAML; `compare` and `mirror` accept `--source-snapshot`, and the web UI exports and imports the same files
- **Ignored tables** — a profile's `ignore:` globs keep tables out of every seed, fill, generate and mirror run
- **Re-seed safely** — seeding a populated table appends: ids, UNIQUE sequences and composite keys continue past existing rows
- **Schema clone for test DBs** — copy schema-only structure from one connected Postgres/MySQL database into another matching local target, preserving compatible table metadata before seeding it with safe fake data; `--objects all` adds views, functions/procedures and triggers
- **Interactive TUI** — wizard for table selection, global config, self-reference depth, per-table row volumes, and review before seeding
-- **Web UI** — `seedstorm serve` exposes an interactive graph workspace with search that zooms to matches, a navigator minimap for large schemas, privilege checks for the connected user, click-to-select tables, self-reference depth, per-table row overrides, truncate-only runs (`Rows = 0` + `truncate`), live SSE job logs with per-table truncate/insert progress, schema clone between connected DBs, and a multi-DB session switcher
+- **Web UI** — runs you can leave and come back to (live progress, elapsed time, reattach), remembered settings per connection, relationship badges on the graph; `seedstorm serve` exposes an interactive graph workspace with search that zooms to matches, a navigator minimap for large schemas, privilege checks for the connected user, click-to-select tables, self-reference depth, per-table row overrides, truncate-only runs (`Rows = 0` + `truncate`), live SSE job logs with per-table truncate/insert progress, schema clone between connected DBs, and a multi-DB session switcher
- **Saved connections** — connections persist on the machine and survive a restart, with test-before-connect, opt-in password storage, and driver-aware connection parameters (with JDBC-to-Go translation) for both Postgres and MySQL
- **Dry-run** — preview the seed plan and INSERT SQL without touching the database
- **Export** — generate fake data as YAML, JSON, SQL or CSV without a live connection, streamed in flat memory; SQL files load as-is
diff --git a/cmd/seedstorm/main.go b/cmd/seedstorm/main.go
index 8e40cda..e2087f2 100644
--- a/cmd/seedstorm/main.go
+++ b/cmd/seedstorm/main.go
@@ -2,18 +2,61 @@ package main
import (
"context"
+ "errors"
"fmt"
"os"
+ "os/signal"
+ "syscall"
+
+ tea "github.com/charmbracelet/bubbletea"
"github.com/AxeForging/seedstorm/internal/app"
+ "github.com/AxeForging/seedstorm/internal/safego"
_ "github.com/go-sql-driver/mysql"
_ "github.com/jackc/pgx/v5/stdlib"
)
+// Exit codes: 1 a run failed, 70 an internal error (a bug: please report it),
+// 130 interrupted.
+const (
+ exitFailed = 1
+ exitInternal = 70
+ exitInterrupted = 130
+)
+
func main() {
- if err := app.New().Run(context.Background(), os.Args); err != nil {
- fmt.Fprintf(os.Stderr, "error: %v\n", err)
- os.Exit(1)
+ os.Exit(run())
+}
+
+func run() int {
+ // The first Ctrl+C cancels the run, so reads stop on the server (MySQL
+ // queries are killed) and the summary is printed; a second one exits now.
+ ctx, stop := signal.NotifyContext(context.Background(), os.Interrupt, syscall.SIGTERM)
+ defer stop()
+ go func() {
+ <-ctx.Done()
+ stop()
+ again := make(chan os.Signal, 1)
+ signal.Notify(again, os.Interrupt, syscall.SIGTERM)
+ <-again
+ fmt.Fprintln(os.Stderr, "interrupted again: exiting now")
+ os.Exit(exitInterrupted)
+ }()
+
+ err := safego.Run("seedstorm", func() error { return app.New().Run(ctx, os.Args) })
+ switch {
+ case err == nil:
+ return 0
+ case ctx.Err() != nil:
+ fmt.Fprintf(os.Stderr, "interrupted: %v\n", err)
+ return exitInterrupted
+ }
+ var p *safego.PanicError
+ if errors.As(err, &p) || errors.Is(err, tea.ErrProgramPanic) {
+ fmt.Fprintf(os.Stderr, "error: %v\nThis is a bug in seedstorm; run with --log-level debug for the stack and please report it.\n", err)
+ return exitInternal
}
+ fmt.Fprintf(os.Stderr, "error: %v\n", err)
+ return exitFailed
}
diff --git a/docs/benchmarks.md b/docs/benchmarks.md
index 9217461..fd6879d 100644
--- a/docs/benchmarks.md
+++ b/docs/benchmarks.md
@@ -74,6 +74,49 @@ The integration tests `TestSeed_WideRowsStayUnderAMemoryBound`,
process's own `VmHWM`: rusage's max RSS also counts the test process at fork,
which under `-race` made every run report 212 MB.
+## Relationship shapes
+
+Shaped seeding keeps the streaming memory bound: a dealer holds a few bytes per
+parent (degrees plus a Fenwick tree over the remaining slots) and is freed when
+its table ends. Dealing 5M slots over 500,000 parents takes well under a
+second (`TestDealDegrees_PoolLimitIsFast`).
+
+| Run | Peak memory |
+|-----|------------:|
+| Shaped seed, 100k parents + 225k shaped children (Postgres) | **124 MB** |
+| Unshaped seed, 100k + 200k rows (bound held by `TestSeed_LargeRunsStreamWithFlatMemory`) | < 150 MB |
+
+```bash
+cd integration && go test -tags integration -count=1 -run TestShapedSeed_MemoryStaysBounded -v ./...
+```
+
+With `--seed` and no `relationships:` in the profile, output is byte-identical
+to the version before shaping existed (the 6 recorded files below).
+
+## Tuning and small instances
+
+Writers for a managed database are limited by its CPU and disk more than by
+seedstorm. Local Docker container shaped like Cloud SQL's smallest instance
+(`cloudsql-micro`: 1 vCPU, 629MB, 300 write IOPS), 3 tables × 2,000 rows,
+one run each (2026-09-17):
+
+| Engine | 1 writer | 2 writers | 4 writers | 8 writers |
+|--------|---------:|----------:|----------:|----------:|
+| MySQL 8.4 | 1,672 rows/s | **1,851 rows/s** | 1,496 rows/s | 1,483 rows/s |
+| Postgres 17 (COPY) | 50,887 rows/s | 54,240 rows/s | 54,714 rows/s | 57,910 rows/s |
+
+On MySQL two writers per vCPU is the best of the four and more writers get
+slower; that is what `tune` recommends for a small instance. The Postgres run
+ends in about 0.1s, too short for the disk limit to matter, so it does not rank
+writers. The numbers vary from run to run; read them as ratios.
+
+Reproduce (numbers go to the JSON report):
+
+```bash
+SEEDSTORM_LOADSIM_MEASURE=/tmp/measure.json SEEDSTORM_LOADSIM_PROFILES=cloudsql-micro \
+ make test-loadsim ARGS=-run=TestLoadsim_MeasureWriters
+```
+
## Reproducibility
`--seed` writes byte-identical data on every run. The generation speedups were
diff --git a/docs/commands.md b/docs/commands.md
index 9f010ae..a311ff4 100644
--- a/docs/commands.md
+++ b/docs/commands.md
@@ -13,11 +13,16 @@ Every seedstorm command, with all flags and examples.
- [`clone-schema`](#clone-schema) — copy schema structure into another DB
- [`compare`](#compare) — row counts, sizes and column drift between two DBs
- [`mirror`](#mirror) — seed a target so its volumes follow a source
-- [`snapshot`](#snapshot) — save row counts to a file for compare/mirror
+- [`snapshot`](#snapshot) — save row counts (and relationship shapes) to a file for compare/mirror
+- [`tune`](#tune) — recommend writers and generators for a database
- [`profile`](#profile) — manage seed profiles (value rules)
- [`serve`](#serve) — local web UI for every feature
- [`version`](#version) / [`completion`](#completion)
+**Exit codes:** `0` success · `1` the command failed (the message names the side, phase and table, e.g. `error: target · write · orders: …`, and a partial run lists what was written) · `70` internal error (a panic, reported with an id; the stack is in `--log-level debug`) · `130` interrupted with Ctrl+C (running queries are cancelled on the server first).
+
+**Unreachable databases** fail within 10 seconds naming the database (`app@db:5432 did not answer`), instead of waiting for the operating system's TCP timeout.
+
---
## `introspect`
@@ -46,6 +51,11 @@ SEEDSTORM_DB=postgres SEEDSTORM_DSN="postgres://..." seedstorm introspect
| `--db` / `$SEEDSTORM_DB` | `postgres` | Database type: `postgres` or `mysql` |
| `--dsn` / `$SEEDSTORM_DSN` | — | Connection string (required) |
| `--out` / `-o` | `schema.yaml` | Output file path |
+| `--relationships` | — | Also measure every foreign key's shape (read-only, exact) and write it with estimated table counts to this [snapshot](#snapshot) file |
+| `--scan-unindexed` | false | With `--relationships`: also scan keys that lead no index (full table scans); otherwise those are estimated |
+| `--read-timeout` | `60s` | With `--relationships`: server-side time limit per foreign key; a slower one is reported as `timed out` and the rest continue |
+
+Introspection logs a line per table on large schemas, so a slow catalog shows progress.
---
@@ -159,7 +169,21 @@ The interactive TUI includes a **Volumes** step after global config. Each select
| `--workers` | `4` | Connections writing at once. A table writes only after every table it references; self-referencing tables write in order (`1` = one at a time) |
| `--gen-workers` | `1` | Tables generated at once on separate cores. Only helps when the database takes rows faster than one core generates them (hundreds of thousands per second, see [benchmarks](benchmarks.md)); ignored with `--seed` so runs stay reproducible |
| `--interactive` / `-i` | false | Launch interactive TUI |
-| `--profile` / `-p` / `$SEEDSTORM_PROFILE` | — | [Seed profile](profiles.md): rules file or saved profile name; its `ignore:` tables are never written |
+| `--profile` / `-p` / `$SEEDSTORM_PROFILE` | — | [Seed profile](profiles.md): rules file or saved profile name; its `ignore:` tables are never written and its `relationships:` shape foreign keys |
+| `--shape-rows` | false | Derive the row count of each shaped child table from its parents (parents × share with children × avg); `--table-rows` still wins |
+| `--production` / `$SEEDSTORM_PRODUCTION` | false | The database is production: writes are refused unless `--allow-production` is also given (dry runs still work) |
+| `--allow-production` | false | Confirm a write to a database marked `--production` |
+
+`--workers auto` picks writers from the database's free connections (see [`tune`](#tune)). Whatever you ask for, a run never opens more connections than the server has free: it lowers the writers and says so (`Using 3 writers instead of 8: the server has 95 of 100 connections in use`).
+
+With a profile that has `relationships:`, foreign keys follow those shapes instead of picking parents evenly: each parent gets a number of children drawn from the histogram, no parent exceeds `max` (exact while the parent table fits the 500,000-row key pool), the share of parents without children and of NULL keys is kept, and the run logs the achieved shape next to the target afterwards:
+
+```
+info Rows derived from relationship shapes rows=14000 table=orders
+info Relationship shape (target → table now) avg="4.00 → 4.00" max="25 → 25" relationship=orders.account_id without_children="30% → 30%"
+```
+
+Shapes that cannot fit the planned rows are adjusted with a warning instead of looping (`14000 rows over 500 parents do not fit max 25: max raised to 28`). Self-references and junction keys are not shaped (reported). A parent table above 500,000 rows is kept as a sample: children are dealt a round at a time over each sample (the run says so), and seeding against a live database rotates the sample so the whole table gets its share. The shape is then close rather than exact — measured on 600,000 parents and 1.6M children: average 3 → 3.4, max 8 → 10, parents without children 10% → 22%.
Any `--rows` is safe: rows are generated and written 20,000 at a time, Postgres takes each chunk through `COPY`, and memory stays flat (600k rows on Postgres: 7s, under 100MB). A dry run prints the SQL the same way.
@@ -248,6 +272,7 @@ Gap Analysis
| `--gen-workers` | `1` | Tables generated at once on separate cores. Only helps when the database takes rows faster than one core generates them (hundreds of thousands per second, see [benchmarks](benchmarks.md)); ignored with `--seed` so runs stay reproducible |
| `--interactive` / `-i` | false | Launch interactive TUI |
| `--profile` / `-p` / `$SEEDSTORM_PROFILE` | — | [Seed profile](profiles.md): rules file or saved profile name; its `ignore:` tables are never filled |
+| `--production` / `--allow-production` | false | As for [`seed`](#seed) |
---
@@ -354,6 +379,7 @@ By default only tables, foreign keys, indexes and comments are cloned. `--views`
| `--objects` | — | Comma-separated kinds to clone: `views`, `routines`, `triggers`, or `all` |
| `--dry-run` / `-n` | false | Print generated DDL, do not execute |
| `--interactive` / `-i` | false | Confirm the clone in the terminal UI |
+| `--production` / `--allow-production` | false | As for [`seed`](#seed): marks the target as production |
Boundaries: `clone-schema` is same-engine only. It does not attempt cross-engine translation, and it does not clone partial/expression indexes, grants, ownership, events, or non-public/non-current schemas.
@@ -376,6 +402,9 @@ seedstorm compare --source-dsn "$PROD" --target-dsn "$STAGE" --format json
# Source from a counts file instead of a connection (see snapshot)
seedstorm compare --source-snapshot prod-counts.yaml --target-dsn "$STAGE"
+
+# Also compare children per parent for every foreign key
+seedstorm compare --source-dsn "$PROD" --target-dsn "$STAGE" --relationships
```
Sample output:
@@ -392,6 +421,18 @@ employees 160 80 -80 72.0 KB 64.0 KB
rows 3856 → 1888 · size 1.7 MB → 1.3 MB
```
+With `--relationships`:
+
+```
+Relationships (children per parent: avg / p95 / max · parents without children)
+RELATIONSHIP SOURCE TARGET STATUS
+orders.user_id → users 5.15 / 15 / 34 · 19% 2.00 / 2 / 2 · 0% differs
+reviews.order_id → orders ~1.00 / ? / ? · 70% 0.00 / 0 / 0 · 100% differs
+2 relationships · same 0 · differs 2 · source only 0 · target only 0 · unknown 0
+```
+
+`~` marks estimates (planner statistics). A relationship that timed out, was locked or failed shows its outcome instead of numbers and counts as `unknown`.
+
| Flag | Default | Description |
|------|---------|-------------|
| `--source-db` / `$SEEDSTORM_SOURCE_DB` | `postgres` | Source database type |
@@ -401,7 +442,12 @@ rows 3856 → 1888 · size 1.7 MB → 1.3 MB
| `--target-dsn` / `$SEEDSTORM_TARGET_DSN` | — | Target connection string (required) |
| `--counts` | `exact` | `exact` (COUNT(*)) or `estimate` (Postgres `reltuples`, MySQL `TABLE_ROWS`); tables with no or zero statistics are counted exactly and estimated counts are marked `~` |
| `--format` / `-f` | `table` | `table` or `json` |
-| `--only-diff` | false | Hide tables whose counts and columns match |
+| `--only-diff` | false | Hide tables (and relationships) that match |
+| `--relationships` | false | Also compare every foreign key's shape on both sides (a `--source-snapshot` must include relationships) |
+| `--scan-unindexed` | false | With `--relationships`: also scan keys that lead no index |
+| `--read-timeout` | `60s` | With `--relationships`: server-side time limit per foreign key |
+
+All reads are safe on a busy database: each runs in a read-only transaction, waits at most 2 seconds for a lock (a table locked by DDL is reported `locked`, not waited on), and is cancelled on the server when you press Ctrl+C. A count that could not be read is `unknown` (`?`), never 0. Both sides are read at the same time; a side that fails is named (`target · count: …`).
Statuses: `same`, `differs`, `source_only`, `target_only`. MySQL sizes come from cached statistics and are approximate.
@@ -432,6 +478,10 @@ seedstorm mirror --source-db mysql --source-dsn "$MYSQL_PROD" \
# Review, preview samples and confirm in the terminal UI
seedstorm mirror --source-dsn "$PROD" --target-dsn "$STAGE" --interactive
+
+# Give the target production's children per parent, not an even spread
+seedstorm mirror --source-dsn "$PROD" --target-dsn "$STAGE" --shape-like-source
+seedstorm mirror --source-snapshot prod-with-relationships.yaml --target-dsn "$STAGE" --shape-like-source
```
Sample dry run:
@@ -460,6 +510,8 @@ How the plan is built:
- Tables missing on the target, or whose source count is unknown, are skipped with a reason. Per-table `rows` in a profile do not apply: volumes come from the source.
- The run refuses when source and target are the same database, however the DSNs are spelled. With `--source-snapshot` that check cannot run, and seedstorm warns about it: double-check `--target-dsn`.
- Tables a profile lists under `ignore:` are skipped (`ignored by profile`) and never truncated; a table that needs an ignored, empty parent is skipped with the reason.
+- A target that refuses every write is refused before anything is planned: a Postgres standby, or MySQL with `super_read_only`. MySQL `read_only` alone only warns, since a user with `SUPER` can still write. When source and target are databases on the same server, the plan says so: reading one and writing the other share its CPU, memory and disk.
+- `--shape-like-source` shapes the target's foreign keys like the source's (see [`seed`](#seed)); the shapes come from the snapshot file's relationships, or are measured read-only on the source first. Keys this schema cannot shape (junction keys, key columns, self-references) are named with their reason and spread evenly — on Keycloak's 87 tables, 29 of 67 measured keys are shaped and the other 38 are reported.
Large volumes are safe to run: rows are generated and written 20,000 at a time, Postgres takes each chunk through `COPY`, and memory stays flat whatever the table size (a 7.2M-row mirror peaks around 150MB). Existing keys are read once per table; tables with more than 500,000 parents reference a rotating sample of them, so children still spread over the whole parent table. After the run, Postgres sequences behind SERIAL/IDENTITY columns are moved past the inserted ids, so the application's own inserts keep working (`advanced orders.id sequence 0 → 2000000`).
@@ -496,6 +548,11 @@ inserted 384 rows · missing 6
| `--yes` / `-y` | false | Skip the `reset` confirmation |
| `--seed` | `0` | Random seed for reproducible generation |
| `--interactive` / `-i` | false | Review, preview and confirm in the TUI |
+| `--shape-like-source` | false | Shape target foreign keys like the source's |
+| `--scan-unindexed` / `--read-timeout` | false / `60s` | With `--shape-like-source` on a live source, as for `compare --relationships` |
+| `--production` / `--allow-production` | false | As for [`seed`](#seed): marks the target as production |
+
+`--workers auto` is accepted as for `seed`.
---
@@ -508,6 +565,9 @@ seedstorm snapshot --db postgres --dsn "$PROD_DSN" --out prod-counts.yaml
seedstorm snapshot --db mysql --dsn "$DSN" --counts estimate --format json > counts.json
seedstorm compare --source-snapshot prod-counts.yaml --target-dsn "$STAGE_DSN"
seedstorm mirror --source-snapshot prod-counts.yaml --target-dsn "$STAGE_DSN" --scale 0.1
+
+# Counts plus every foreign key's shape (writes version 2)
+seedstorm snapshot --dsn "$PROD_DSN" --relationships --out prod-with-relationships.yaml
```
```yaml
@@ -535,6 +595,8 @@ tables:
orders: 5000
```
+With `--relationships` the file is `version: 2` and adds a `relationships:` list: per foreign key the parent and child counts, NULL keys, share of parents without children, min / avg / p50 / p95 / max children per parent and a histogram (exact buckets 1–16, then powers of two). Relationship scans run read-only with a 60-second limit per key, two at a time, cheapest tables first; keys that lead no index are estimated unless `--scan-unindexed` is given, and a key that times out is recorded as `timed out` while the others continue. A file without relationships stays `version: 1`, readable by older seedstorm versions.
+
Tables match the target case-insensitively, so a Postgres snapshot works against MySQL. `-1` means unknown, and mirror skips that table. The web UI's **Compare** page exports either side of a comparison in the same format and imports files or pasted text as a source.
| Flag | Default | Description |
@@ -544,6 +606,43 @@ Tables match the target case-insensitively, so a Postgres snapshot works against
| `--counts` | `exact` | `exact` (COUNT(*)) or `estimate` (planner statistics) |
| `--format` / `-f` | `yaml` | `yaml` or `json` |
| `--out` / `-o` | stdout | File to write |
+| `--relationships` | false | Also measure every foreign key's shape (read-only; exact or estimate per `--counts`) |
+| `--scan-unindexed` | false | With `--relationships`: also scan keys that lead no index |
+| `--read-timeout` | `60s` | With `--relationships`: server-side time limit per foreign key |
+
+---
+
+## `tune`
+
+Recommends `--workers` and `--gen-workers` for one database from what it reports (free connections, buffers, used space; read-only) and what SQL cannot tell (vCPU, memory, disk), and checks that the rows fit on the disk.
+
+```bash
+seedstorm tune --dsn "$DSN" --vcpu 1 --memory-mb 629 --storage network-ssd --storage-gb 10 --rows 2000000 --avg-row-bytes 300
+```
+
+```
+Host 16 CPUs, 31188MB memory (this machine or container, running seedstorm)
+Database postgres 15.18, 6/100 connections in use
+Writers 2 (--workers)
+Generators 1 (--gen-workers)
+
+Why:
+ - writers 2: 1 vCPU on the database, 2 writers each
+ - generators 1: one generator keeps up with this database
+
+Disk: About 1.3GB of the 10.0GB free will be used.
+```
+
+The recommendation is an estimate from benchmarks on one machine: watch the rate during the first minute and adjust. Production, shared or high-availability databases get fewer writers. The web UI's **Recommend** dialog (workspace → Tuning) runs the same rules.
+
+| Flag | Default | Description |
+|------|---------|-------------|
+| `--db` / `--dsn` | — | As for `seed` |
+| `--vcpu` / `--memory-mb` | — | The database's CPU and memory |
+| `--storage` | — | `local-ssd`, `network-ssd` or `hdd` |
+| `--storage-gb` / `--iops` | — | Disk size and provisioned IOPS of a network disk |
+| `--rows` / `--avg-row-bytes` | — / `256` | Rows the run will write, for the disk check |
+| `--shared` / `--ha` / `--production` | false | Other workloads use it / synchronous replica / production: stay conservative |
---
@@ -588,6 +687,13 @@ What the UI gives you:
- **Multi-session** — hold several DBs open at once and switch from the topbar dropdown; a saved connection that is already live offers **Switch to** instead of a second connect. The workspace can clone schema from the active connection into another matching connected database.
- **Compare & mirror** — `/compare` puts any two connections (live or saved, engines may differ) side by side: per-table mirrored gauges of source and target rows, size, delta and column drift, filterable. The mirror panel plans a top-up or reset at any scale for the ticked tables, shows the plan, truncate list, skipped tables and sample rows in a review dialog, and runs only after you confirm (reset needs an extra acknowledgement). Counts refresh when the run ends. The last comparison of each pair is kept in the browser and restored when you come back, marked with its age. **Export counts** downloads or copies either side as a [`snapshot`](#snapshot) file (JSON or YAML); **Import counts** takes a file, a drop or pasted text and uses it as the source for compare and mirror.
- **Seed profiles** — `/profiles` builds [value rules](profiles.md) with a generator palette (click or drag tokens into a template), ordered column patterns with live example values and match counts, and a table explorer that shows what every column will get plus sample rows from the active connection. Profiles save to disk, import (file, drop or paste) and export (download or copy) as YAML, and are selectable in the workspace action bar and on Compare. **Ignored tables** lists globs never written, with the tables each one matches; the workspace greys those tables out and lists them in an **Ignored** tab.
+- **Runs you can leave** — every page shows a run strip with the phase, progress, elapsed time and whether the server is still answering (`running`, `quiet`, `reconnecting`, `lost`). Leave a page while a seed, compare or mirror runs and come back: the run is reattached with live progress, or its result is shown if it finished. A failure names the side, phase and table and lists what completed; if `serve` restarted, the page says the run was lost instead of spinning.
+- **Remembered settings** — rows, tuning, profile and compare options are kept per connection in the browser. Truncate, disable-FK and drop options are never remembered.
+- **Fast return to the workspace** — the graph draws from the cached schema first; row counts fill in afterwards (cached per connection, **↻** recounts) and a table that cannot be counted says so.
+- **Production connections** — tick *Production database* on a saved connection: writes to it (seed, fill empty, mirror, clone into it) need its label typed back, and relationship scans read planner estimates, one query at a time, unless you confirm an exact scan.
+- **Recommend** — the Tuning panel's dialog takes vCPU, memory, storage type and size, and recommends writers and generators with the reasons (`tune` on the command line).
+- **Snapshot counts** — saves the active connection's row counts (optionally with its relationships) to YAML or JSON, or hands them to Compare as an imported source; **Calibrate from a file** opens Compare with this connection as target.
+- **Analyze relationships** — measures children per parent for every foreign key in the background (exact or estimate, unindexed keys opt-in); each arrow in the graph gets an `avg · max` badge as its key finishes, and the table panel shows the details. On Compare, **Compare relationships** shows the same per key for both sides, the export can include them, and **Shape relationships like the source** makes a mirror follow them.
- **Standalone tools** — `/generate`, `/enrich`, `/export` mirror the CLI commands as forms.
- **Bounded resources** — text a job returns to the browser is capped at 20MB: a dry run's SQL stops there with a note of the rows left out, while generate and export refuse and point at the CLI `--out` flags, since a cut document would be broken. Job requests are limited to 21MB, and a connection's pooled database connections close after two idle minutes (the next query reopens one).
diff --git a/docs/development.md b/docs/development.md
index 337c2bb..9c5fd4b 100644
--- a/docs/development.md
+++ b/docs/development.md
@@ -111,6 +111,15 @@ scratch databases they create and drop, so they test exactly what a user runs:
| `TestSeed_ManyTablesDoNotKeepEveryKeyPool` | 100 tables × 30k rows peak 82MB; keeping every key pool peaked at 262MB |
| `TestWebJobs_SeedProgressIsTruthful` / `…CancelIsPromptAndLeavesAUsableServer` / `…GapsFillAndMirrorReportProgress` / `…ConcurrentSeedJobs` | In-process server on real DBs: progress is monotonic and ends at done == total == `COUNT(*)`, one `Table written` per table, cancel ends within 15s with no queries left, concurrent jobs with viewers joining and leaving stay correct. Run with `-race` (found two job-manager races) |
| `TestCloneSchema_Objects` | `clone-schema --objects all`: view on view, function, procedure and trigger work on the clone; nothing extra without the flags |
+| `TestReadScope_*` | Reads run in a read-only transaction the server enforces, time out server-side, stop waiting behind a DDL lock, and a cancelled MySQL query is killed on the server (it keeps running with the driver alone) |
+| `TestProduction_CLIRefusesWritesUnlessAllowed` | `--production` refuses seed/gaps/mirror/clone writes without `--allow-production`; nothing is written; dry runs work |
+| `TestCLI_ExitCodesSayWhatKindOfFailureItWas` / `TestCLI_InterruptExitsWith130` / `TestServe_PanicInOneJobLeavesTheServerAndOtherJobsRunning` | Built with `-tags faultinject`: a panic exits 70 with the phase and table, an error exits 1, Ctrl+C exits 130; in `serve` a panicking job fails alone while another job completes and the server keeps answering |
+| `TestFaultInjection_AbsentFromTheDefaultBinary` | The release binary contains no fault-injection code |
+| `TestIntrospect_PostgresConstraintsArePairedAndVisibleToReadOnlyRoles` / `TestSeed_ForeignKeysToNonKeyColumnsInsertValidReferences` | FK columns paired from `pg_constraint` (composite and cross-schema), and FKs to a non-key UNIQUE column reference real values |
+| `TestPartitionedTables_*` | Postgres partitioned tables are counted once and seeded inside their partition bounds; an expression partition key is refused before writing |
+| `TestDetectServer_ReadsCapacityOnBothEngines` / `TestTune_CLIRecommendsAndSeedUsesAuto` / `TestServers_TwoDatabasesOnOneServerAreShared` | Server capacity reads, `tune` output and `--workers auto`, shared-server detection |
+| `TestRelationships_*` | Hand-computed shapes on both engines, the unindexed-key gate, a timed-out key while the others finish, cancel keeping finished keys, `snapshot`/`introspect`/`compare --relationships` through the binary |
+| `TestShapedSeed_*` | A shaped profile seeds with no parent above max and the target average and zero share; `mirror --shape-like-source` copies a skewed source; shaped seeding keeps the memory bound (325k rows, ~124MB) |
Scratch databases on MySQL are created as `root` (`SEEDSTORM_MYSQL_ROOT_PASSWORD`, default `root`).
@@ -145,6 +154,11 @@ SEEDSTORM_E2E_KEEP=1 make test-e2e # keep the scratch databases for debuggin
| `access` | SELECT-only MySQL user: read-only badge, banner, warning chips |
| `profiles` | Ignore glob with live matches, saved to disk, Ignored tab in the workspace, exported YAML |
| `mobile` | 390 and 320 wide with search and a wide preview open: no overlap, no sideways scroll |
+| `memory` | Workspace and compare settings survive leaving the page (truncate is never kept); a run started before leaving is reattached |
+| `production` | Seeding a connection marked production asks for its label first |
+| `tuning` | Recommend dialog for a small managed database applies its writers |
+| `snapshot` | Snapshot counts from the workspace, download, compare another database against the file, calibrate from a file |
+| `relationships` | Analyze relationships (estimate, then exact), include them in a counts file, compare drift, export with relationships, mirror plan shaped like the source, import into a profile |
Specs find elements only through `data-testid` names kept in `e2e/support/selectors.ts`: when you change markup a journey uses, keep or move the test id and update that file. Each journey checks what the user sees and, when it writes, the database via SQL. Static assets are embedded in the binary, so the suite always runs against a fresh build.
@@ -169,18 +183,42 @@ All tests run automatically on every PR via GitHub Actions (`.github/workflows/p
| Job | What it checks |
|-----|---------------|
-| `title` | Conventional Commits format |
+| `pr-title` | Conventional Commits format |
| `review` | AI code review via reviewforge (Gemini) |
-| `validate` | Directory/file structure via structlint |
-| `test` | `go test ./...` + `make build` |
+| `gauntlet` | structlint, dupehound, and unit tests with `-race` (its `gotest` gate) |
| `lint` | `golangci-lint` |
| `integration` | Full 29-table suite, scenario evals and schema-clone tests with `-race` on each Postgres/MySQL pair |
+| `loadsim` | Resource-limited evals in Cloud SQL-shaped containers (below); the job summary prints the runner's size and every skip |
| `e2e` | Playwright journeys against a built binary (Postgres 15, MySQL 8.0) |
+`loadsim-measure.yml` measures throughput per profile and writer count when started by hand (`workflow_dispatch`); it never gates a PR.
+
The integration job in CI uses `-race -timeout 1500s` across the database-version matrix. Use the same timeout locally when running both engines back-to-back.
Throughput and memory numbers, and how to reproduce them, are in [benchmarks.md](benchmarks.md).
+### Resource-limited evals (loadsim)
+
+`integration/loadsim_*_test.go` (build tags `integration loadsim`) start throwaway database containers shaped like managed instances and run the binary against them, some of it inside a container with its own CPU and memory limits:
+
+| Profile | Database container | Settings from |
+|---------|-------------------|---------------|
+| `cloudsql-micro` | 1 vCPU, 629MB, 300 write IOPS | Cloud SQL docs (Postgres); `SHOW VARIABLES` of a MySQL 8.4 1 vCPU / 628.74MB instance |
+| `cloudsql-2vcpu` | 2 vCPU, 7.5GB, 3000 write IOPS | Cloud SQL docs; MySQL values estimated from the micro reference |
+
+They assert outcomes, never speed: seedstorm sees a container's CPU and memory limits (not the host's), few free connections lower the writers and the run completes, a full disk fails naming the table and the cause, a million rows seed under a 256MB container, and the read safeguards hold on the smallest instance. A test skips (visibly) when the machine lacks the memory for its profile or a throttle does not apply.
+
+```bash
+make dev-up
+make test-loadsim # needs Docker
+make test-loadsim ARGS=-run=TestLoadsim_Read # one test
+SEEDSTORM_LOADSIM_MEASURE=/tmp/measure.json SEEDSTORM_LOADSIM_PROFILES=cloudsql-micro make test-loadsim ARGS=-run=TestLoadsim_MeasureWriters
+```
+
+### Fault injection
+
+Reliability tests build the binary with `-tags faultinject`; `SEEDSTORM_FAULT=point:table:action` then makes a point fail (`write`, `generate`, `copy`, `job`, `introspect`; action `panic`, `error` or `hang`), e.g. `SEEDSTORM_FAULT=write:orders:panic`. The default build contains none of it.
+
### Supported database versions
| Postgres | MySQL | Notes |
@@ -215,7 +253,10 @@ The engine-specific suites (`TestMySQLIntegration`, `TestMySQLGaps`, `TestMySQLS
| `SEEDSTORM_PROFILE` | Default `--profile` for `seed`, `gaps`, `generate`, `mirror` |
| `SEEDSTORM_PROFILES` | Path of the saved-profile store (default `~/.config/seedstorm/profiles.yaml`) |
| `SEEDSTORM_SOURCE_DSN` / `SEEDSTORM_TARGET_DSN` | Defaults for `compare`, `mirror`, `clone-schema` |
+| `SEEDSTORM_PRODUCTION` | Same as `--production` on commands that write |
| `SEEDSTORM_PG_HOST` / `SEEDSTORM_PG_PORT` / `SEEDSTORM_MYSQL_HOST` / `SEEDSTORM_MYSQL_PORT` | Integration tests: where the databases listen |
+| `SEEDSTORM_FAULT` | Fault-injection builds only: `point:table:panic\|error\|hang` |
+| `SEEDSTORM_LOADSIM_MEASURE` / `SEEDSTORM_LOADSIM_PROFILES` / `SEEDSTORM_LOADSIM_ENGINES` | Loadsim measure: report path, profiles and engines |
---
@@ -226,6 +267,8 @@ make build Build for current platform → bin/seedstorm
make build-all Build for linux/darwin amd64+arm64 → dist/
make test Run unit tests
make test-integration Run integration tests (requires make dev-up)
+make test-loadsim Run resource-limited evals (requires Docker and make dev-up)
+make test-e2e Run Playwright journeys (requires make dev-up)
make lint Run golangci-lint
make fmt Format with gofumpt
make tidy go mod tidy
diff --git a/docs/profiles.md b/docs/profiles.md
index b3aaa05..2459969 100644
--- a/docs/profiles.md
+++ b/docs/profiles.md
@@ -42,8 +42,68 @@ tables: # explicit settings per table
columns:
role: { value: guest }
phone: { setNull: true }
+relationships: # children per parent for a foreign key (version 2)
+ orders.user_id:
+ min: 1
+ avg: 5.2
+ max: 34
+ zeroShare: 0.19 # share of parents with no children
+ nullShare: 0 # share of NULL keys (nullable columns)
+ histogram: # optional: parents per degree range
+ - {min: 1, max: 1, parents: 187}
+ - {min: 2, max: 3, parents: 340}
+ - {min: 4, max: 7, parents: 310}
+ - {min: 8, max: 34, parents: 133}
```
+### Relationships
+
+`relationships:` makes foreign keys look like real data instead of an even
+spread: most users with a few orders, some with many, some with none. Each key
+is `table.column` (matched to the database ignoring case). Seed, fill empty,
+generate and mirror deal parents so that:
+
+- no parent gets more than `max` children, and each parent with children gets at least `min` (exact while the parent table fits the key pool of 500,000 — see below);
+- the share of parents without children is `zeroShare`, and exactly `nullShare` of the rows have a NULL key;
+- degrees follow the `histogram` when given (bucket by its parent weight, then a value inside it), otherwise they spread around `avg`.
+
+Every foreign key of a table can be shaped at once; the keys are dealt
+independently, so they are not correlated (the users with many orders are not
+necessarily the ones with many reviews).
+
+When the planned rows cannot fit the shape, seedstorm adjusts in one bounded
+pass and warns instead of retrying: `max raised`, `share of parents without
+children lowered`, or fewer parents with children. More rows than planned
+(enum coverage) pick parents evenly and are reported. After a run that writes,
+the achieved shape is measured and logged next to the target.
+
+Not shaped (every run names them with the reason, and the profile validation
+warns too): self-references, junction tables whose key is made of foreign keys,
+and foreign keys that are part of the primary key. Real schemas hit this often:
+on Keycloak's 87 tables, 29 of 67 measured keys are shaped. A parent table above 500,000 rows is held as a
+sample: children are then dealt a round at a time over each sample, and the run
+says so. Seeding against a live database rotates the sample, so the whole
+parent table gets its share, and the shape lands close rather than exact — a
+parent caught in two samples can pass `max`, and parents never sampled raise
+the share without children (measured on 600,000 parents and 1.6M children:
+average 3 → 3.4, max 8 → 10, without children 10% → 22%). `generate` (no
+connection) can only reach the parents in the sample.
+
+Where shapes come from:
+
+- **Analyze relationships** in the workspace, `seedstorm snapshot --relationships`
+ or `introspect --relationships` measure them from a database; **Import from a
+ counts file** on the Profiles page turns such a file into `relationships:`.
+- `seed --shape-rows` derives the row count of each shaped child table from its
+ parents (parents × (1 − zeroShare) × avg ÷ (1 − nullShare)); `--table-rows`
+ still wins.
+- `mirror --shape-like-source` (or **Shape relationships like the source** on
+ Compare) uses the source's shapes directly, without a profile.
+
+A profile with relationships is written as `version: 2`; older seedstorm
+versions refuse it rather than seeding it unshaped. Without relationships a
+profile stays `version: 1`.
+
### Ignored tables
`ignore:` lists table-name globs that seedstorm never writes: seed, gaps, generate and mirror skip them (no inserts, no truncation). Globs match the whole name and ignore case, so `flyway_*` also matches MySQL's `FLYWAY_SCHEMA_HISTORY`; `*` matches any run of characters and `?` exactly one.
@@ -109,7 +169,8 @@ Validation (web UI issues panel, `seedstorm profile validate --dsn …`) reports
| Level | Examples |
|-------|----------|
| error | no action / two actions, unknown generator or token, bad glob, explicit rule on a key, `setNull` on `NOT NULL`, a value that cannot fit the column type |
-| warning | table or column not in this database (profiles stay portable), a rule that matches nothing or is shadowed by an earlier rule, a UNIQUE column (or a column inside a multi-column UNIQUE) given a fixed value or short list, a UNIQUE template relying on `{{seq}}` alone |
+| warning | table or column not in this database (profiles stay portable), a rule that matches nothing or is shadowed by an earlier rule, a UNIQUE column (or a column inside a multi-column UNIQUE) given a fixed value or short list, a UNIQUE template relying on `{{seq}}` alone, a relationship on a key that cannot be shaped |
+| error (relationships) | a key not written as `table.column`, max below min, avg outside min … max, shares outside 0 … 1, a bucket with min above max |
## CLI
diff --git a/docs/specs/001-connect-flow-overhaul.md b/docs/specs/001-connect-flow-overhaul.md
new file mode 100644
index 0000000..c61b248
--- /dev/null
+++ b/docs/specs/001-connect-flow-overhaul.md
@@ -0,0 +1,375 @@
+# 001 — Connect flow overhaul: extra params, test connection, saved connections
+
+> Working document. **Not committed** — this file stays untracked and out of the PR.
+
+## Context
+
+`seedstorm serve` opens on `/connect`, a single form that is the only door into the
+whole web UI. Three problems make that door harder to walk through than it should be.
+
+**1. No way to pass driver params from the structured form.** `buildDSN`
+(`internal/web/dsn.go:15`) hardcodes the query string per driver — `sslmode` for
+Postgres, `parseTime=true&multiStatements=true` for MySQL — with no seam for anything
+else. Any connection that needs an extra driver parameter is unreachable through the
+structured fields. Real drivers demand these routinely: `go-sql-driver/mysql` refuses
+to authenticate against a `caching_sha2_password` account over a non-TLS socket unless
+`allowPublicKeyRetrieval=1` is set, and against a cleartext-plugin account unless
+`allowCleartextPasswords=1`. The failure mode today is a red banner quoting the driver
+("please add 'allowCleartextPasswords=1' to your DSN") next to a form that has no field
+to add it to. The only workaround is abandoning the structured fields entirely and
+hand-assembling a raw DSN in the **Connection string** box — which also means
+hand-escaping the password.
+
+**2. Validating a connection costs a full page navigation.** The only way to find out
+whether a connection works is to submit the form. `handleConnect`
+(`internal/web/handlers_pages.go:22`) either opens a session and 303s to the workspace,
+or re-renders the whole connect page with `Error:` set. Iterating on one wrong parameter
+means a full round trip through a re-rendered page each time, and every re-render drops
+the password field.
+
+**3. Saved connections are browser-local, half-wired, and never the starting point.**
+The server already supports several live connections at once — `SessionRegistry`
+(`internal/web/session.go:52`) holds many, `/api/connections`, `/switch` and
+`/disconnect`+`Pick` all exist, and the header menu lists them. But *saved* connections
+are `localStorage` only (`PRESET_KEY = "seedstorm.connections.v1"`,
+`internal/web/static/app.js:6`): they are invisible to a second browser, lost on a cache
+clear, and — because sessions live only in server memory — every `serve` restart drops
+all live connections while the saved list is stranded client-side. Worse, the header
+menu disables any preset saved without a password (`btn.disabled = !p.password && !p.dsn`,
+`app.js:202`), so a security-conscious save produces an entry that cannot be opened.
+And on `/connect` itself, saved connections appear only as a bare `