Schema reconciliation + index hygiene, and the weekly base-image refresh #243
@@ -17,9 +17,10 @@
|
|||||||
# still has them. That is what makes this the proof for step 1 and the
|
# still has them. That is what makes this the proof for step 1 and the
|
||||||
# pre-flight for step 2, rather than a one-shot script.
|
# pre-flight for step 2, rather than a one-shot script.
|
||||||
#
|
#
|
||||||
# `generate: true` additionally autogenerates a candidate baseline from the
|
# While the chain is still present it also autogenerates a candidate baseline
|
||||||
# models and uploads it. That is a starting point, NOT the answer: autogenerate
|
# from the models and prints it. That is a starting point, NOT the answer:
|
||||||
# reads SQLAlchemy metadata, and three things in this schema do not live there —
|
# autogenerate reads SQLAlchemy metadata, and three things here do not live
|
||||||
|
# there —
|
||||||
# * CREATE EXTENSION vector (0001)
|
# * CREATE EXTENSION vector (0001)
|
||||||
# * CREATE EXTENSION tsm_system_rows (0004)
|
# * CREATE EXTENSION tsm_system_rows (0004)
|
||||||
# * the HNSW index on image_record.siglip_embedding, which is raw SQL
|
# * the HNSW index on image_record.siglip_embedding, which is raw SQL
|
||||||
@@ -36,10 +37,6 @@ on:
|
|||||||
description: 'Commit/tag that still carries the full 0001..0087 chain'
|
description: 'Commit/tag that still carries the full 0001..0087 chain'
|
||||||
type: string
|
type: string
|
||||||
default: '0a5bbe8'
|
default: '0a5bbe8'
|
||||||
generate:
|
|
||||||
description: 'Also autogenerate a candidate baseline from the models'
|
|
||||||
type: boolean
|
|
||||||
default: false
|
|
||||||
|
|
||||||
jobs:
|
jobs:
|
||||||
compare:
|
compare:
|
||||||
@@ -111,13 +108,23 @@ jobs:
|
|||||||
--no-owner --no-privileges -d fc_chain > chain.sql
|
--no-owner --no-privileges -d fc_chain > chain.sql
|
||||||
wc -l chain.sql
|
wc -l chain.sql
|
||||||
|
|
||||||
# Optional: a candidate baseline, autogenerated from the models against an
|
# A candidate baseline, autogenerated from the models against an EMPTY
|
||||||
# EMPTY database so every table shows up as a create. Uploaded for a human
|
# database so every table shows up as a create. Printed for a human to
|
||||||
# to finish — it will be missing the three raw-SQL items named at the top.
|
# finish — it will be missing the three raw-SQL items named at the top.
|
||||||
|
#
|
||||||
|
# Gated on the TREE, not on a workflow input. A `type: boolean` input
|
||||||
|
# read back as `github.event.inputs.generate == 'true'` silently
|
||||||
|
# evaluated false on this runner (run 4960 skipped this step entirely
|
||||||
|
# with no diagnostic) — the same `github.event.inputs` typing quirk
|
||||||
|
# build.yml already works around. The file count is the real question
|
||||||
|
# anyway: there is nothing to generate once the chain is collapsed.
|
||||||
- name: Autogenerate a candidate baseline
|
- name: Autogenerate a candidate baseline
|
||||||
if: ${{ github.event.inputs.generate == 'true' }}
|
|
||||||
run: |
|
run: |
|
||||||
set -eux
|
set -eux
|
||||||
|
if [ "$(ls alembic/versions/*.py | wc -l)" -le 1 ]; then
|
||||||
|
echo "already collapsed — nothing to generate"
|
||||||
|
exit 0
|
||||||
|
fi
|
||||||
docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_gen
|
docker exec "$PG_CONTAINER" createdb -U fabledcurator fc_gen
|
||||||
# Hide the existing revisions so alembic sees an empty history and
|
# Hide the existing revisions so alembic sees an empty history and
|
||||||
# emits the whole schema rather than a delta.
|
# emits the whole schema rather than a delta.
|
||||||
@@ -153,16 +160,25 @@ jobs:
|
|||||||
#
|
#
|
||||||
# pg_dump orders dumpable objects by name within type, not by creation
|
# pg_dump orders dumpable objects by name within type, not by creation
|
||||||
# order, so two schemas built by different routes are directly
|
# order, so two schemas built by different routes are directly
|
||||||
# comparable. The only normalisation applied is dropping blank lines,
|
# comparable. Normalisation is deliberately minimal, because a filter
|
||||||
# comment lines and the alembic_version row-count noise — deliberately
|
# that hides a real difference is the one way this check passes when it
|
||||||
# minimal, because a filter that hides a real difference is the one way
|
# should fail — blank lines, SQL comments, trailing whitespace, and:
|
||||||
# this check passes when it should fail. Whatever is normalised is
|
#
|
||||||
# printed, so the filtering itself is reviewable.
|
# \restrict / \unrestrict — a per-invocation RANDOM NONCE that newer
|
||||||
|
# pg_dump emits to fence the dump against injection during restore. It
|
||||||
|
# differs on every run by construction, so it is noise by definition,
|
||||||
|
# not a schema difference. Measured on run 4960, the control: two dumps
|
||||||
|
# of the SAME schema came back 1123 lines each and differed on exactly
|
||||||
|
# these two lines and nothing else. That control is what licenses this
|
||||||
|
# filter — it was observed to be the only false positive, rather than
|
||||||
|
# assumed to be one.
|
||||||
- name: Diff
|
- name: Diff
|
||||||
run: |
|
run: |
|
||||||
set -eu
|
set -eu
|
||||||
norm() {
|
norm() {
|
||||||
grep -vE '^\s*(--|$)' "$1" | sed 's/[[:space:]]*$//'
|
grep -vE '^\s*(--|$)' "$1" \
|
||||||
|
| grep -vE '^\\(un)?restrict ' \
|
||||||
|
| sed 's/[[:space:]]*$//'
|
||||||
}
|
}
|
||||||
norm chain.sql > a.txt
|
norm chain.sql > a.txt
|
||||||
norm baseline.sql > b.txt
|
norm baseline.sql > b.txt
|
||||||
|
|||||||
Reference in New Issue
Block a user