MEASURED FIRST, then fixed. On the shipped v0.229.0, on demo-hp, an app's Tier-2 copy went from 120 082 104 B (4 database dumps + 3 named-volume tars) to 7 036 B (none of either) in ONE nightly run, and the run recorded itself a success: 'Tier 2 copied docmost -> ... (14.9 KB, 0 leg(s), 0s)'. Evidence: felhom.eu/documentation/audits/DRILL-r403-tier2-delete-2026-08-31/. The mechanism was three individually-correct lines: RunTier2 guards the unit leg with os.Stat only (does the folder exist), rsyncMirror is rsync -a --delete, and nothing between them compared source to destination. An EMPTY unit is a folder that exists. THE GUARD. One predicate, unitCarriesData/unitIsHollow (r403_hollow.go), asking the MANIFEST and never the byte size - a big compose tree with no dumps is dangerous, a tiny unit for a tiny app is fine. Fail closed on an absent or unparseable manifest. RunTier2 skips the unit leg when the source is hollow AND the destination is not; the other legs still run, the run is not failed, and the skip is recorded for the SURFACE (CrossDriveBackup.UnitLegSkipped + UnitPackageDate) as well as logged. --delete STAYS and shrinking stays legal. 07 section 8 row 5's derived-copy rule is unchanged; the fence is exactly one shape. TestR403_DataLegShrinkIsUnaffected is the guard on the guard. THE HONESTY. A preserved package is older than the run that preserved it, so the card carries a notice and the unit-restore confirm names the PACKAGE's date - read from the mirrored manifest's own created_at, not from the status record - plus a clause saying why it is older. THE CAUSE. RestoreTier2Unit now refills a hollow or absent primary unit from the mirror it just restored from, INSIDE the call before returning. The hollow manifest was written two seconds after a restore by the 5-minute capture job; any follow-up job races it. The capture itself is NOT guarded: a capture describing an empty drive as empty is correct, and with the primary refilled there is no hollow state left to describe. Never over a complete primary, never after a failed restore. recordTier2Success and tier2UnitConfirmMsg keep their old signatures as thin callers, so no existing test needed editing. New seam unitRehydrate, separate from tier2Mirror on purpose. 22 new Go tests. Red-proofs run and reverted: A6 (predicate -> size threshold), B1 (guard removed -> the copy's 3 files are DELETED and the seam is called), B6 (a general never-shrink rule -> the shrink case fails), C2 (only-when-hollow dropped -> the complete primary is overwritten).
This commit is contained in:
@@ -365,8 +365,25 @@ func (m *Manager) RunTier2(stackName string) error {
|
||||
}
|
||||
}
|
||||
|
||||
// Unit leg (always).
|
||||
if err := mirror(unitDir, filepath.Join(destBase, "recovery-unit")); err != nil {
|
||||
// Unit leg — always, EXCEPT the one case R-403 measured (see r403_hollow.go for the mechanism).
|
||||
//
|
||||
// `mirror` is `rsync -a --delete`. That is correct for a derived copy and it stays. What was
|
||||
// missing is the precondition: a source unit that carries NO data must not be mirrored over a
|
||||
// destination unit that does, because `--delete` then removes the customer's last package. Proven
|
||||
// on demo-hp 2026-08-31: 120 082 104 B → 7 036 B in one run, reported as a success.
|
||||
//
|
||||
// The fence is EXACTLY this shape and no wider. Complete→complete, complete→hollow and
|
||||
// hollow→hollow all mirror as before; a data leg that legitimately shrinks is untouched (this
|
||||
// guard is on the unit leg only). §8 row 5's derived-copy rule is unchanged.
|
||||
destUnit := filepath.Join(destBase, "recovery-unit")
|
||||
unitLegSkipped := unitIsHollow(unitDir) && unitCarriesData(destUnit)
|
||||
if unitLegSkipped {
|
||||
// Loud, and it names the app, the reason and the consequence. Paths and counts only — a unit's
|
||||
// compose/app.yaml carries portable secrets and nothing from inside it is logged here.
|
||||
m.logger.Printf("[WARN] [backup] Tier 2 %s: unit leg SKIPPED — the recovery unit on the source drive lists no database dumps and no volume tars, while the existing copy at %s does. The copy was PRESERVED rather than replaced with an empty one (R-403). The other legs continue.",
|
||||
stackName, destUnit)
|
||||
warns = append(warns, tier2UnitPreservedWarning)
|
||||
} else if err := mirror(unitDir, destUnit); err != nil {
|
||||
m.recordTier2Failure(stackName, target, err)
|
||||
if m.tier2Notify != nil {
|
||||
m.tier2Notify(stackName, target.Label, time.Since(start), err)
|
||||
@@ -395,13 +412,19 @@ func (m *Manager) RunTier2(stackName string) error {
|
||||
}
|
||||
|
||||
dur := time.Since(start)
|
||||
m.recordTier2Success(stackName, target, mirroredSize, strings.Join(warns, " "), dur)
|
||||
// R-403: the package date is read from the DESTINATION unit's own manifest, so it describes what
|
||||
// is actually in the copy whether the leg mirrored or was preserved. Reading the artifact rather
|
||||
// than assuming the run's own timestamp is what stops the surface calling a preserved package
|
||||
// fresh — a data loss traded for a comforting lie is not a fix.
|
||||
m.recordTier2SuccessWithUnit(stackName, target, mirroredSize, strings.Join(warns, " "), dur,
|
||||
unitLegSkipped, unitPackageDate(destUnit))
|
||||
if m.tier2Notify != nil {
|
||||
m.tier2Notify(stackName, target.Label, dur, nil)
|
||||
}
|
||||
m.logger.Printf("[INFO] [backup] Tier 2 copied %s → %s (%s, %d leg(s), %s)%s",
|
||||
m.logger.Printf("[INFO] [backup] Tier 2 copied %s → %s (%s, %d leg(s), %s)%s%s",
|
||||
stackName, destBase, humanizeBytes(mirroredSize), len(legs), dur.Round(time.Second),
|
||||
map[bool]string{true: " [SSD: state-only]", false: ""}[target.StateOnly])
|
||||
map[bool]string{true: " [SSD: state-only]", false: ""}[target.StateOnly],
|
||||
map[bool]string{true: " [unit leg SKIPPED — existing package preserved, R-403]", false: ""}[unitLegSkipped])
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -571,9 +594,24 @@ func (m *Manager) tier2Update(stackName string, mutate func(*settings.CrossDrive
|
||||
}
|
||||
}
|
||||
|
||||
// recordTier2Success records a completed run for a path that HAS NO UNIT LEG — the shares
|
||||
// pseudo-stack mirrors share legs and carries no recovery unit, so there is no leg to skip and no
|
||||
// package whose date could be older than the run. It is the thin caller; ONE implementation lives
|
||||
// below.
|
||||
func (m *Manager) recordTier2Success(stackName string, target *Tier2Target, sizeBytes int64, warning string, dur time.Duration) {
|
||||
m.recordTier2SuccessWithUnit(stackName, target, sizeBytes, warning, dur, false, "")
|
||||
}
|
||||
|
||||
// recordTier2SuccessWithUnit records a completed run INCLUDING what happened to its unit leg.
|
||||
// `unitLegSkipped` and `unitPkgDate` (R-403) travel with the rest rather than through a second write,
|
||||
// because two writers to one record is how a status and the thing it describes drift apart.
|
||||
// `unitPkgDate` is read from the destination unit's OWN manifest, so it is a fact about the copy and
|
||||
// not about the run.
|
||||
func (m *Manager) recordTier2SuccessWithUnit(stackName string, target *Tier2Target, sizeBytes int64, warning string, dur time.Duration, unitLegSkipped bool, unitPkgDate string) {
|
||||
now := time.Now().Format(time.RFC3339)
|
||||
m.tier2Update(stackName, func(c *settings.CrossDriveBackup) {
|
||||
c.UnitLegSkipped = unitLegSkipped
|
||||
c.UnitPackageDate = unitPkgDate
|
||||
c.Enabled = true
|
||||
c.Method = "rsync"
|
||||
c.DestinationPath = target.NamespaceRoot
|
||||
@@ -652,6 +690,21 @@ func rsyncMirror(src, dst string) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// tier2UnitPreservedWarning is the customer-facing half of the R-403 skip. It rides in
|
||||
// CrossDriveBackup.LastWarning, which the per-app card already renders, so the refusal reaches the
|
||||
// SURFACE and not only the log — the shape recordTier2NoTarget established.
|
||||
const tier2UnitPreservedWarning = "A fő meghajtón lévő adatcsomag hiányos volt, ezért a másodlagos másolatban meglévő, teljes csomagot megőriztük. A másolat adatcsomagja ezért régebbi, mint ez a mentés."
|
||||
|
||||
// unitPackageDate returns the `created_at` of a recovery unit's manifest — WHEN the package in that
|
||||
// directory was captured. "" when the manifest is absent or unparseable, which the surface must read
|
||||
// as UNKNOWN and never as "now".
|
||||
func unitPackageDate(unitDir string) string {
|
||||
if man := readManifest(UnitManifestFile(unitDir)); man != nil {
|
||||
return man.CreatedAt
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// dirSizeBytes returns the total size of a directory via `du -sb` (0 if absent/error).
|
||||
func dirSizeBytes(dir string) int64 {
|
||||
if _, err := os.Stat(dir); err != nil {
|
||||
|
||||
Reference in New Issue
Block a user