Pre-check PACK header object count before pushing oversized batches · Entire

Pre-check PACK header object count before pushing oversized batches

fe63356→main·

Soph·3mo ago·2 files·+146 added/-0 removed

When a batch's FetchPack returns, peek at the first 12 bytes of the pack stream to read the PACK header's object count. Multiply by 750 bytes/object (a conservative compressed-object average) to estimate total pack size. If the estimate exceeds BatchMaxPack, subdivide immediately and re-fetch smaller ranges — avoiding a multi-GiB transfer that the target would reject anyway.

This is a zero-cost optimization: 12 bytes read + one multiplication. The PACK header is prepended back to the reader via MultiReader so the push receives a valid packfile if the check passes.

Combined with the post-push subdivide from the previous commit, the batch loop now has two layers of protection:

  1. Pre-push: PACK header estimate catches obviously-oversized batches before any significant transfer (this commit).
  2. Post-push: target body-limit rejection triggers subdivide for batches that passed the estimate but still exceeded the target's actual limit (previous commit).

For the linux kernel against a 2 GiB target:

Co-Authored-By: Claude Opus 4.6 (1M context) noreply@anthropic.com

Sessions

dbedffba0bc5View transcript

Changes

2

2 unmodified lines

3
4
5
6
7
8
9
242 unmodified lines

252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
144 unmodified lines

428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480

2 unmodified lines

package bootstrap

import (
    "bytes"
    "context"
    "encoding/json"
    "errors"
242 unmodified lines

return result, fmt.Errorf("fetch source batch pack for %s: %w", batch.Plan.TargetRef, err)
        }
        packReader = closeOnce(packReader)

// Peek at the PACK header (12 bytes) to get the object count.
        // If the estimated pack size exceeds the batch limit, subdivide
        // immediately instead of pushing a pack the target will reject.
        // This avoids wasting a multi-GiB transfer on a doomed push.
        if p.BatchMaxPack > 0 && len(batch.chain) > 0 {
            packReader, err = checkPackSizeAndSubdivide(packReader, p.BatchMaxPack, func() bool {
                expanded := subdivideCheckpoints(batch.chain, current, batch.Checkpoints[idx:])
                if len(expanded) > len(batch.Checkpoints[idx:]) {
                    p.log("bootstrap batch subdividing before push (pack header estimate)",
                    "branch", batch.Plan.TargetRef.String(),
                    "old_remaining", len(batch.Checkpoints[idx:]),
                    "new_remaining", len(expanded))
                    batch.Checkpoints = append(batch.Checkpoints[:idx], expanded...)
                    return true
                }
                return false
            })
            if err != nil {
                return result, fmt.Errorf("check pack size for %s: %w", batch.Plan.TargetRef, err)
            }
            if packReader == nil {
                continue // subdivided, retry at same idx
            }
        }

cmds := convert.PlansToPushCommands(stagePlans)
        if err := p.TargetPusher.PushPack(ctx, cmds, packReader); err != nil {
            _ = packReader.Close()
144 unmodified lines

return n
}

// estimatedBytesPerObject is a conservative average for compressed git objects
// in a packfile. Used with the PACK header's object count to estimate total
// pack size before streaming the full pack. Real values range from ~200 bytes
// (tiny commits in a sparse repo) to ~2 KiB (blob-heavy repos), with most
// mature repos averaging 500–1000 bytes. 750 is a reasonable middle ground.
const estimatedBytesPerObject = 750

// checkPackSizeAndSubdivide reads the 12-byte PACK header to get the object
// count, estimates total pack size, and if it exceeds batchLimit, closes the
// reader and calls subdivide(). Returns (nil, nil) when subdivided (caller
// should continue to retry), or (prepended reader, nil) to proceed with push.
func checkPackSizeAndSubdivide(
    r io.ReadCloser,
    batchLimit int64,
    subdivide func() bool,
) (io.ReadCloser, error) {
    var header [12]byte
    n, err := io.ReadFull(r, header[:])
    if err != nil {
        // Short pack or error — let the push handle it
        prefixed := io.MultiReader(bytes.NewReader(header[:n]), r)
        return &wrappedMultiRC{Reader: prefixed, Closer: r}, nil
    }
    if string(header[:4]) != "PACK" {
        // Not a standard packfile — can't estimate, proceed
        prefixed := io.MultiReader(bytes.NewReader(header[:]), r)
        return &wrappedMultiRC{Reader: prefixed, Closer: r}, nil
    }
    objectCount := int64(header[8])<<24 | int64(header[9])<<16 | int64(header[10])<<8 | int64(header[11])
    estimated := objectCount * estimatedBytesPerObject

if estimated > batchLimit && subdivide() {
        _ = r.Close()
        return nil, nil
    }

prefixed := io.MultiReader(bytes.NewReader(header[:]), r)
    return &wrappedMultiRC{Reader: prefixed, Closer: r}, nil
}

type wrappedMultiRC struct {
    io.Reader
    io.Closer
}

func (w *wrappedMultiRC) Read(p []byte) (int, error) { return w.Reader.Read(p) }

// subdivideCheckpoints splits each remaining checkpoint range in half using
// the full commit chain. Called when a batch push is rejected for exceeding
// the target's body-size limit. Returns the expanded checkpoint list; if no

``