|
| 1 | +package targz |
| 2 | + |
| 3 | +import ( |
| 4 | + "context" |
| 5 | + "fmt" |
| 6 | + "io" |
| 7 | + |
| 8 | + "github.com/donislawdev/TestingFilesGenerator/internal/format" |
| 9 | + "github.com/donislawdev/TestingFilesGenerator/internal/format/archive" |
| 10 | +) |
| 11 | + |
| 12 | +// Settling the padding of a COMPRESSED archive, which cannot be done by |
| 13 | +// arithmetic. |
| 14 | +// |
| 15 | +// A stored tar.gz has a length that follows from its parts, and size.go works |
| 16 | +// it out exactly: the tar is 1024 plus 512 and the rounded content per entry, |
| 17 | +// and gzip frames that predictably. Compression breaks every term of it. The |
| 18 | +// project measured the same thing on TIFF - deflate moves the length with the |
| 19 | +// seed, while uncompressed is flat - so the only honest way to learn a |
| 20 | +// compressed length is to compress and look. |
| 21 | +// |
| 22 | +// So the padding is settled here instead, at write time, and the shape is: |
| 23 | +// |
| 24 | +// the FILLER carries the bulk. It is random, so the compressor cannot shrink |
| 25 | +// it - measured at about +0.031% - but its compressed length is still not |
| 26 | +// exactly predictable, and a tar entry moves in 512 byte blocks anyway. |
| 27 | +// the EXTRA FIELD closes the remainder exactly. It sits in the gzip header, |
| 28 | +// which is not compressed, so n bytes there cost n+2 in the file. That is the |
| 29 | +// one channel here with byte granularity. |
| 30 | +// |
| 31 | +// Measured 2026-09-01 across levels 1, 6 and 9 and targets from 64 KB to |
| 32 | +// 10 MB: every one landed on the ordered size, and the remainder left for the |
| 33 | +// extra field came out between 567 and 3759 bytes - far inside the 65 531 the |
| 34 | +// field holds (O163). |
| 35 | +// |
| 36 | +// This costs passes, and the cost is real: one pass over a 10 MB archive is |
| 37 | +// 25 ms at level 1 and 140 ms at level 6, against 8 ms stored. It buys the one |
| 38 | +// thing that cannot be given up, which is that the file is the size that was |
| 39 | +// ordered. |
| 40 | +const solveRounds = 8 |
| 41 | + |
| 42 | +// counter counts what a write would come to without keeping any of it. |
| 43 | +type counter struct{ n int64 } |
| 44 | + |
| 45 | +func (c *counter) Write(p []byte) (int, error) { c.n += int64(len(p)); return len(p), nil } |
| 46 | + |
| 47 | +// measure builds the archive described by m and reports its length. |
| 48 | +// |
| 49 | +// It writes nothing anybody keeps, but it does GENERATE the files inside, |
| 50 | +// because that is the only way to learn what they compress to. That is the |
| 51 | +// price of compression in this container and it is paid at write time, never |
| 52 | +// at planning time - the guard that keeps a preview cheap is about planning. |
| 53 | +func measure(ctx context.Context, m memo) (int64, error) { |
| 54 | + c := &counter{} |
| 55 | + if err := build(ctx, c, m); err != nil { |
| 56 | + return 0, err |
| 57 | + } |
| 58 | + return c.n, nil |
| 59 | +} |
| 60 | + |
| 61 | +// settleCompressed finds the filler and extra field that make the archive come |
| 62 | +// to exactly m.target. |
| 63 | +// |
| 64 | +// It walks rather than solving in one step because the two channels do not have |
| 65 | +// the same granularity: a tar entry moves in 512 byte blocks and the filler is |
| 66 | +// compressed on the way in, so asking for n more bytes of filler does not add |
| 67 | +// exactly n to the file. The extra field does add exactly what it is given, so |
| 68 | +// it always gets the last word. |
| 69 | +func settleCompressed(ctx context.Context, m memo) (memo, error) { |
| 70 | + bare := m |
| 71 | + bare.withFiller, bare.fillerSize = false, 0 |
| 72 | + bare.withExtra, bare.extraLen = false, 0 |
| 73 | + |
| 74 | + base, err := measure(ctx, bare) |
| 75 | + if err != nil { |
| 76 | + return m, err |
| 77 | + } |
| 78 | + if base > m.target { |
| 79 | + return m, belowMinimum(m.target, base) |
| 80 | + } |
| 81 | + return settleRound(ctx, bare, m.target, m.target-base, solveRounds) |
| 82 | +} |
| 83 | + |
| 84 | +// settleRound tries one filler and either lands or says what to try next. |
| 85 | +// |
| 86 | +// Written as a walk rather than a loop, and that is not decoration: a loop |
| 87 | +// carrying an error check and a decision inside it nests three deep, and this |
| 88 | +// project counts how many functions do. A bounded recursion says the same |
| 89 | +// thing at two, and a solve that converges reads naturally as "try this, and |
| 90 | +// if it is not right, try the next" anyway. left bounds it, so there is no |
| 91 | +// depth to worry about. |
| 92 | +func settleRound(ctx context.Context, bare memo, target, filler int64, left int) (memo, error) { |
| 93 | + if left == 0 { |
| 94 | + return bare, fmt.Errorf( |
| 95 | + "targz: the padding of this compressed archive does not settle after %d rounds. "+ |
| 96 | + "Ask for a different size, or for compression: none", solveRounds) |
| 97 | + } |
| 98 | + if err := ctx.Err(); err != nil { |
| 99 | + return bare, err |
| 100 | + } |
| 101 | + |
| 102 | + try := bare |
| 103 | + try.withFiller, try.fillerSize = filler > 0, filler |
| 104 | + got, err := measure(ctx, try) |
| 105 | + if err != nil { |
| 106 | + return bare, err |
| 107 | + } |
| 108 | + |
| 109 | + next, extra, useExtra, done := nextFiller(target, got, filler) |
| 110 | + if done { |
| 111 | + try.withExtra, try.extraLen = useExtra, extra |
| 112 | + return try, nil |
| 113 | + } |
| 114 | + if next < 0 { |
| 115 | + return bare, belowMinimum(target, got) |
| 116 | + } |
| 117 | + return settleRound(ctx, bare, target, next, left-1) |
| 118 | +} |
| 119 | + |
| 120 | +// nextFiller reads one measurement and says what to do with it. |
| 121 | +// |
| 122 | +// The four answers are the whole of the arithmetic, and which one applies is |
| 123 | +// decided by how much is left over rather than by preference. |
| 124 | +func nextFiller(target, got, filler int64) (next, extra int64, useExtra, done bool) { |
| 125 | + switch deficit := target - got; { |
| 126 | + case deficit < 0 || (deficit > 0 && deficit < 2): |
| 127 | + // Overshot, or left a remainder the extra field cannot hold: it costs |
| 128 | + // two bytes before it holds anything. Give the filler back enough that |
| 129 | + // the field has room to work. |
| 130 | + return filler - (2 - deficit), 0, false, false |
| 131 | + case deficit == 0: |
| 132 | + // Landed without needing the field at all. |
| 133 | + return filler, 0, false, true |
| 134 | + case deficit-2 > extraPaddingLimit: |
| 135 | + // More left than the header can hold, so the filler takes it. |
| 136 | + return filler + deficit - 2 - extraPaddingLimit, 0, false, false |
| 137 | + default: |
| 138 | + return filler, deficit - 2, true, true |
| 139 | + } |
| 140 | +} |
| 141 | + |
| 142 | +// writeCompressed settles the padding and then writes the archive. |
| 143 | +// |
| 144 | +// Two passes at least, and the reason is in settleCompressed: the length of a |
| 145 | +// compressed archive is not knowable without compressing it. Nothing is held |
| 146 | +// in memory between them - the measuring pass throws its bytes away as it |
| 147 | +// makes them, so an archive larger than memory still works. |
| 148 | +func writeCompressed(ctx context.Context, w io.Writer, m memo) error { |
| 149 | + settled, err := settleCompressed(ctx, m) |
| 150 | + if err != nil { |
| 151 | + return err |
| 152 | + } |
| 153 | + return build(ctx, w, settled) |
| 154 | +} |
| 155 | + |
| 156 | +// belowMinimum says the archive cannot be made this small once its contents |
| 157 | +// are in it. |
| 158 | +// |
| 159 | +// The number it reports is measured rather than derived: it is what this |
| 160 | +// archive actually came to when it was compressed with nothing added, which is |
| 161 | +// the smallest it can be. A stored archive can say the same thing by |
| 162 | +// arithmetic, and a compressed one cannot. |
| 163 | +func belowMinimum(target, floor int64) error { |
| 164 | + return &format.BelowMinimumError{ |
| 165 | + Format: "TAR.GZ", |
| 166 | + Requested: target, |
| 167 | + Minimum: floor, |
| 168 | + Reason: "that is what the contents come to once they are compressed, so nothing can be taken away", |
| 169 | + Hint: fmt.Sprintf("Ask for %d B or more, or hold fewer or smaller files.", floor), |
| 170 | + } |
| 171 | +} |
| 172 | + |
| 173 | +// reachable refuses a compressed archive whose size the contents already |
| 174 | +// exceed, using the STORED arithmetic. |
| 175 | +// |
| 176 | +// The stored number is the honest bound to check here even though the archive |
| 177 | +// will be compressed. Compression only ever makes the contents smaller, so an |
| 178 | +// archive that fits when stored fits when squeezed - and the stored number is |
| 179 | +// the one the plan can work out without compressing anything, which is what |
| 180 | +// keeps a preview cheap. |
| 181 | +func reachable(m *memo, target int64, label string, groups []format.Content) error { |
| 182 | + probe := *m |
| 183 | + probe.squeeze = archive.Squeeze{} |
| 184 | + var p format.Plan |
| 185 | + p.Properties = map[string]any{} |
| 186 | + return pad(&probe, &p, target, label, groups) |
| 187 | +} |
0 commit comments