fix(nwsync): stream emit so peak memory tracks the largest resource (#76) (#78)
build-binaries / build-binaries (push) Successful in 2m13s
build-binaries / build-binaries (push) Successful in 2m13s
Closes #76. Part of #54. `nwsync emit` held roughly 5× the artifact size in RAM, so ovh-main (7 GB, no swap) OOM-killed it on any hak over ~1.4 GB. That blocked the backfill in sow-assets-manifest and would have killed the next release rebuilding a large hak. ## What it does - `erf.ReadIndex` / `erf.ReadPayload` — parse the header and resource table only, read one payload on demand. `erf.Read` keeps its shape but returns payloads as subslices of the buffer instead of fresh copies, which removes one full copy for the `pipeline` callers too. Callers must not mutate `Resource.Data`; the doc comment says so and no caller does. - `nwsync.Emit` opens the artifact, hashes it by streaming for the key check (through a section reader, so the file offset stays put), then hashes, compresses and stores one resource at a time. The archive is never resident. Shadowed duplicate resrefs are now never read at all. - Bounds checks in `ReadIndex` moved to `int64`, so key/resource-list offsets can no longer overflow. ## Not in the issue, but memory-motivated The zstd blob encoder ran at the default concurrency, which is one encoder per CPU, each holding a window-sized history — about 200 MB of live heap doing nothing on a 24-core runner. `EncodeAll` is single-threaded per call, so concurrency 1 costs nothing. `TestSingleThreadedEncoderMatchesDefault` pins the claim that blobs come out byte-identical. ## Measured Peak heap during emit, sampled 1 ms: | hak | before | after | |-----|--------|-------| | 8 MB | 155 MB | 21 MB | | 64 MB | 289 MB | 22 MB | Flat, as the acceptance asks. `TestEmitPeakMemoryDoesNotScaleWithArtifactSize` fails if the 64 MB fixture costs more than the 8 MB one plus 24 MB of slack. ## Gaps - The regression check measures Go heap, not RSS, and its largest fixture is 64 MB — a multi-GB run was not done here. A 2.15 GB hak now needs about the same ~22 MB the 64 MB one does, so the 5 GB budget is not close, but that is inference from the flat curve, not a measurement. - mmap was suggested in the issue and skipped. Payload buffers are still anonymous, but they are one resource each (≤15 MiB), so making them file-backed buys nothing now. 🤖 Generated with [Claude Code](https://claude.com/claude-code)Reviewed-on: #78 Reviewed-by: xtul <mpiasecki720@protonmail.com> Co-authored-by: vickydotbat <vickydotbat@tutamail.com>
This commit was merged in pull request #78.
This commit is contained in:
+92
-45
@@ -311,63 +311,110 @@ func Write(w io.Writer, archive Archive) error {
|
||||
return nil
|
||||
}
|
||||
|
||||
// IndexEntry locates one resource inside an archive without holding its
|
||||
// payload. Streaming callers read one payload at a time from these, so peak
|
||||
// memory tracks the largest resource instead of the whole archive.
|
||||
type IndexEntry struct {
|
||||
Name string
|
||||
Type uint16
|
||||
Offset int64
|
||||
Size int64
|
||||
}
|
||||
|
||||
// Index is the header plus the resource table of an ERF: everything except the
|
||||
// payloads.
|
||||
type Index struct {
|
||||
FileType string
|
||||
Version string
|
||||
Entries []IndexEntry
|
||||
}
|
||||
|
||||
// ReadIndex parses the tables of an ERF of the given size, reading only the
|
||||
// header, the key list and the resource list.
|
||||
func ReadIndex(r io.ReaderAt, size int64) (Index, error) {
|
||||
if size < headerSize {
|
||||
return Index{}, fmt.Errorf("erf file too small: %d bytes", size)
|
||||
}
|
||||
|
||||
var hdr header
|
||||
if err := binary.Read(io.NewSectionReader(r, 0, headerSize), binary.LittleEndian, &hdr); err != nil {
|
||||
return Index{}, fmt.Errorf("decode erf header: %w", err)
|
||||
}
|
||||
|
||||
if int64(hdr.KeyListOffset)+int64(hdr.EntryCount)*24 > size {
|
||||
return Index{}, fmt.Errorf("erf key list exceeds file bounds")
|
||||
}
|
||||
keys := make([]keyEntry, hdr.EntryCount)
|
||||
keyReader := io.NewSectionReader(r, int64(hdr.KeyListOffset), int64(hdr.EntryCount)*24)
|
||||
if err := binary.Read(keyReader, binary.LittleEndian, &keys); err != nil {
|
||||
return Index{}, fmt.Errorf("decode key list: %w", err)
|
||||
}
|
||||
|
||||
if int64(hdr.ResourceListOffset)+int64(hdr.EntryCount)*8 > size {
|
||||
return Index{}, fmt.Errorf("erf resource list exceeds file bounds")
|
||||
}
|
||||
entries := make([]resourceEntry, hdr.EntryCount)
|
||||
entryReader := io.NewSectionReader(r, int64(hdr.ResourceListOffset), int64(hdr.EntryCount)*8)
|
||||
if err := binary.Read(entryReader, binary.LittleEndian, &entries); err != nil {
|
||||
return Index{}, fmt.Errorf("decode resource list: %w", err)
|
||||
}
|
||||
|
||||
index := Index{
|
||||
FileType: string(hdr.FileType[:]),
|
||||
Version: string(hdr.Version[:]),
|
||||
Entries: make([]IndexEntry, 0, hdr.EntryCount),
|
||||
}
|
||||
for position, key := range keys {
|
||||
entry := entries[position]
|
||||
if int64(entry.Offset)+int64(entry.Size) > size {
|
||||
return Index{}, fmt.Errorf("resource %d exceeds file bounds", position)
|
||||
}
|
||||
index.Entries = append(index.Entries, IndexEntry{
|
||||
Name: string(bytes.TrimRight(key.ResRef[:], "\x00")),
|
||||
Type: key.ResourceType,
|
||||
Offset: int64(entry.Offset),
|
||||
Size: int64(entry.Size),
|
||||
})
|
||||
}
|
||||
return index, nil
|
||||
}
|
||||
|
||||
// ReadPayload returns one resource's bytes.
|
||||
func ReadPayload(r io.ReaderAt, entry IndexEntry) ([]byte, error) {
|
||||
payload := make([]byte, entry.Size)
|
||||
if _, err := r.ReadAt(payload, entry.Offset); err != nil {
|
||||
return nil, fmt.Errorf("read resource %q: %w", entry.Name, err)
|
||||
}
|
||||
return payload, nil
|
||||
}
|
||||
|
||||
// Read materialises a whole archive. Payloads are subslices of the buffer the
|
||||
// archive was read into, so nothing is copied twice: a caller must not mutate
|
||||
// Data. Callers that only need one resource at a time should use ReadIndex
|
||||
// instead, which never holds the archive at all.
|
||||
func Read(r io.Reader) (Archive, error) {
|
||||
data, err := io.ReadAll(r)
|
||||
if err != nil {
|
||||
return Archive{}, fmt.Errorf("read erf: %w", err)
|
||||
}
|
||||
if len(data) < headerSize {
|
||||
return Archive{}, fmt.Errorf("erf file too small: %d bytes", len(data))
|
||||
index, err := ReadIndex(bytes.NewReader(data), int64(len(data)))
|
||||
if err != nil {
|
||||
return Archive{}, err
|
||||
}
|
||||
|
||||
var hdr header
|
||||
if err := binary.Read(bytes.NewReader(data[:headerSize]), binary.LittleEndian, &hdr); err != nil {
|
||||
return Archive{}, fmt.Errorf("decode erf header: %w", err)
|
||||
}
|
||||
|
||||
keyStart := int(hdr.KeyListOffset)
|
||||
keyEnd := keyStart + int(hdr.EntryCount)*24
|
||||
if keyEnd > len(data) {
|
||||
return Archive{}, fmt.Errorf("erf key list exceeds file bounds")
|
||||
}
|
||||
keys := make([]keyEntry, hdr.EntryCount)
|
||||
if err := binary.Read(bytes.NewReader(data[keyStart:keyEnd]), binary.LittleEndian, &keys); err != nil {
|
||||
return Archive{}, fmt.Errorf("decode key list: %w", err)
|
||||
}
|
||||
|
||||
resourceStart := int(hdr.ResourceListOffset)
|
||||
resourceEnd := resourceStart + int(hdr.EntryCount)*8
|
||||
if resourceEnd > len(data) {
|
||||
return Archive{}, fmt.Errorf("erf resource list exceeds file bounds")
|
||||
}
|
||||
entries := make([]resourceEntry, hdr.EntryCount)
|
||||
if err := binary.Read(bytes.NewReader(data[resourceStart:resourceEnd]), binary.LittleEndian, &entries); err != nil {
|
||||
return Archive{}, fmt.Errorf("decode resource list: %w", err)
|
||||
}
|
||||
|
||||
resources := make([]Resource, 0, hdr.EntryCount)
|
||||
for index, key := range keys {
|
||||
entry := entries[index]
|
||||
start := int(entry.Offset)
|
||||
end := start + int(entry.Size)
|
||||
if end > len(data) {
|
||||
return Archive{}, fmt.Errorf("resource %d exceeds file bounds", index)
|
||||
}
|
||||
|
||||
resref := string(bytes.TrimRight(key.ResRef[:], "\x00"))
|
||||
payload := make([]byte, entry.Size)
|
||||
copy(payload, data[start:end])
|
||||
resources := make([]Resource, 0, len(index.Entries))
|
||||
for _, entry := range index.Entries {
|
||||
resources = append(resources, Resource{
|
||||
Name: resref,
|
||||
Type: key.ResourceType,
|
||||
Data: payload,
|
||||
Size: int64(entry.Size),
|
||||
Name: entry.Name,
|
||||
Type: entry.Type,
|
||||
Data: data[entry.Offset : entry.Offset+entry.Size],
|
||||
Size: entry.Size,
|
||||
})
|
||||
}
|
||||
|
||||
return Archive{
|
||||
FileType: string(hdr.FileType[:]),
|
||||
Version: string(hdr.Version[:]),
|
||||
FileType: index.FileType,
|
||||
Version: index.Version,
|
||||
Resources: resources,
|
||||
}, nil
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user