reftable/block: reuse uncompressed blocks

The reftable backend stores reflog entries in a compressed format and thus needs to uncompress blocks before one can read records from it. For each reflog block we thus have to allocate an array that we can decompress the block contents into. This block is being discarded whenever the table iterator moves to the next block. Consequently, we reallocate a new array on every block, which is quite wasteful. Refactor the code to reuse the uncompressed block data when moving the block reader to a new block. This significantly reduces the number of allocations when iterating through many compressed blocks. The following measurements are done with `git reflog list` when listing 100k reflogs. Before: HEAP SUMMARY: in use at exit: 13,473 bytes in 122 blocks total heap usage: 45,755 allocs, 45,633 frees, 254,779,456 bytes allocated After: HEAP SUMMARY: in use at exit: 13,473 bytes in 122 blocks total heap usage: 23,028 allocs, 22,906 frees, 162,813,547 bytes allocated Signed-off-by: Patrick Steinhardt <ps@pks.im> Signed-off-by: Junio C Hamano <gitster@pobox.com>

Patrick Steinhardt committed Apr 8, 2024 at 14:16 UTC dd347bbce6053538d3332e6e8e499a3bebdbd251
3 files changed +26 -19
reftable/block.c
+6 -8
@@ -186,7 +186,6 @@ int block_reader_init(struct block_reader *br, struct reftable_block *block,
186 uint16_t restart_count = 0;
187 uint32_t restart_start = 0;
188 uint8_t *restart_bytes = NULL;
189 - uint8_t *uncompressed = NULL;
189
190 reftable_block_done(&br->block);
191
@@ -202,14 +201,15 @@ int block_reader_init(struct block_reader *br, struct reftable_block *block,
201 uLongf src_len = block->len - block_header_skip;
202
203 /* Log blocks specify the *uncompressed* size in their header. */
205 - REFTABLE_ALLOC_ARRAY(uncompressed, sz);
204 + REFTABLE_ALLOC_GROW(br->uncompressed_data, sz,
205 + br->uncompressed_cap);
206
207 /* Copy over the block header verbatim. It's not compressed. */
208 - memcpy(uncompressed, block->data, block_header_skip);
208 + memcpy(br->uncompressed_data, block->data, block_header_skip);
209
210 /* Uncompress */
211 if (Z_OK !=
212 - uncompress2(uncompressed + block_header_skip, &dst_len,
212 + uncompress2(br->uncompressed_data + block_header_skip, &dst_len,
213 block->data + block_header_skip, &src_len)) {
214 err = REFTABLE_ZLIB_ERROR;
215 goto done;
@@ -222,10 +222,8 @@ int block_reader_init(struct block_reader *br, struct reftable_block *block,
222
223 /* We're done with the input data. */
224 reftable_block_done(block);
225 - block->data = uncompressed;
226 - uncompressed = NULL;
225 + block->data = br->uncompressed_data;
226 block->len = sz;
228 - block->source = malloc_block_source();
227 full_block_size = src_len + block_header_skip;
228 } else if (full_block_size == 0) {
229 full_block_size = sz;
@@ -254,12 +252,12 @@ int block_reader_init(struct block_reader *br, struct reftable_block *block,
252 br->restart_bytes = restart_bytes;
253
254 done:
257 - reftable_free(uncompressed);
255 return err;
256 }
257
258 void block_reader_release(struct block_reader *br)
259 {
260 + reftable_free(br->uncompressed_data);
261 reftable_block_done(&br->block);
262 }
263
reftable/block.h
+4
@@ -66,6 +66,10 @@ struct block_reader {
66 struct reftable_block block;
67 int hash_size;
68
69 + /* Uncompressed data for log entries. */
70 + unsigned char *uncompressed_data;
71 + size_t uncompressed_cap;
72 +
73 /* size of the data, excluding restart data. */
74 uint32_t block_len;
75 uint8_t *restart_bytes;
reftable/reader.c
+16 -11
@@ -459,6 +459,8 @@ static int reader_seek_linear(struct table_iter *ti,
459 * we would not do a linear search there anymore.
460 */
461 memset(&next.br.block, 0, sizeof(next.br.block));
462 + next.br.uncompressed_data = NULL;
463 + next.br.uncompressed_cap = 0;
464
465 err = table_iter_next_block(&next);
466 if (err < 0)
@@ -599,25 +601,28 @@ static int reader_seek_internal(struct reftable_reader *r,
601 struct reftable_reader_offsets *offs =
602 reader_offsets_for(r, reftable_record_type(rec));
603 uint64_t idx = offs->index_offset;
602 - struct table_iter ti = TABLE_ITER_INIT;
603 - int err = 0;
604 + struct table_iter ti = TABLE_ITER_INIT, *p;
605 + int err;
606 +
607 if (idx > 0)
608 return reader_seek_indexed(r, it, rec);
609
610 err = reader_start(r, &ti, reftable_record_type(rec), 0);
611 if (err < 0)
609 - return err;
612 + goto out;
613 +
614 err = reader_seek_linear(&ti, rec);
615 if (err < 0)
612 - return err;
613 - else {
614 - struct table_iter *p =
615 - reftable_malloc(sizeof(struct table_iter));
616 - *p = ti;
617 - iterator_from_table_iter(it, p);
618 - }
616 + goto out;
617
620 - return 0;
618 + REFTABLE_ALLOC_ARRAY(p, 1);
619 + *p = ti;
620 + iterator_from_table_iter(it, p);
621 +
622 +out:
623 + if (err)
624 + table_iter_close(&ti);
625 + return err;
626 }
627
628 static int reader_seek(struct reftable_reader *r, struct reftable_iterator *it,