1 /* SPDX-License-Identifier: LGPL-2.1-or-later */
6 #if HAVE_VALGRIND_VALGRIND_H
7 # include <valgrind/valgrind.h>
10 #include "alloc-util.h"
11 #include "extract-word.h"
14 #include "logarithm.h"
15 #include "memory-util.h"
17 #include "process-util.h"
18 #include "random-util.h"
20 #include "siphash24.h"
21 #include "sort-util.h"
22 #include "string-util.h"
25 #if ENABLE_DEBUG_HASHMAP
30 * Implementation of hashmaps.
32 * - uses less RAM compared to closed addressing (chaining), because
33 * our entries are small (especially in Sets, which tend to contain
34 * the majority of entries in systemd).
35 * Collision resolution: Robin Hood
36 * - tends to equalize displacement of entries from their optimal buckets.
37 * Probe sequence: linear
38 * - though theoretically worse than random probing/uniform hashing/double
39 * hashing, it is good for cache locality.
42 * Celis, P. 1986. Robin Hood Hashing.
43 * Ph.D. Dissertation. University of Waterloo, Waterloo, Ont., Canada, Canada.
44 * https://cs.uwaterloo.ca/research/tr/1986/CS-86-14.pdf
45 * - The results are derived for random probing. Suggests deletion with
46 * tombstones and two mean-centered search methods. None of that works
47 * well for linear probing.
49 * Janson, S. 2005. Individual displacements for linear probing hashing with different insertion policies.
50 * ACM Trans. Algorithms 1, 2 (October 2005), 177-213.
51 * DOI=10.1145/1103963.1103964 http://doi.acm.org/10.1145/1103963.1103964
52 * http://www.math.uu.se/~svante/papers/sj157.pdf
53 * - Applies to Robin Hood with linear probing. Contains remarks on
54 * the unsuitability of mean-centered search with linear probing.
56 * Viola, A. 2005. Exact distribution of individual displacements in linear probing hashing.
57 * ACM Trans. Algorithms 1, 2 (October 2005), 214-242.
58 * DOI=10.1145/1103963.1103965 http://doi.acm.org/10.1145/1103963.1103965
59 * - Similar to Janson. Note that Viola writes about C_{m,n} (number of probes
60 * in a successful search), and Janson writes about displacement. C = d + 1.
62 * Goossaert, E. 2013. Robin Hood hashing: backward shift deletion.
63 * http://codecapsule.com/2013/11/17/robin-hood-hashing-backward-shift-deletion/
64 * - Explanation of backward shift deletion with pictures.
66 * Khuong, P. 2013. The Other Robin Hood Hashing.
67 * http://www.pvk.ca/Blog/2013/11/26/the-other-robin-hood-hashing/
68 * - Short summary of random vs. linear probing, and tombstones vs. backward shift.
72 * XXX Ideas for improvement:
73 * For unordered hashmaps, randomize iteration order, similarly to Perl:
74 * http://blog.booking.com/hardening-perls-hash-function.html
77 /* INV_KEEP_FREE = 1 / (1 - max_load_factor)
78 * e.g. 1 / (1 - 0.8) = 5 ... keep one fifth of the buckets free. */
79 #define INV_KEEP_FREE 5U
81 /* Fields common to entries of all hashmap/set types */
82 struct hashmap_base_entry
{
86 /* Entry types for specific hashmap/set types
87 * hashmap_base_entry must be at the beginning of each entry struct. */
89 struct plain_hashmap_entry
{
90 struct hashmap_base_entry b
;
94 struct ordered_hashmap_entry
{
95 struct plain_hashmap_entry p
;
96 unsigned iterate_next
, iterate_previous
;
100 struct hashmap_base_entry b
;
103 /* In several functions it is advantageous to have the hash table extended
104 * virtually by a couple of additional buckets. We reserve special index values
105 * for these "swap" buckets. */
106 #define _IDX_SWAP_BEGIN (UINT_MAX - 3)
107 #define IDX_PUT (_IDX_SWAP_BEGIN + 0)
108 #define IDX_TMP (_IDX_SWAP_BEGIN + 1)
109 #define _IDX_SWAP_END (_IDX_SWAP_BEGIN + 2)
111 #define IDX_FIRST (UINT_MAX - 1) /* special index for freshly initialized iterators */
112 #define IDX_NIL UINT_MAX /* special index value meaning "none" or "end" */
114 assert_cc(IDX_FIRST
== _IDX_SWAP_END
);
115 assert_cc(IDX_FIRST
== _IDX_ITERATOR_FIRST
);
117 /* Storage space for the "swap" buckets.
118 * All entry types can fit into an ordered_hashmap_entry. */
119 struct swap_entries
{
120 struct ordered_hashmap_entry e
[_IDX_SWAP_END
- _IDX_SWAP_BEGIN
];
123 /* Distance from Initial Bucket */
124 typedef uint8_t dib_raw_t
;
125 #define DIB_RAW_OVERFLOW ((dib_raw_t)0xfdU) /* indicates DIB value is greater than representable */
126 #define DIB_RAW_REHASH ((dib_raw_t)0xfeU) /* entry yet to be rehashed during in-place resize */
127 #define DIB_RAW_FREE ((dib_raw_t)0xffU) /* a free bucket */
128 #define DIB_RAW_INIT ((char)DIB_RAW_FREE) /* a byte to memset a DIB store with when initializing */
130 #define DIB_FREE UINT_MAX
132 #if ENABLE_DEBUG_HASHMAP
133 struct hashmap_debug_info
{
134 LIST_FIELDS(struct hashmap_debug_info
, debug_list
);
135 unsigned max_entries
; /* high watermark of n_entries */
137 /* fields to detect modification while iterating */
138 unsigned put_count
; /* counts puts into the hashmap */
139 unsigned rem_count
; /* counts removals from hashmap */
140 unsigned last_rem_idx
; /* remembers last removal index */
143 /* Tracks all existing hashmaps. Get at it from gdb. See sd_dump_hashmaps.py */
144 static LIST_HEAD(struct hashmap_debug_info
, hashmap_debug_list
);
145 static pthread_mutex_t hashmap_debug_list_mutex
= PTHREAD_MUTEX_INITIALIZER
;
150 HASHMAP_TYPE_ORDERED
,
155 struct _packed_ indirect_storage
{
156 void *storage
; /* where buckets and DIBs are stored */
157 uint8_t hash_key
[HASH_KEY_SIZE
]; /* hash key; changes during resize */
159 unsigned n_entries
; /* number of stored entries */
160 unsigned n_buckets
; /* number of buckets */
162 unsigned idx_lowest_entry
; /* Index below which all buckets are free.
163 Makes "while (hashmap_steal_first())" loops
164 O(n) instead of O(n^2) for unordered hashmaps. */
165 uint8_t _pad
[3]; /* padding for the whole HashmapBase */
166 /* The bitfields in HashmapBase complete the alignment of the whole thing. */
169 struct direct_storage
{
170 /* This gives us 39 bytes on 64-bit, or 35 bytes on 32-bit.
171 * That's room for 4 set_entries + 4 DIB bytes + 3 unused bytes on 64-bit,
172 * or 7 set_entries + 7 DIB bytes + 0 unused bytes on 32-bit. */
173 uint8_t storage
[sizeof(struct indirect_storage
)];
176 #define DIRECT_BUCKETS(entry_t) \
177 (sizeof(struct direct_storage) / (sizeof(entry_t) + sizeof(dib_raw_t)))
179 /* We should be able to store at least one entry directly. */
180 assert_cc(DIRECT_BUCKETS(struct ordered_hashmap_entry
) >= 1);
182 /* We have 3 bits for n_direct_entries. */
183 assert_cc(DIRECT_BUCKETS(struct set_entry
) < (1 << 3));
185 /* Hashmaps with directly stored entries all use this shared hash key.
186 * It's no big deal if the key is guessed, because there can be only
187 * a handful of directly stored entries in a hashmap. When a hashmap
188 * outgrows direct storage, it gets its own key for indirect storage. */
189 static uint8_t shared_hash_key
[HASH_KEY_SIZE
];
191 /* Fields that all hashmap/set types must have */
193 const struct hash_ops
*hash_ops
; /* hash and compare ops to use */
196 struct indirect_storage indirect
; /* if has_indirect */
197 struct direct_storage direct
; /* if !has_indirect */
200 enum HashmapType type
:2; /* HASHMAP_TYPE_* */
201 bool has_indirect
:1; /* whether indirect storage is used */
202 unsigned n_direct_entries
:3; /* Number of entries in direct storage.
203 * Only valid if !has_indirect. */
204 bool from_pool
:1; /* whether was allocated from mempool */
205 bool dirty
:1; /* whether dirtied since last iterated_cache_get() */
206 bool cached
:1; /* whether this hashmap is being cached */
208 #if ENABLE_DEBUG_HASHMAP
209 struct hashmap_debug_info debug
;
213 /* Specific hash types
214 * HashmapBase must be at the beginning of each hashmap struct. */
217 struct HashmapBase b
;
220 struct OrderedHashmap
{
221 struct HashmapBase b
;
222 unsigned iterate_list_head
, iterate_list_tail
;
226 struct HashmapBase b
;
229 typedef struct CacheMem
{
235 struct IteratedCache
{
236 HashmapBase
*hashmap
;
237 CacheMem keys
, values
;
240 DEFINE_MEMPOOL(hashmap_pool
, Hashmap
, 8);
241 DEFINE_MEMPOOL(ordered_hashmap_pool
, OrderedHashmap
, 8);
242 /* No need for a separate Set pool */
243 assert_cc(sizeof(Hashmap
) == sizeof(Set
));
245 struct hashmap_type_info
{
248 struct mempool
*mempool
;
249 unsigned n_direct_buckets
;
252 static _used_
const struct hashmap_type_info hashmap_type_info
[_HASHMAP_TYPE_MAX
] = {
253 [HASHMAP_TYPE_PLAIN
] = {
254 .head_size
= sizeof(Hashmap
),
255 .entry_size
= sizeof(struct plain_hashmap_entry
),
256 .mempool
= &hashmap_pool
,
257 .n_direct_buckets
= DIRECT_BUCKETS(struct plain_hashmap_entry
),
259 [HASHMAP_TYPE_ORDERED
] = {
260 .head_size
= sizeof(OrderedHashmap
),
261 .entry_size
= sizeof(struct ordered_hashmap_entry
),
262 .mempool
= &ordered_hashmap_pool
,
263 .n_direct_buckets
= DIRECT_BUCKETS(struct ordered_hashmap_entry
),
265 [HASHMAP_TYPE_SET
] = {
266 .head_size
= sizeof(Set
),
267 .entry_size
= sizeof(struct set_entry
),
268 .mempool
= &hashmap_pool
,
269 .n_direct_buckets
= DIRECT_BUCKETS(struct set_entry
),
273 void hashmap_trim_pools(void) {
276 /* The pool is only allocated by the main thread, but the memory can be passed to other
277 * threads. Let's clean up if we are the main thread and no other threads are live. */
279 /* We build our own is_main_thread() here, which doesn't use C11 TLS based caching of the
280 * result. That's because valgrind apparently doesn't like TLS to be used from a GCC destructor. */
281 if (getpid() != gettid())
282 return (void) log_debug("Not cleaning up memory pools, not in main thread.");
284 r
= get_process_threads(0);
286 return (void) log_debug_errno(r
, "Failed to determine number of threads, not cleaning up memory pools: %m");
288 return (void) log_debug("Not cleaning up memory pools, running in multi-threaded process.");
290 mempool_trim(&hashmap_pool
);
291 mempool_trim(&ordered_hashmap_pool
);
294 #if HAVE_VALGRIND_VALGRIND_H
295 _destructor_
static void cleanup_pools(void) {
296 /* Be nice to valgrind */
297 if (RUNNING_ON_VALGRIND
)
298 hashmap_trim_pools();
302 static unsigned n_buckets(HashmapBase
*h
) {
304 return h
->has_indirect
? h
->indirect
.n_buckets
305 : hashmap_type_info
[h
->type
].n_direct_buckets
;
308 static unsigned n_entries(HashmapBase
*h
) {
310 return h
->has_indirect
? h
->indirect
.n_entries
311 : h
->n_direct_entries
;
314 static void n_entries_inc(HashmapBase
*h
) {
318 h
->indirect
.n_entries
++;
320 h
->n_direct_entries
++;
323 static void n_entries_dec(HashmapBase
*h
) {
327 h
->indirect
.n_entries
--;
329 h
->n_direct_entries
--;
332 static void* storage_ptr(HashmapBase
*h
) {
334 return h
->has_indirect
? h
->indirect
.storage
338 static uint8_t* hash_key(HashmapBase
*h
) {
340 return h
->has_indirect
? h
->indirect
.hash_key
344 static unsigned base_bucket_hash(HashmapBase
*h
, const void *p
) {
345 struct siphash state
;
350 siphash24_init(&state
, hash_key(h
));
352 h
->hash_ops
->hash(p
, &state
);
354 hash
= siphash24_finalize(&state
);
356 return (unsigned) (hash
% n_buckets(h
));
358 #define bucket_hash(h, p) base_bucket_hash(HASHMAP_BASE(h), p)
360 static void base_set_dirty(HashmapBase
*h
) {
365 #define hashmap_set_dirty(h) base_set_dirty(HASHMAP_BASE(h))
367 static void get_hash_key(uint8_t hash_key
[HASH_KEY_SIZE
], bool reuse_is_ok
) {
368 static uint8_t current
[HASH_KEY_SIZE
];
369 static bool current_initialized
= false;
371 /* Returns a hash function key to use. In order to keep things
372 * fast we will not generate a new key each time we allocate a
373 * new hash table. Instead, we'll just reuse the most recently
374 * generated one, except if we never generated one or when we
375 * are rehashing an entire hash table because we reached a
378 if (!current_initialized
|| !reuse_is_ok
) {
379 random_bytes(current
, sizeof(current
));
380 current_initialized
= true;
383 memcpy(hash_key
, current
, sizeof(current
));
386 static struct hashmap_base_entry
* bucket_at(HashmapBase
*h
, unsigned idx
) {
388 return CAST_ALIGN_PTR(
389 struct hashmap_base_entry
,
390 (uint8_t *) storage_ptr(h
) + idx
* hashmap_type_info
[h
->type
].entry_size
);
393 static struct plain_hashmap_entry
* plain_bucket_at(Hashmap
*h
, unsigned idx
) {
394 return (struct plain_hashmap_entry
*) bucket_at(HASHMAP_BASE(h
), idx
);
397 static struct ordered_hashmap_entry
* ordered_bucket_at(OrderedHashmap
*h
, unsigned idx
) {
398 return (struct ordered_hashmap_entry
*) bucket_at(HASHMAP_BASE(h
), idx
);
401 static struct set_entry
*set_bucket_at(Set
*h
, unsigned idx
) {
402 return (struct set_entry
*) bucket_at(HASHMAP_BASE(h
), idx
);
405 static struct ordered_hashmap_entry
* bucket_at_swap(struct swap_entries
*swap
, unsigned idx
) {
407 return &swap
->e
[idx
- _IDX_SWAP_BEGIN
];
410 /* Returns a pointer to the bucket at index idx.
411 * Understands real indexes and swap indexes, hence "_virtual". */
412 static struct hashmap_base_entry
* bucket_at_virtual(HashmapBase
*h
, struct swap_entries
*swap
,
414 if (idx
< _IDX_SWAP_BEGIN
)
415 return bucket_at(h
, idx
);
417 if (idx
< _IDX_SWAP_END
)
418 return &bucket_at_swap(swap
, idx
)->p
.b
;
420 assert_not_reached();
423 static dib_raw_t
* dib_raw_ptr(HashmapBase
*h
) {
426 ((uint8_t*) storage_ptr(h
) + hashmap_type_info
[h
->type
].entry_size
* n_buckets(h
));
429 static unsigned bucket_distance(HashmapBase
*h
, unsigned idx
, unsigned from
) {
430 return idx
>= from
? idx
- from
431 : n_buckets(h
) + idx
- from
;
434 static unsigned bucket_calculate_dib(HashmapBase
*h
, unsigned idx
, dib_raw_t raw_dib
) {
435 unsigned initial_bucket
;
437 if (raw_dib
== DIB_RAW_FREE
)
440 if (_likely_(raw_dib
< DIB_RAW_OVERFLOW
))
444 * Having an overflow DIB value is very unlikely. The hash function
445 * would have to be bad. For example, in a table of size 2^24 filled
446 * to load factor 0.9 the maximum observed DIB is only about 60.
447 * In theory (assuming I used Maxima correctly), for an infinite size
448 * hash table with load factor 0.8 the probability of a given entry
449 * having DIB > 40 is 1.9e-8.
450 * This returns the correct DIB value by recomputing the hash value in
451 * the unlikely case. XXX Hitting this case could be a hint to rehash.
453 initial_bucket
= bucket_hash(h
, bucket_at(h
, idx
)->key
);
454 return bucket_distance(h
, idx
, initial_bucket
);
457 static void bucket_set_dib(HashmapBase
*h
, unsigned idx
, unsigned dib
) {
458 dib_raw_ptr(h
)[idx
] = dib
!= DIB_FREE
? MIN(dib
, DIB_RAW_OVERFLOW
) : DIB_RAW_FREE
;
461 static unsigned skip_free_buckets(HashmapBase
*h
, unsigned idx
) {
464 dibs
= dib_raw_ptr(h
);
466 for ( ; idx
< n_buckets(h
); idx
++)
467 if (dibs
[idx
] != DIB_RAW_FREE
)
473 static void bucket_mark_free(HashmapBase
*h
, unsigned idx
) {
476 memzero(bucket_at(h
, idx
), hashmap_type_info
[h
->type
].entry_size
);
477 bucket_set_dib(h
, idx
, DIB_FREE
);
480 static void bucket_move_entry(HashmapBase
*h
, struct swap_entries
*swap
,
481 unsigned from
, unsigned to
) {
482 struct hashmap_base_entry
*e_from
, *e_to
;
487 e_from
= bucket_at_virtual(h
, swap
, from
);
488 e_to
= bucket_at_virtual(h
, swap
, to
);
490 memcpy(e_to
, e_from
, hashmap_type_info
[h
->type
].entry_size
);
492 if (h
->type
== HASHMAP_TYPE_ORDERED
) {
493 OrderedHashmap
*lh
= (OrderedHashmap
*) h
;
494 struct ordered_hashmap_entry
*le
, *le_to
;
496 le_to
= (struct ordered_hashmap_entry
*) e_to
;
498 if (le_to
->iterate_next
!= IDX_NIL
) {
499 le
= (struct ordered_hashmap_entry
*)
500 bucket_at_virtual(h
, swap
, le_to
->iterate_next
);
501 le
->iterate_previous
= to
;
504 if (le_to
->iterate_previous
!= IDX_NIL
) {
505 le
= (struct ordered_hashmap_entry
*)
506 bucket_at_virtual(h
, swap
, le_to
->iterate_previous
);
507 le
->iterate_next
= to
;
510 if (lh
->iterate_list_head
== from
)
511 lh
->iterate_list_head
= to
;
512 if (lh
->iterate_list_tail
== from
)
513 lh
->iterate_list_tail
= to
;
517 static unsigned next_idx(HashmapBase
*h
, unsigned idx
) {
518 return (idx
+ 1U) % n_buckets(h
);
521 static unsigned prev_idx(HashmapBase
*h
, unsigned idx
) {
522 return (n_buckets(h
) + idx
- 1U) % n_buckets(h
);
525 static void* entry_value(HashmapBase
*h
, struct hashmap_base_entry
*e
) {
531 case HASHMAP_TYPE_PLAIN
:
532 case HASHMAP_TYPE_ORDERED
:
533 return ((struct plain_hashmap_entry
*)e
)->value
;
535 case HASHMAP_TYPE_SET
:
536 return (void*) e
->key
;
539 assert_not_reached();
543 static void base_remove_entry(HashmapBase
*h
, unsigned idx
) {
544 unsigned left
, right
, prev
, dib
;
545 dib_raw_t raw_dib
, *dibs
;
547 dibs
= dib_raw_ptr(h
);
548 assert(dibs
[idx
] != DIB_RAW_FREE
);
550 #if ENABLE_DEBUG_HASHMAP
552 h
->debug
.rem_count
++;
553 h
->debug
.last_rem_idx
= idx
;
557 /* Find the stop bucket ("right"). It is either free or has DIB == 0. */
558 for (right
= next_idx(h
, left
); ; right
= next_idx(h
, right
)) {
559 raw_dib
= dibs
[right
];
560 if (IN_SET(raw_dib
, 0, DIB_RAW_FREE
))
563 /* The buckets are not supposed to be all occupied and with DIB > 0.
564 * That would mean we could make everyone better off by shifting them
565 * backward. This scenario is impossible. */
566 assert(left
!= right
);
569 if (h
->type
== HASHMAP_TYPE_ORDERED
) {
570 OrderedHashmap
*lh
= (OrderedHashmap
*) h
;
571 struct ordered_hashmap_entry
*le
= ordered_bucket_at(lh
, idx
);
573 if (le
->iterate_next
!= IDX_NIL
)
574 ordered_bucket_at(lh
, le
->iterate_next
)->iterate_previous
= le
->iterate_previous
;
576 lh
->iterate_list_tail
= le
->iterate_previous
;
578 if (le
->iterate_previous
!= IDX_NIL
)
579 ordered_bucket_at(lh
, le
->iterate_previous
)->iterate_next
= le
->iterate_next
;
581 lh
->iterate_list_head
= le
->iterate_next
;
584 /* Now shift all buckets in the interval (left, right) one step backwards */
585 for (prev
= left
, left
= next_idx(h
, left
); left
!= right
;
586 prev
= left
, left
= next_idx(h
, left
)) {
587 dib
= bucket_calculate_dib(h
, left
, dibs
[left
]);
589 bucket_move_entry(h
, NULL
, left
, prev
);
590 bucket_set_dib(h
, prev
, dib
- 1);
593 bucket_mark_free(h
, prev
);
597 #define remove_entry(h, idx) base_remove_entry(HASHMAP_BASE(h), idx)
599 static unsigned hashmap_iterate_in_insertion_order(OrderedHashmap
*h
, Iterator
*i
) {
600 struct ordered_hashmap_entry
*e
;
606 if (i
->idx
== IDX_NIL
)
609 if (i
->idx
== IDX_FIRST
&& h
->iterate_list_head
== IDX_NIL
)
612 if (i
->idx
== IDX_FIRST
) {
613 idx
= h
->iterate_list_head
;
614 e
= ordered_bucket_at(h
, idx
);
617 e
= ordered_bucket_at(h
, idx
);
619 * We allow removing the current entry while iterating, but removal may cause
620 * a backward shift. The next entry may thus move one bucket to the left.
621 * To detect when it happens, we remember the key pointer of the entry we were
622 * going to iterate next. If it does not match, there was a backward shift.
624 if (e
->p
.b
.key
!= i
->next_key
) {
625 idx
= prev_idx(HASHMAP_BASE(h
), idx
);
626 e
= ordered_bucket_at(h
, idx
);
628 assert(e
->p
.b
.key
== i
->next_key
);
631 #if ENABLE_DEBUG_HASHMAP
635 if (e
->iterate_next
!= IDX_NIL
) {
636 struct ordered_hashmap_entry
*n
;
637 i
->idx
= e
->iterate_next
;
638 n
= ordered_bucket_at(h
, i
->idx
);
639 i
->next_key
= n
->p
.b
.key
;
650 static unsigned hashmap_iterate_in_internal_order(HashmapBase
*h
, Iterator
*i
) {
656 if (i
->idx
== IDX_NIL
)
659 if (i
->idx
== IDX_FIRST
) {
660 /* fast forward to the first occupied bucket */
661 if (h
->has_indirect
) {
662 i
->idx
= skip_free_buckets(h
, h
->indirect
.idx_lowest_entry
);
663 h
->indirect
.idx_lowest_entry
= i
->idx
;
665 i
->idx
= skip_free_buckets(h
, 0);
667 if (i
->idx
== IDX_NIL
)
670 struct hashmap_base_entry
*e
;
674 e
= bucket_at(h
, i
->idx
);
676 * We allow removing the current entry while iterating, but removal may cause
677 * a backward shift. The next entry may thus move one bucket to the left.
678 * To detect when it happens, we remember the key pointer of the entry we were
679 * going to iterate next. If it does not match, there was a backward shift.
681 if (e
->key
!= i
->next_key
)
682 e
= bucket_at(h
, --i
->idx
);
684 assert(e
->key
== i
->next_key
);
688 #if ENABLE_DEBUG_HASHMAP
692 i
->idx
= skip_free_buckets(h
, i
->idx
+ 1);
693 if (i
->idx
!= IDX_NIL
)
694 i
->next_key
= bucket_at(h
, i
->idx
)->key
;
705 static unsigned hashmap_iterate_entry(HashmapBase
*h
, Iterator
*i
) {
711 #if ENABLE_DEBUG_HASHMAP
712 if (i
->idx
== IDX_FIRST
) {
713 i
->put_count
= h
->debug
.put_count
;
714 i
->rem_count
= h
->debug
.rem_count
;
716 /* While iterating, must not add any new entries */
717 assert(i
->put_count
== h
->debug
.put_count
);
718 /* ... or remove entries other than the current one */
719 assert(i
->rem_count
== h
->debug
.rem_count
||
720 (i
->rem_count
== h
->debug
.rem_count
- 1 &&
721 i
->prev_idx
== h
->debug
.last_rem_idx
));
722 /* Reset our removals counter */
723 i
->rem_count
= h
->debug
.rem_count
;
727 return h
->type
== HASHMAP_TYPE_ORDERED
? hashmap_iterate_in_insertion_order((OrderedHashmap
*) h
, i
)
728 : hashmap_iterate_in_internal_order(h
, i
);
731 bool _hashmap_iterate(HashmapBase
*h
, Iterator
*i
, void **value
, const void **key
) {
732 struct hashmap_base_entry
*e
;
736 idx
= hashmap_iterate_entry(h
, i
);
737 if (idx
== IDX_NIL
) {
746 e
= bucket_at(h
, idx
);
747 data
= entry_value(h
, e
);
756 #define HASHMAP_FOREACH_IDX(idx, h, i) \
757 for ((i) = ITERATOR_FIRST, (idx) = hashmap_iterate_entry((h), &(i)); \
759 (idx) = hashmap_iterate_entry((h), &(i)))
761 IteratedCache
* _hashmap_iterated_cache_new(HashmapBase
*h
) {
762 IteratedCache
*cache
;
770 cache
= new0(IteratedCache
, 1);
780 static void reset_direct_storage(HashmapBase
*h
) {
781 const struct hashmap_type_info
*hi
= &hashmap_type_info
[h
->type
];
784 assert(!h
->has_indirect
);
786 p
= mempset(h
->direct
.storage
, 0, hi
->entry_size
* hi
->n_direct_buckets
);
787 memset(p
, DIB_RAW_INIT
, sizeof(dib_raw_t
) * hi
->n_direct_buckets
);
790 static void shared_hash_key_initialize(void) {
791 random_bytes(shared_hash_key
, sizeof(shared_hash_key
));
794 static struct HashmapBase
* hashmap_base_new(const struct hash_ops
*hash_ops
, enum HashmapType type
) {
796 const struct hashmap_type_info
*hi
= &hashmap_type_info
[type
];
798 bool use_pool
= mempool_enabled
&& mempool_enabled(); /* mempool_enabled is a weak symbol */
800 h
= use_pool
? mempool_alloc0_tile(hi
->mempool
) : malloc0(hi
->head_size
);
805 h
->from_pool
= use_pool
;
806 h
->hash_ops
= hash_ops
?: &trivial_hash_ops
;
808 if (type
== HASHMAP_TYPE_ORDERED
) {
809 OrderedHashmap
*lh
= (OrderedHashmap
*)h
;
810 lh
->iterate_list_head
= lh
->iterate_list_tail
= IDX_NIL
;
813 reset_direct_storage(h
);
815 static pthread_once_t once
= PTHREAD_ONCE_INIT
;
816 assert_se(pthread_once(&once
, shared_hash_key_initialize
) == 0);
818 #if ENABLE_DEBUG_HASHMAP
819 assert_se(pthread_mutex_lock(&hashmap_debug_list_mutex
) == 0);
820 LIST_PREPEND(debug_list
, hashmap_debug_list
, &h
->debug
);
821 assert_se(pthread_mutex_unlock(&hashmap_debug_list_mutex
) == 0);
827 Hashmap
*hashmap_new(const struct hash_ops
*hash_ops
) {
828 return (Hashmap
*) hashmap_base_new(hash_ops
, HASHMAP_TYPE_PLAIN
);
831 OrderedHashmap
*ordered_hashmap_new(const struct hash_ops
*hash_ops
) {
832 return (OrderedHashmap
*) hashmap_base_new(hash_ops
, HASHMAP_TYPE_ORDERED
);
835 Set
*set_new(const struct hash_ops
*hash_ops
) {
836 return (Set
*) hashmap_base_new(hash_ops
, HASHMAP_TYPE_SET
);
839 static int hashmap_base_ensure_allocated(HashmapBase
**h
, const struct hash_ops
*hash_ops
,
840 enum HashmapType type
) {
846 assert((*h
)->hash_ops
== (hash_ops
?: &trivial_hash_ops
));
850 q
= hashmap_base_new(hash_ops
, type
);
858 int hashmap_ensure_allocated(Hashmap
**h
, const struct hash_ops
*hash_ops
) {
859 return hashmap_base_ensure_allocated((HashmapBase
**)h
, hash_ops
, HASHMAP_TYPE_PLAIN
);
862 int ordered_hashmap_ensure_allocated(OrderedHashmap
**h
, const struct hash_ops
*hash_ops
) {
863 return hashmap_base_ensure_allocated((HashmapBase
**)h
, hash_ops
, HASHMAP_TYPE_ORDERED
);
866 int set_ensure_allocated(Set
**s
, const struct hash_ops
*hash_ops
) {
867 return hashmap_base_ensure_allocated((HashmapBase
**)s
, hash_ops
, HASHMAP_TYPE_SET
);
870 int hashmap_ensure_put(Hashmap
**h
, const struct hash_ops
*hash_ops
, const void *key
, void *value
) {
875 r
= hashmap_ensure_allocated(h
, hash_ops
);
879 return hashmap_put(*h
, key
, value
);
882 int ordered_hashmap_ensure_put(OrderedHashmap
**h
, const struct hash_ops
*hash_ops
, const void *key
, void *value
) {
887 r
= ordered_hashmap_ensure_allocated(h
, hash_ops
);
891 return ordered_hashmap_put(*h
, key
, value
);
894 int ordered_hashmap_ensure_replace(OrderedHashmap
**h
, const struct hash_ops
*hash_ops
, const void *key
, void *value
) {
899 r
= ordered_hashmap_ensure_allocated(h
, hash_ops
);
903 return ordered_hashmap_replace(*h
, key
, value
);
906 int hashmap_ensure_replace(Hashmap
**h
, const struct hash_ops
*hash_ops
, const void *key
, void *value
) {
911 r
= hashmap_ensure_allocated(h
, hash_ops
);
915 return hashmap_replace(*h
, key
, value
);
918 static void hashmap_free_no_clear(HashmapBase
*h
) {
919 assert(!h
->has_indirect
);
920 assert(h
->n_direct_entries
== 0);
922 #if ENABLE_DEBUG_HASHMAP
923 assert_se(pthread_mutex_lock(&hashmap_debug_list_mutex
) == 0);
924 LIST_REMOVE(debug_list
, hashmap_debug_list
, &h
->debug
);
925 assert_se(pthread_mutex_unlock(&hashmap_debug_list_mutex
) == 0);
929 /* Ensure that the object didn't get migrated between threads. */
930 assert_se(is_main_thread());
931 mempool_free_tile(hashmap_type_info
[h
->type
].mempool
, h
);
936 HashmapBase
* _hashmap_free(HashmapBase
*h
) {
939 hashmap_free_no_clear(h
);
945 void _hashmap_clear(HashmapBase
*h
) {
949 if (h
->hash_ops
->free_key
|| h
->hash_ops
->free_value
) {
951 /* If destructor calls are defined, let's destroy things defensively: let's take the item out of the
952 * hash table, and only then call the destructor functions. If these destructors then try to unregister
953 * themselves from our hash table a second time, the entry is already gone. */
955 while (_hashmap_size(h
) > 0) {
959 v
= _hashmap_first_key_and_value(h
, true, &k
);
961 if (h
->hash_ops
->free_key
)
962 h
->hash_ops
->free_key(k
);
964 if (h
->hash_ops
->free_value
)
965 h
->hash_ops
->free_value(v
);
969 if (h
->has_indirect
) {
970 free(h
->indirect
.storage
);
971 h
->has_indirect
= false;
974 h
->n_direct_entries
= 0;
975 reset_direct_storage(h
);
977 if (h
->type
== HASHMAP_TYPE_ORDERED
) {
978 OrderedHashmap
*lh
= (OrderedHashmap
*) h
;
979 lh
->iterate_list_head
= lh
->iterate_list_tail
= IDX_NIL
;
985 static int resize_buckets(HashmapBase
*h
, unsigned entries_add
);
988 * Finds an empty bucket to put an entry into, starting the scan at 'idx'.
989 * Performs Robin Hood swaps as it goes. The entry to put must be placed
990 * by the caller into swap slot IDX_PUT.
991 * If used for in-place resizing, may leave a displaced entry in swap slot
992 * IDX_PUT. Caller must rehash it next.
993 * Returns: true if it left a displaced entry to rehash next in IDX_PUT,
996 static bool hashmap_put_robin_hood(HashmapBase
*h
, unsigned idx
,
997 struct swap_entries
*swap
) {
998 dib_raw_t raw_dib
, *dibs
;
999 unsigned dib
, distance
;
1001 #if ENABLE_DEBUG_HASHMAP
1003 h
->debug
.put_count
++;
1006 dibs
= dib_raw_ptr(h
);
1008 for (distance
= 0; ; distance
++) {
1009 raw_dib
= dibs
[idx
];
1010 if (IN_SET(raw_dib
, DIB_RAW_FREE
, DIB_RAW_REHASH
)) {
1011 if (raw_dib
== DIB_RAW_REHASH
)
1012 bucket_move_entry(h
, swap
, idx
, IDX_TMP
);
1014 if (h
->has_indirect
&& h
->indirect
.idx_lowest_entry
> idx
)
1015 h
->indirect
.idx_lowest_entry
= idx
;
1017 bucket_set_dib(h
, idx
, distance
);
1018 bucket_move_entry(h
, swap
, IDX_PUT
, idx
);
1019 if (raw_dib
== DIB_RAW_REHASH
) {
1020 bucket_move_entry(h
, swap
, IDX_TMP
, IDX_PUT
);
1027 dib
= bucket_calculate_dib(h
, idx
, raw_dib
);
1029 if (dib
< distance
) {
1030 /* Found a wealthier entry. Go Robin Hood! */
1031 bucket_set_dib(h
, idx
, distance
);
1033 /* swap the entries */
1034 bucket_move_entry(h
, swap
, idx
, IDX_TMP
);
1035 bucket_move_entry(h
, swap
, IDX_PUT
, idx
);
1036 bucket_move_entry(h
, swap
, IDX_TMP
, IDX_PUT
);
1041 idx
= next_idx(h
, idx
);
1046 * Puts an entry into a hashmap, boldly - no check whether key already exists.
1047 * The caller must place the entry (only its key and value, not link indexes)
1048 * in swap slot IDX_PUT.
1049 * Caller must ensure: the key does not exist yet in the hashmap.
1050 * that resize is not needed if !may_resize.
1051 * Returns: 1 if entry was put successfully.
1052 * -ENOMEM if may_resize==true and resize failed with -ENOMEM.
1053 * Cannot return -ENOMEM if !may_resize.
1055 static int hashmap_base_put_boldly(HashmapBase
*h
, unsigned idx
,
1056 struct swap_entries
*swap
, bool may_resize
) {
1057 struct ordered_hashmap_entry
*new_entry
;
1060 assert(idx
< n_buckets(h
));
1062 new_entry
= bucket_at_swap(swap
, IDX_PUT
);
1065 r
= resize_buckets(h
, 1);
1069 idx
= bucket_hash(h
, new_entry
->p
.b
.key
);
1071 assert(n_entries(h
) < n_buckets(h
));
1073 if (h
->type
== HASHMAP_TYPE_ORDERED
) {
1074 OrderedHashmap
*lh
= (OrderedHashmap
*) h
;
1076 new_entry
->iterate_next
= IDX_NIL
;
1077 new_entry
->iterate_previous
= lh
->iterate_list_tail
;
1079 if (lh
->iterate_list_tail
!= IDX_NIL
) {
1080 struct ordered_hashmap_entry
*old_tail
;
1082 old_tail
= ordered_bucket_at(lh
, lh
->iterate_list_tail
);
1083 assert(old_tail
->iterate_next
== IDX_NIL
);
1084 old_tail
->iterate_next
= IDX_PUT
;
1087 lh
->iterate_list_tail
= IDX_PUT
;
1088 if (lh
->iterate_list_head
== IDX_NIL
)
1089 lh
->iterate_list_head
= IDX_PUT
;
1092 assert_se(hashmap_put_robin_hood(h
, idx
, swap
) == false);
1095 #if ENABLE_DEBUG_HASHMAP
1096 h
->debug
.max_entries
= MAX(h
->debug
.max_entries
, n_entries(h
));
1103 #define hashmap_put_boldly(h, idx, swap, may_resize) \
1104 hashmap_base_put_boldly(HASHMAP_BASE(h), idx, swap, may_resize)
1107 * Returns 0 if resize is not needed.
1108 * 1 if successfully resized.
1109 * -ENOMEM on allocation failure.
1111 static int resize_buckets(HashmapBase
*h
, unsigned entries_add
) {
1112 struct swap_entries swap
;
1114 dib_raw_t
*old_dibs
, *new_dibs
;
1115 const struct hashmap_type_info
*hi
;
1116 unsigned idx
, optimal_idx
;
1117 unsigned old_n_buckets
, new_n_buckets
, n_rehashed
, new_n_entries
;
1123 hi
= &hashmap_type_info
[h
->type
];
1124 new_n_entries
= n_entries(h
) + entries_add
;
1127 if (_unlikely_(new_n_entries
< entries_add
))
1130 /* For direct storage we allow 100% load, because it's tiny. */
1131 if (!h
->has_indirect
&& new_n_entries
<= hi
->n_direct_buckets
)
1135 * Load factor = n/m = 1 - (1/INV_KEEP_FREE).
1136 * From it follows: m = n + n/(INV_KEEP_FREE - 1)
1138 new_n_buckets
= new_n_entries
+ new_n_entries
/ (INV_KEEP_FREE
- 1);
1140 if (_unlikely_(new_n_buckets
< new_n_entries
))
1143 if (_unlikely_(new_n_buckets
> UINT_MAX
/ (hi
->entry_size
+ sizeof(dib_raw_t
))))
1146 old_n_buckets
= n_buckets(h
);
1148 if (_likely_(new_n_buckets
<= old_n_buckets
))
1151 new_shift
= log2u_round_up(MAX(
1152 new_n_buckets
* (hi
->entry_size
+ sizeof(dib_raw_t
)),
1153 2 * sizeof(struct direct_storage
)));
1155 /* Realloc storage (buckets and DIB array). */
1156 new_storage
= realloc(h
->has_indirect
? h
->indirect
.storage
: NULL
,
1161 /* Must upgrade direct to indirect storage. */
1162 if (!h
->has_indirect
) {
1163 memcpy(new_storage
, h
->direct
.storage
,
1164 old_n_buckets
* (hi
->entry_size
+ sizeof(dib_raw_t
)));
1165 h
->indirect
.n_entries
= h
->n_direct_entries
;
1166 h
->indirect
.idx_lowest_entry
= 0;
1167 h
->n_direct_entries
= 0;
1170 /* Get a new hash key. If we've just upgraded to indirect storage,
1171 * allow reusing a previously generated key. It's still a different key
1172 * from the shared one that we used for direct storage. */
1173 get_hash_key(h
->indirect
.hash_key
, !h
->has_indirect
);
1175 h
->has_indirect
= true;
1176 h
->indirect
.storage
= new_storage
;
1177 h
->indirect
.n_buckets
= (1U << new_shift
) /
1178 (hi
->entry_size
+ sizeof(dib_raw_t
));
1180 old_dibs
= (dib_raw_t
*)((uint8_t*) new_storage
+ hi
->entry_size
* old_n_buckets
);
1181 new_dibs
= dib_raw_ptr(h
);
1184 * Move the DIB array to the new place, replacing valid DIB values with
1185 * DIB_RAW_REHASH to indicate all of the used buckets need rehashing.
1186 * Note: Overlap is not possible, because we have at least doubled the
1187 * number of buckets and dib_raw_t is smaller than any entry type.
1189 for (idx
= 0; idx
< old_n_buckets
; idx
++) {
1190 assert(old_dibs
[idx
] != DIB_RAW_REHASH
);
1191 new_dibs
[idx
] = old_dibs
[idx
] == DIB_RAW_FREE
? DIB_RAW_FREE
1195 /* Zero the area of newly added entries (including the old DIB area) */
1196 memzero(bucket_at(h
, old_n_buckets
),
1197 (n_buckets(h
) - old_n_buckets
) * hi
->entry_size
);
1199 /* The upper half of the new DIB array needs initialization */
1200 memset(&new_dibs
[old_n_buckets
], DIB_RAW_INIT
,
1201 (n_buckets(h
) - old_n_buckets
) * sizeof(dib_raw_t
));
1203 /* Rehash entries that need it */
1205 for (idx
= 0; idx
< old_n_buckets
; idx
++) {
1206 if (new_dibs
[idx
] != DIB_RAW_REHASH
)
1209 optimal_idx
= bucket_hash(h
, bucket_at(h
, idx
)->key
);
1212 * Not much to do if by luck the entry hashes to its current
1213 * location. Just set its DIB.
1215 if (optimal_idx
== idx
) {
1221 new_dibs
[idx
] = DIB_RAW_FREE
;
1222 bucket_move_entry(h
, &swap
, idx
, IDX_PUT
);
1223 /* bucket_move_entry does not clear the source */
1224 memzero(bucket_at(h
, idx
), hi
->entry_size
);
1228 * Find the new bucket for the current entry. This may make
1229 * another entry homeless and load it into IDX_PUT.
1231 rehash_next
= hashmap_put_robin_hood(h
, optimal_idx
, &swap
);
1234 /* Did the current entry displace another one? */
1236 optimal_idx
= bucket_hash(h
, bucket_at_swap(&swap
, IDX_PUT
)->p
.b
.key
);
1237 } while (rehash_next
);
1240 assert_se(n_rehashed
== n_entries(h
));
1246 * Finds an entry with a matching key
1247 * Returns: index of the found entry, or IDX_NIL if not found.
1249 static unsigned base_bucket_scan(HashmapBase
*h
, unsigned idx
, const void *key
) {
1250 struct hashmap_base_entry
*e
;
1251 unsigned dib
, distance
;
1252 dib_raw_t
*dibs
= dib_raw_ptr(h
);
1254 assert(idx
< n_buckets(h
));
1256 for (distance
= 0; ; distance
++) {
1257 if (dibs
[idx
] == DIB_RAW_FREE
)
1260 dib
= bucket_calculate_dib(h
, idx
, dibs
[idx
]);
1264 if (dib
== distance
) {
1265 e
= bucket_at(h
, idx
);
1266 if (h
->hash_ops
->compare(e
->key
, key
) == 0)
1270 idx
= next_idx(h
, idx
);
1273 #define bucket_scan(h, idx, key) base_bucket_scan(HASHMAP_BASE(h), idx, key)
1275 int hashmap_put(Hashmap
*h
, const void *key
, void *value
) {
1276 struct swap_entries swap
;
1277 struct plain_hashmap_entry
*e
;
1282 hash
= bucket_hash(h
, key
);
1283 idx
= bucket_scan(h
, hash
, key
);
1284 if (idx
!= IDX_NIL
) {
1285 e
= plain_bucket_at(h
, idx
);
1286 if (e
->value
== value
)
1291 e
= &bucket_at_swap(&swap
, IDX_PUT
)->p
;
1294 return hashmap_put_boldly(h
, hash
, &swap
, true);
1297 int set_put(Set
*s
, const void *key
) {
1298 struct swap_entries swap
;
1299 struct hashmap_base_entry
*e
;
1304 hash
= bucket_hash(s
, key
);
1305 idx
= bucket_scan(s
, hash
, key
);
1309 e
= &bucket_at_swap(&swap
, IDX_PUT
)->p
.b
;
1311 return hashmap_put_boldly(s
, hash
, &swap
, true);
1314 int set_ensure_put(Set
**s
, const struct hash_ops
*hash_ops
, const void *key
) {
1319 r
= set_ensure_allocated(s
, hash_ops
);
1323 return set_put(*s
, key
);
1326 int set_ensure_consume(Set
**s
, const struct hash_ops
*hash_ops
, void *key
) {
1329 r
= set_ensure_put(s
, hash_ops
, key
);
1331 if (hash_ops
&& hash_ops
->free_key
)
1332 hash_ops
->free_key(key
);
1333 else if (hash_ops
&& hash_ops
->free_value
)
1334 /* Sets store their element in the key slot but may carry a value destructor. */
1335 hash_ops
->free_value(key
);
1343 int hashmap_replace(Hashmap
*h
, const void *key
, void *value
) {
1344 struct swap_entries swap
;
1345 struct plain_hashmap_entry
*e
;
1350 hash
= bucket_hash(h
, key
);
1351 idx
= bucket_scan(h
, hash
, key
);
1352 if (idx
!= IDX_NIL
) {
1353 e
= plain_bucket_at(h
, idx
);
1354 #if ENABLE_DEBUG_HASHMAP
1355 /* Although the key is equal, the key pointer may have changed,
1356 * and this would break our assumption for iterating. So count
1357 * this operation as incompatible with iteration. */
1358 if (e
->b
.key
!= key
) {
1359 h
->b
.debug
.put_count
++;
1360 h
->b
.debug
.rem_count
++;
1361 h
->b
.debug
.last_rem_idx
= idx
;
1366 hashmap_set_dirty(h
);
1371 e
= &bucket_at_swap(&swap
, IDX_PUT
)->p
;
1374 return hashmap_put_boldly(h
, hash
, &swap
, true);
1377 int hashmap_update(Hashmap
*h
, const void *key
, void *value
) {
1378 struct plain_hashmap_entry
*e
;
1383 hash
= bucket_hash(h
, key
);
1384 idx
= bucket_scan(h
, hash
, key
);
1388 e
= plain_bucket_at(h
, idx
);
1390 hashmap_set_dirty(h
);
1395 void* _hashmap_get(HashmapBase
*h
, const void *key
) {
1396 struct hashmap_base_entry
*e
;
1402 hash
= bucket_hash(h
, key
);
1403 idx
= bucket_scan(h
, hash
, key
);
1407 e
= bucket_at(h
, idx
);
1408 return entry_value(h
, e
);
1411 void* hashmap_get2(Hashmap
*h
, const void *key
, void **ret
) {
1412 struct plain_hashmap_entry
*e
;
1418 hash
= bucket_hash(h
, key
);
1419 idx
= bucket_scan(h
, hash
, key
);
1423 e
= plain_bucket_at(h
, idx
);
1425 *ret
= (void*) e
->b
.key
;
1430 bool _hashmap_contains(HashmapBase
*h
, const void *key
) {
1436 hash
= bucket_hash(h
, key
);
1437 return bucket_scan(h
, hash
, key
) != IDX_NIL
;
1440 void* _hashmap_remove(HashmapBase
*h
, const void *key
) {
1441 struct hashmap_base_entry
*e
;
1448 hash
= bucket_hash(h
, key
);
1449 idx
= bucket_scan(h
, hash
, key
);
1453 e
= bucket_at(h
, idx
);
1454 data
= entry_value(h
, e
);
1455 remove_entry(h
, idx
);
1460 void* hashmap_remove2(Hashmap
*h
, const void *key
, void **ret
) {
1461 struct plain_hashmap_entry
*e
;
1471 hash
= bucket_hash(h
, key
);
1472 idx
= bucket_scan(h
, hash
, key
);
1473 if (idx
== IDX_NIL
) {
1479 e
= plain_bucket_at(h
, idx
);
1482 *ret
= (void*) e
->b
.key
;
1484 remove_entry(h
, idx
);
1489 int hashmap_remove_and_put(Hashmap
*h
, const void *old_key
, const void *new_key
, void *value
) {
1490 struct swap_entries swap
;
1491 struct plain_hashmap_entry
*e
;
1492 unsigned old_hash
, new_hash
, idx
;
1497 old_hash
= bucket_hash(h
, old_key
);
1498 idx
= bucket_scan(h
, old_hash
, old_key
);
1502 new_hash
= bucket_hash(h
, new_key
);
1503 if (bucket_scan(h
, new_hash
, new_key
) != IDX_NIL
)
1506 remove_entry(h
, idx
);
1508 e
= &bucket_at_swap(&swap
, IDX_PUT
)->p
;
1511 assert_se(hashmap_put_boldly(h
, new_hash
, &swap
, false) == 1);
1516 int set_remove_and_put(Set
*s
, const void *old_key
, const void *new_key
) {
1517 struct swap_entries swap
;
1518 struct hashmap_base_entry
*e
;
1519 unsigned old_hash
, new_hash
, idx
;
1524 old_hash
= bucket_hash(s
, old_key
);
1525 idx
= bucket_scan(s
, old_hash
, old_key
);
1529 new_hash
= bucket_hash(s
, new_key
);
1530 if (bucket_scan(s
, new_hash
, new_key
) != IDX_NIL
)
1533 remove_entry(s
, idx
);
1535 e
= &bucket_at_swap(&swap
, IDX_PUT
)->p
.b
;
1537 assert_se(hashmap_put_boldly(s
, new_hash
, &swap
, false) == 1);
1542 int hashmap_remove_and_replace(Hashmap
*h
, const void *old_key
, const void *new_key
, void *value
) {
1543 struct swap_entries swap
;
1544 struct plain_hashmap_entry
*e
;
1545 unsigned old_hash
, new_hash
, idx_old
, idx_new
;
1550 old_hash
= bucket_hash(h
, old_key
);
1551 idx_old
= bucket_scan(h
, old_hash
, old_key
);
1552 if (idx_old
== IDX_NIL
)
1555 old_key
= bucket_at(HASHMAP_BASE(h
), idx_old
)->key
;
1557 new_hash
= bucket_hash(h
, new_key
);
1558 idx_new
= bucket_scan(h
, new_hash
, new_key
);
1559 if (idx_new
!= IDX_NIL
)
1560 if (idx_old
!= idx_new
) {
1561 remove_entry(h
, idx_new
);
1562 /* Compensate for a possible backward shift. */
1563 if (old_key
!= bucket_at(HASHMAP_BASE(h
), idx_old
)->key
)
1564 idx_old
= prev_idx(HASHMAP_BASE(h
), idx_old
);
1565 assert(old_key
== bucket_at(HASHMAP_BASE(h
), idx_old
)->key
);
1568 remove_entry(h
, idx_old
);
1570 e
= &bucket_at_swap(&swap
, IDX_PUT
)->p
;
1573 assert_se(hashmap_put_boldly(h
, new_hash
, &swap
, false) == 1);
1578 void* _hashmap_remove_value(HashmapBase
*h
, const void *key
, void *value
) {
1579 struct hashmap_base_entry
*e
;
1585 hash
= bucket_hash(h
, key
);
1586 idx
= bucket_scan(h
, hash
, key
);
1590 e
= bucket_at(h
, idx
);
1591 if (entry_value(h
, e
) != value
)
1594 remove_entry(h
, idx
);
1599 static unsigned find_first_entry(HashmapBase
*h
) {
1600 Iterator i
= ITERATOR_FIRST
;
1602 if (!h
|| !n_entries(h
))
1605 return hashmap_iterate_entry(h
, &i
);
1608 void* _hashmap_first_key_and_value(HashmapBase
*h
, bool remove
, void **ret_key
) {
1609 struct hashmap_base_entry
*e
;
1613 idx
= find_first_entry(h
);
1614 if (idx
== IDX_NIL
) {
1620 e
= bucket_at(h
, idx
);
1621 key
= (void*) e
->key
;
1622 data
= entry_value(h
, e
);
1625 remove_entry(h
, idx
);
1633 unsigned _hashmap_size(HashmapBase
*h
) {
1637 return n_entries(h
);
1640 unsigned _hashmap_buckets(HashmapBase
*h
) {
1644 return n_buckets(h
);
1647 int _hashmap_merge(Hashmap
*h
, Hashmap
*other
) {
1653 HASHMAP_FOREACH_IDX(idx
, HASHMAP_BASE(other
), i
) {
1654 struct plain_hashmap_entry
*pe
= plain_bucket_at(other
, idx
);
1657 r
= hashmap_put(h
, pe
->b
.key
, pe
->value
);
1658 if (r
< 0 && r
!= -EEXIST
)
1665 int set_merge(Set
*s
, Set
*other
) {
1671 HASHMAP_FOREACH_IDX(idx
, HASHMAP_BASE(other
), i
) {
1672 struct set_entry
*se
= set_bucket_at(other
, idx
);
1675 r
= set_put(s
, se
->b
.key
);
1683 int _hashmap_reserve(HashmapBase
*h
, unsigned entries_add
) {
1688 r
= resize_buckets(h
, entries_add
);
1696 * The same as hashmap_merge(), but every new item from other is moved to h.
1697 * Keys already in h are skipped and stay in other.
1698 * Returns: 0 on success.
1699 * -ENOMEM on alloc failure, in which case no move has been done.
1701 int _hashmap_move(HashmapBase
*h
, HashmapBase
*other
) {
1702 struct swap_entries swap
;
1703 struct hashmap_base_entry
*e
, *n
;
1713 assert(other
->type
== h
->type
);
1716 * This reserves buckets for the worst case, where none of other's
1717 * entries are yet present in h. This is preferable to risking
1718 * an allocation failure in the middle of the moving and having to
1719 * rollback or return a partial result.
1721 r
= resize_buckets(h
, n_entries(other
));
1725 HASHMAP_FOREACH_IDX(idx
, other
, i
) {
1728 e
= bucket_at(other
, idx
);
1729 h_hash
= bucket_hash(h
, e
->key
);
1730 if (bucket_scan(h
, h_hash
, e
->key
) != IDX_NIL
)
1733 n
= &bucket_at_swap(&swap
, IDX_PUT
)->p
.b
;
1735 if (h
->type
!= HASHMAP_TYPE_SET
)
1736 ((struct plain_hashmap_entry
*) n
)->value
=
1737 ((struct plain_hashmap_entry
*) e
)->value
;
1738 assert_se(hashmap_put_boldly(h
, h_hash
, &swap
, false) == 1);
1740 remove_entry(other
, idx
);
1746 int _hashmap_move_one(HashmapBase
*h
, HashmapBase
*other
, const void *key
) {
1747 struct swap_entries swap
;
1748 unsigned h_hash
, other_hash
, idx
;
1749 struct hashmap_base_entry
*e
, *n
;
1754 h_hash
= bucket_hash(h
, key
);
1755 if (bucket_scan(h
, h_hash
, key
) != IDX_NIL
)
1761 assert(other
->type
== h
->type
);
1763 other_hash
= bucket_hash(other
, key
);
1764 idx
= bucket_scan(other
, other_hash
, key
);
1768 e
= bucket_at(other
, idx
);
1770 n
= &bucket_at_swap(&swap
, IDX_PUT
)->p
.b
;
1772 if (h
->type
!= HASHMAP_TYPE_SET
)
1773 ((struct plain_hashmap_entry
*) n
)->value
=
1774 ((struct plain_hashmap_entry
*) e
)->value
;
1775 r
= hashmap_put_boldly(h
, h_hash
, &swap
, true);
1779 remove_entry(other
, idx
);
1783 HashmapBase
* _hashmap_copy(HashmapBase
*h
) {
1789 copy
= hashmap_base_new(h
->hash_ops
, h
->type
);
1794 case HASHMAP_TYPE_PLAIN
:
1795 case HASHMAP_TYPE_ORDERED
:
1796 r
= hashmap_merge((Hashmap
*)copy
, (Hashmap
*)h
);
1798 case HASHMAP_TYPE_SET
:
1799 r
= set_merge((Set
*)copy
, (Set
*)h
);
1802 assert_not_reached();
1806 return _hashmap_free(copy
);
1811 char** _hashmap_get_strv(HashmapBase
*h
) {
1817 return new0(char*, 1);
1819 sv
= new(char*, n_entries(h
)+1);
1824 HASHMAP_FOREACH_IDX(idx
, h
, i
)
1825 sv
[n
++] = entry_value(h
, bucket_at(h
, idx
));
1831 char** set_to_strv(Set
**s
) {
1834 /* This is similar to set_get_strv(), but invalidates the set on success. */
1836 char **v
= new(char*, set_size(*s
) + 1);
1840 for (char **p
= v
; (*p
= set_steal_first(*s
)); p
++)
1843 assert(set_isempty(*s
));
1848 void* ordered_hashmap_next(OrderedHashmap
*h
, const void *key
) {
1849 struct ordered_hashmap_entry
*e
;
1855 hash
= bucket_hash(h
, key
);
1856 idx
= bucket_scan(h
, hash
, key
);
1860 e
= ordered_bucket_at(h
, idx
);
1861 if (e
->iterate_next
== IDX_NIL
)
1863 return ordered_bucket_at(h
, e
->iterate_next
)->p
.value
;
1866 int set_consume(Set
*s
, void *value
) {
1872 r
= set_put(s
, value
);
1879 int hashmap_put_strdup_full(Hashmap
**h
, const struct hash_ops
*hash_ops
, const char *k
, const char *v
) {
1884 r
= hashmap_ensure_allocated(h
, hash_ops
);
1888 _cleanup_free_
char *kdup
= NULL
, *vdup
= NULL
;
1900 r
= hashmap_put(*h
, kdup
, vdup
);
1902 if (r
== -EEXIST
&& streq_ptr(v
, hashmap_get(*h
, kdup
)))
1907 /* 0 with non-null vdup would mean vdup is already in the hashmap, which cannot be */
1908 assert(vdup
== NULL
|| r
> 0);
1915 int set_put_strndup_full(Set
**s
, const struct hash_ops
*hash_ops
, const char *p
, size_t n
) {
1922 r
= set_ensure_allocated(s
, hash_ops
);
1926 if (n
== SIZE_MAX
) {
1927 if (set_contains(*s
, (char*) p
))
1936 return set_consume(*s
, c
);
1939 int set_put_strdupv_full(Set
**s
, const struct hash_ops
*hash_ops
, char **l
) {
1944 STRV_FOREACH(i
, l
) {
1945 r
= set_put_strndup_full(s
, hash_ops
, *i
, SIZE_MAX
);
1955 int set_put_strsplit(Set
*s
, const char *v
, const char *separators
, ExtractFlags flags
) {
1956 const char *p
= ASSERT_PTR(v
);
1964 r
= extract_first_word(&p
, &word
, separators
, flags
);
1968 r
= set_consume(s
, word
);
1974 /* expand the cachemem if needed, return true if newly (re)activated. */
1975 static int cachemem_maintain(CacheMem
*mem
, size_t size
) {
1978 if (!GREEDY_REALLOC(mem
->ptr
, size
)) {
1991 int iterated_cache_get(IteratedCache
*cache
, const void ***res_keys
, const void ***res_values
, unsigned *res_n_entries
) {
1992 bool sync_keys
= false, sync_values
= false;
1997 assert(cache
->hashmap
);
1999 size
= n_entries(cache
->hashmap
);
2002 r
= cachemem_maintain(&cache
->keys
, size
);
2008 cache
->keys
.active
= false;
2011 r
= cachemem_maintain(&cache
->values
, size
);
2017 cache
->values
.active
= false;
2019 if (cache
->hashmap
->dirty
) {
2020 if (cache
->keys
.active
)
2022 if (cache
->values
.active
)
2025 cache
->hashmap
->dirty
= false;
2028 if (sync_keys
|| sync_values
) {
2033 HASHMAP_FOREACH_IDX(idx
, cache
->hashmap
, iter
) {
2034 struct hashmap_base_entry
*e
;
2036 e
= bucket_at(cache
->hashmap
, idx
);
2039 cache
->keys
.ptr
[i
] = e
->key
;
2041 cache
->values
.ptr
[i
] = entry_value(cache
->hashmap
, e
);
2047 *res_keys
= cache
->keys
.ptr
;
2049 *res_values
= cache
->values
.ptr
;
2051 *res_n_entries
= size
;
2056 IteratedCache
* iterated_cache_free(IteratedCache
*cache
) {
2058 free(cache
->keys
.ptr
);
2059 free(cache
->values
.ptr
);
2062 return mfree(cache
);
2065 int set_strjoin(Set
*s
, const char *separator
, bool wrap_with_separator
, char **ret
) {
2066 _cleanup_free_
char *str
= NULL
;
2067 size_t separator_len
, len
= 0;
2073 if (set_isempty(s
)) {
2078 separator_len
= strlen_ptr(separator
);
2080 if (separator_len
== 0)
2081 wrap_with_separator
= false;
2083 first
= !wrap_with_separator
;
2085 SET_FOREACH(value
, s
) {
2086 size_t l
= strlen_ptr(value
);
2091 if (!GREEDY_REALLOC(str
, len
+ l
+ (first
? 0 : separator_len
) + (wrap_with_separator
? separator_len
: 0) + 1))
2094 if (separator_len
> 0 && !first
) {
2095 memcpy(str
+ len
, separator
, separator_len
);
2096 len
+= separator_len
;
2099 memcpy(str
+ len
, value
, l
);
2104 if (wrap_with_separator
) {
2105 memcpy(str
+ len
, separator
, separator_len
);
2106 len
+= separator_len
;
2111 *ret
= TAKE_PTR(str
);
2115 bool set_equal(Set
*a
, Set
*b
) {
2118 /* Checks whether each entry of 'a' is also in 'b' and vice versa, i.e. the two sets contain the same
2124 if (set_isempty(a
) && set_isempty(b
))
2127 if (set_size(a
) != set_size(b
)) /* Cheap check that hopefully catches a lot of inequality cases
2132 if (!set_contains(b
, p
))
2135 /* If we have the same hashops, then we don't need to check things backwards given we compared the
2136 * size and that all of a is in b. */
2137 if (a
->b
.hash_ops
== b
->b
.hash_ops
)
2141 if (!set_contains(a
, p
))
2147 static bool set_fnmatch_one(Set
*patterns
, const char *needle
) {
2152 /* Any failure of fnmatch() is treated as equivalent to FNM_NOMATCH, i.e. as non-matching pattern */
2154 SET_FOREACH(p
, patterns
)
2155 if (fnmatch(p
, needle
, 0) == 0)
2161 bool set_fnmatch(Set
*include_patterns
, Set
*exclude_patterns
, const char *needle
) {
2164 if (set_fnmatch_one(exclude_patterns
, needle
))
2167 if (set_isempty(include_patterns
))
2170 return set_fnmatch_one(include_patterns
, needle
);
2173 static int hashmap_entry_compare(
2174 struct hashmap_base_entry
* const *a
,
2175 struct hashmap_base_entry
* const *b
,
2176 compare_func_t compare
) {
2182 return compare((*a
)->key
, (*b
)->key
);
2185 static int _hashmap_dump_entries_sorted(
2189 _cleanup_free_
void **entries
= NULL
;
2197 if (_hashmap_size(h
) == 0) {
2203 /* We append one more element than needed so that the resulting array can be used as a strv. We
2204 * don't count this entry in the returned size. */
2205 entries
= new(void*, _hashmap_size(h
) + 1);
2209 HASHMAP_FOREACH_IDX(idx
, h
, iter
)
2210 entries
[n
++] = bucket_at(h
, idx
);
2212 assert(n
== _hashmap_size(h
));
2215 typesafe_qsort_r((struct hashmap_base_entry
**) entries
, n
,
2216 hashmap_entry_compare
, h
->hash_ops
->compare
);
2218 *ret
= TAKE_PTR(entries
);
2223 int _hashmap_dump_keys_sorted(HashmapBase
*h
, void ***ret
, size_t *ret_n
) {
2224 _cleanup_free_
void **entries
= NULL
;
2230 r
= _hashmap_dump_entries_sorted(h
, &entries
, &n
);
2234 /* Reuse the array. */
2235 FOREACH_ARRAY(e
, entries
, n
)
2236 *e
= (void*) (*(struct hashmap_base_entry
**) e
)->key
;
2238 *ret
= TAKE_PTR(entries
);
2244 int _hashmap_dump_sorted(HashmapBase
*h
, void ***ret
, size_t *ret_n
) {
2245 _cleanup_free_
void **entries
= NULL
;
2251 r
= _hashmap_dump_entries_sorted(h
, &entries
, &n
);
2255 /* Reuse the array. */
2256 FOREACH_ARRAY(e
, entries
, n
)
2257 *e
= entry_value(h
, *(struct hashmap_base_entry
**) e
);
2259 *ret
= TAKE_PTR(entries
);