blob: 3eef0802a0cdb24251fac396eaaa9bbef6e24884 [file] [log] [blame]
Thomas Graf7e1e7762014-08-02 11:47:44 +02001/*
2 * Resizable, Scalable, Concurrent Hash Table
3 *
Herbert Xudc0ee262015-03-20 21:57:06 +11004 * Copyright (c) 2015 Herbert Xu <herbert@gondor.apana.org.au>
Thomas Grafb5e2c152015-03-24 20:42:19 +00005 * Copyright (c) 2014-2015 Thomas Graf <tgraf@suug.ch>
Thomas Graf7e1e7762014-08-02 11:47:44 +02006 * Copyright (c) 2008-2014 Patrick McHardy <kaber@trash.net>
7 *
Thomas Graf7e1e7762014-08-02 11:47:44 +02008 * Code partially derived from nft_hash
Herbert Xudc0ee262015-03-20 21:57:06 +11009 * Rewritten with rehash code from br_multicast plus single list
10 * pointer as suggested by Josh Triplett
Thomas Graf7e1e7762014-08-02 11:47:44 +020011 *
12 * This program is free software; you can redistribute it and/or modify
13 * it under the terms of the GNU General Public License version 2 as
14 * published by the Free Software Foundation.
15 */
16
17#ifndef _LINUX_RHASHTABLE_H
18#define _LINUX_RHASHTABLE_H
19
Herbert Xu07ee0722015-05-15 11:30:47 +080020#include <linux/atomic.h>
Herbert Xuf2dba9c2015-02-04 07:33:23 +110021#include <linux/compiler.h>
Herbert Xu3cf92222015-12-03 20:41:29 +080022#include <linux/err.h>
Herbert Xu6626af62015-03-20 18:18:45 -040023#include <linux/errno.h>
Herbert Xu31ccde22015-03-24 00:50:21 +110024#include <linux/jhash.h>
Thomas Graff89bd6f2015-01-02 23:00:21 +010025#include <linux/list_nulls.h>
Thomas Graf97defe12015-01-02 23:00:20 +010026#include <linux/workqueue.h>
Ying Xue86b35b62015-01-04 15:25:09 +080027#include <linux/mutex.h>
Herbert Xu02fd97c2015-03-20 21:57:00 +110028#include <linux/rcupdate.h>
Thomas Graf7e1e7762014-08-02 11:47:44 +020029
Thomas Graff89bd6f2015-01-02 23:00:21 +010030/*
31 * The end of the chain is marked with a special nulls marks which has
32 * the following format:
33 *
34 * +-------+-----------------------------------------------------+-+
35 * | Base | Hash |1|
36 * +-------+-----------------------------------------------------+-+
37 *
38 * Base (4 bits) : Reserved to distinguish between multiple tables.
39 * Specified via &struct rhashtable_params.nulls_base.
40 * Hash (27 bits): Full hash (unmasked) of first element added to bucket
41 * 1 (1 bit) : Nulls marker (always set)
42 *
43 * The remaining bits of the next pointer remain unused for now.
44 */
45#define RHT_BASE_BITS 4
46#define RHT_HASH_BITS 27
47#define RHT_BASE_SHIFT RHT_HASH_BITS
48
Herbert Xu02fd97c2015-03-20 21:57:00 +110049/* Base bits plus 1 bit for nulls marker */
50#define RHT_HASH_RESERVED_SPACE (RHT_BASE_BITS + 1)
51
Thomas Graf7e1e7762014-08-02 11:47:44 +020052struct rhash_head {
Thomas Graf5300fdc2014-08-13 16:38:29 +020053 struct rhash_head __rcu *next;
Thomas Graf7e1e7762014-08-02 11:47:44 +020054};
55
Thomas Graf97defe12015-01-02 23:00:20 +010056/**
57 * struct bucket_table - Table of hash buckets
58 * @size: Number of hash buckets
Herbert Xu63d512d2015-03-14 13:57:24 +110059 * @rehash: Current bucket being rehashed
Herbert Xu988dfbd2015-03-10 09:27:55 +110060 * @hash_rnd: Random seed to fold into hash
Thomas Graf97defe12015-01-02 23:00:20 +010061 * @locks_mask: Mask to apply before accessing locks[]
62 * @locks: Array of spinlocks protecting individual buckets
Herbert Xueddee5ba2015-03-14 13:57:20 +110063 * @walkers: List of active walkers
Herbert Xu9d901bc2015-03-14 13:57:23 +110064 * @rcu: RCU structure for freeing the table
Herbert Xuc4db8842015-03-14 13:57:25 +110065 * @future_tbl: Table under construction during rehashing
Thomas Graf97defe12015-01-02 23:00:20 +010066 * @buckets: size * hash buckets
67 */
Thomas Graf7e1e7762014-08-02 11:47:44 +020068struct bucket_table {
Herbert Xu63d512d2015-03-14 13:57:24 +110069 unsigned int size;
70 unsigned int rehash;
Herbert Xu988dfbd2015-03-10 09:27:55 +110071 u32 hash_rnd;
Eric Dumazetb9ebafb2015-02-20 06:48:57 -080072 unsigned int locks_mask;
73 spinlock_t *locks;
Herbert Xueddee5ba2015-03-14 13:57:20 +110074 struct list_head walkers;
Herbert Xu9d901bc2015-03-14 13:57:23 +110075 struct rcu_head rcu;
Eric Dumazetb9ebafb2015-02-20 06:48:57 -080076
Herbert Xuc4db8842015-03-14 13:57:25 +110077 struct bucket_table __rcu *future_tbl;
78
Eric Dumazetb9ebafb2015-02-20 06:48:57 -080079 struct rhash_head __rcu *buckets[] ____cacheline_aligned_in_smp;
Thomas Graf7e1e7762014-08-02 11:47:44 +020080};
81
Herbert Xu02fd97c2015-03-20 21:57:00 +110082/**
83 * struct rhashtable_compare_arg - Key for the function rhashtable_compare
84 * @ht: Hash table
85 * @key: Key to compare against
86 */
87struct rhashtable_compare_arg {
88 struct rhashtable *ht;
89 const void *key;
90};
91
Thomas Graf7e1e7762014-08-02 11:47:44 +020092typedef u32 (*rht_hashfn_t)(const void *data, u32 len, u32 seed);
Patrick McHardy49f7b332015-03-25 13:07:45 +000093typedef u32 (*rht_obj_hashfn_t)(const void *data, u32 len, u32 seed);
Herbert Xu02fd97c2015-03-20 21:57:00 +110094typedef int (*rht_obj_cmpfn_t)(struct rhashtable_compare_arg *arg,
95 const void *obj);
Thomas Graf7e1e7762014-08-02 11:47:44 +020096
97struct rhashtable;
98
99/**
100 * struct rhashtable_params - Hash table construction parameters
101 * @nelem_hint: Hint on number of elements, should be 75% of desired size
102 * @key_len: Length of key
103 * @key_offset: Offset of key in struct to be hashed
104 * @head_offset: Offset of rhash_head in struct to be hashed
Herbert Xu07ee0722015-05-15 11:30:47 +0800105 * @insecure_max_entries: Maximum number of entries (may be exceeded)
Herbert Xuc2e213c2015-03-18 20:01:16 +1100106 * @max_size: Maximum size while expanding
107 * @min_size: Minimum size while shrinking
Thomas Graff89bd6f2015-01-02 23:00:21 +0100108 * @nulls_base: Base value to generate nulls marker
Herbert Xuccd57b12015-03-24 00:50:28 +1100109 * @insecure_elasticity: Set to true to disable chain length checks
Thomas Grafb5e2c152015-03-24 20:42:19 +0000110 * @automatic_shrinking: Enable automatic shrinking of tables
Thomas Graf97defe12015-01-02 23:00:20 +0100111 * @locks_mul: Number of bucket locks to allocate per cpu (default: 128)
Herbert Xu31ccde22015-03-24 00:50:21 +1100112 * @hashfn: Hash function (default: jhash2 if !(key_len % 4), or jhash)
Thomas Graf7e1e7762014-08-02 11:47:44 +0200113 * @obj_hashfn: Function to hash object
Herbert Xu02fd97c2015-03-20 21:57:00 +1100114 * @obj_cmpfn: Function to compare key with object
Thomas Graf7e1e7762014-08-02 11:47:44 +0200115 */
116struct rhashtable_params {
117 size_t nelem_hint;
118 size_t key_len;
119 size_t key_offset;
120 size_t head_offset;
Herbert Xu07ee0722015-05-15 11:30:47 +0800121 unsigned int insecure_max_entries;
Herbert Xuc2e213c2015-03-18 20:01:16 +1100122 unsigned int max_size;
123 unsigned int min_size;
Thomas Graff89bd6f2015-01-02 23:00:21 +0100124 u32 nulls_base;
Herbert Xuccd57b12015-03-24 00:50:28 +1100125 bool insecure_elasticity;
Thomas Grafb5e2c152015-03-24 20:42:19 +0000126 bool automatic_shrinking;
Thomas Graf97defe12015-01-02 23:00:20 +0100127 size_t locks_mul;
Thomas Graf7e1e7762014-08-02 11:47:44 +0200128 rht_hashfn_t hashfn;
129 rht_obj_hashfn_t obj_hashfn;
Herbert Xu02fd97c2015-03-20 21:57:00 +1100130 rht_obj_cmpfn_t obj_cmpfn;
Thomas Graf7e1e7762014-08-02 11:47:44 +0200131};
132
133/**
134 * struct rhashtable - Hash table handle
135 * @tbl: Bucket table
136 * @nelems: Number of elements in table
Herbert Xu31ccde22015-03-24 00:50:21 +1100137 * @key_len: Key length for hashfn
Herbert Xuccd57b12015-03-24 00:50:28 +1100138 * @elasticity: Maximum chain length before rehash
Thomas Graf7e1e7762014-08-02 11:47:44 +0200139 * @p: Configuration parameters
Thomas Graf97defe12015-01-02 23:00:20 +0100140 * @run_work: Deferred worker to expand/shrink asynchronously
141 * @mutex: Mutex to protect current/future table swapping
Herbert Xuba7c95e2015-03-24 09:53:17 +1100142 * @lock: Spin lock to protect walker list
Thomas Graf7e1e7762014-08-02 11:47:44 +0200143 */
144struct rhashtable {
145 struct bucket_table __rcu *tbl;
Thomas Graf97defe12015-01-02 23:00:20 +0100146 atomic_t nelems;
Herbert Xu31ccde22015-03-24 00:50:21 +1100147 unsigned int key_len;
Herbert Xuccd57b12015-03-24 00:50:28 +1100148 unsigned int elasticity;
Thomas Graf7e1e7762014-08-02 11:47:44 +0200149 struct rhashtable_params p;
Ying Xue57699a42015-01-16 11:13:09 +0800150 struct work_struct run_work;
Thomas Graf97defe12015-01-02 23:00:20 +0100151 struct mutex mutex;
Herbert Xuba7c95e2015-03-24 09:53:17 +1100152 spinlock_t lock;
Thomas Graf7e1e7762014-08-02 11:47:44 +0200153};
154
Herbert Xuf2dba9c2015-02-04 07:33:23 +1100155/**
156 * struct rhashtable_walker - Hash table walker
157 * @list: List entry on list of walkers
Herbert Xueddee5ba2015-03-14 13:57:20 +1100158 * @tbl: The table that we were walking over
Herbert Xuf2dba9c2015-02-04 07:33:23 +1100159 */
160struct rhashtable_walker {
161 struct list_head list;
Herbert Xueddee5ba2015-03-14 13:57:20 +1100162 struct bucket_table *tbl;
Herbert Xuf2dba9c2015-02-04 07:33:23 +1100163};
164
165/**
166 * struct rhashtable_iter - Hash table iterator, fits into netlink cb
167 * @ht: Table to iterate through
168 * @p: Current pointer
169 * @walker: Associated rhashtable walker
170 * @slot: Current slot
171 * @skip: Number of entries to skip in slot
172 */
173struct rhashtable_iter {
174 struct rhashtable *ht;
175 struct rhash_head *p;
176 struct rhashtable_walker *walker;
177 unsigned int slot;
178 unsigned int skip;
179};
180
Thomas Graff89bd6f2015-01-02 23:00:21 +0100181static inline unsigned long rht_marker(const struct rhashtable *ht, u32 hash)
182{
183 return NULLS_MARKER(ht->p.nulls_base + hash);
184}
185
186#define INIT_RHT_NULLS_HEAD(ptr, ht, hash) \
187 ((ptr) = (typeof(ptr)) rht_marker(ht, hash))
188
189static inline bool rht_is_a_nulls(const struct rhash_head *ptr)
190{
191 return ((unsigned long) ptr & 1);
192}
193
194static inline unsigned long rht_get_nulls_value(const struct rhash_head *ptr)
195{
196 return ((unsigned long) ptr) >> 1;
197}
198
Herbert Xu02fd97c2015-03-20 21:57:00 +1100199static inline void *rht_obj(const struct rhashtable *ht,
200 const struct rhash_head *he)
201{
202 return (char *)he - ht->p.head_offset;
203}
204
205static inline unsigned int rht_bucket_index(const struct bucket_table *tbl,
206 unsigned int hash)
207{
208 return (hash >> RHT_HASH_RESERVED_SPACE) & (tbl->size - 1);
209}
210
211static inline unsigned int rht_key_hashfn(
212 struct rhashtable *ht, const struct bucket_table *tbl,
213 const void *key, const struct rhashtable_params params)
214{
Thomas Graf299e5c32015-03-24 14:18:17 +0100215 unsigned int hash;
Herbert Xude91b252015-03-24 00:50:20 +1100216
Herbert Xu31ccde22015-03-24 00:50:21 +1100217 /* params must be equal to ht->p if it isn't constant. */
218 if (!__builtin_constant_p(params.key_len))
219 hash = ht->p.hashfn(key, ht->key_len, tbl->hash_rnd);
220 else if (params.key_len) {
Thomas Graf299e5c32015-03-24 14:18:17 +0100221 unsigned int key_len = params.key_len;
Herbert Xu31ccde22015-03-24 00:50:21 +1100222
223 if (params.hashfn)
224 hash = params.hashfn(key, key_len, tbl->hash_rnd);
225 else if (key_len & (sizeof(u32) - 1))
226 hash = jhash(key, key_len, tbl->hash_rnd);
227 else
228 hash = jhash2(key, key_len / sizeof(u32),
229 tbl->hash_rnd);
230 } else {
Thomas Graf299e5c32015-03-24 14:18:17 +0100231 unsigned int key_len = ht->p.key_len;
Herbert Xu31ccde22015-03-24 00:50:21 +1100232
233 if (params.hashfn)
234 hash = params.hashfn(key, key_len, tbl->hash_rnd);
235 else
236 hash = jhash(key, key_len, tbl->hash_rnd);
237 }
238
239 return rht_bucket_index(tbl, hash);
Herbert Xu02fd97c2015-03-20 21:57:00 +1100240}
241
242static inline unsigned int rht_head_hashfn(
243 struct rhashtable *ht, const struct bucket_table *tbl,
244 const struct rhash_head *he, const struct rhashtable_params params)
245{
246 const char *ptr = rht_obj(ht, he);
247
248 return likely(params.obj_hashfn) ?
Patrick McHardy49f7b332015-03-25 13:07:45 +0000249 rht_bucket_index(tbl, params.obj_hashfn(ptr, params.key_len ?:
250 ht->p.key_len,
251 tbl->hash_rnd)) :
Herbert Xu02fd97c2015-03-20 21:57:00 +1100252 rht_key_hashfn(ht, tbl, ptr + params.key_offset, params);
253}
254
255/**
256 * rht_grow_above_75 - returns true if nelems > 0.75 * table-size
257 * @ht: hash table
258 * @tbl: current table
259 */
260static inline bool rht_grow_above_75(const struct rhashtable *ht,
261 const struct bucket_table *tbl)
262{
263 /* Expand table when exceeding 75% load */
264 return atomic_read(&ht->nelems) > (tbl->size / 4 * 3) &&
265 (!ht->p.max_size || tbl->size < ht->p.max_size);
266}
267
268/**
269 * rht_shrink_below_30 - returns true if nelems < 0.3 * table-size
270 * @ht: hash table
271 * @tbl: current table
272 */
273static inline bool rht_shrink_below_30(const struct rhashtable *ht,
274 const struct bucket_table *tbl)
275{
276 /* Shrink table beneath 30% load */
277 return atomic_read(&ht->nelems) < (tbl->size * 3 / 10) &&
278 tbl->size > ht->p.min_size;
279}
280
Herbert Xuccd57b12015-03-24 00:50:28 +1100281/**
282 * rht_grow_above_100 - returns true if nelems > table-size
283 * @ht: hash table
284 * @tbl: current table
285 */
286static inline bool rht_grow_above_100(const struct rhashtable *ht,
287 const struct bucket_table *tbl)
288{
Johannes Berg1d8dc3d2015-04-23 16:38:43 +0200289 return atomic_read(&ht->nelems) > tbl->size &&
290 (!ht->p.max_size || tbl->size < ht->p.max_size);
Herbert Xuccd57b12015-03-24 00:50:28 +1100291}
292
Herbert Xu07ee0722015-05-15 11:30:47 +0800293/**
294 * rht_grow_above_max - returns true if table is above maximum
295 * @ht: hash table
296 * @tbl: current table
297 */
298static inline bool rht_grow_above_max(const struct rhashtable *ht,
299 const struct bucket_table *tbl)
300{
301 return ht->p.insecure_max_entries &&
302 atomic_read(&ht->nelems) >= ht->p.insecure_max_entries;
303}
304
Herbert Xu02fd97c2015-03-20 21:57:00 +1100305/* The bucket lock is selected based on the hash and protects mutations
306 * on a group of hash buckets.
307 *
308 * A maximum of tbl->size/2 bucket locks is allocated. This ensures that
309 * a single lock always covers both buckets which may both contains
310 * entries which link to the same bucket of the old table during resizing.
311 * This allows to simplify the locking as locking the bucket in both
312 * tables during resize always guarantee protection.
313 *
314 * IMPORTANT: When holding the bucket lock of both the old and new table
315 * during expansions and shrinking, the old bucket lock must always be
316 * acquired first.
317 */
318static inline spinlock_t *rht_bucket_lock(const struct bucket_table *tbl,
319 unsigned int hash)
320{
321 return &tbl->locks[hash & tbl->locks_mask];
322}
323
Thomas Graf7e1e7762014-08-02 11:47:44 +0200324#ifdef CONFIG_PROVE_LOCKING
Thomas Graf97defe12015-01-02 23:00:20 +0100325int lockdep_rht_mutex_is_held(struct rhashtable *ht);
Thomas Graf88d6ed12015-01-02 23:00:16 +0100326int lockdep_rht_bucket_is_held(const struct bucket_table *tbl, u32 hash);
Thomas Graf7e1e7762014-08-02 11:47:44 +0200327#else
Thomas Graf97defe12015-01-02 23:00:20 +0100328static inline int lockdep_rht_mutex_is_held(struct rhashtable *ht)
Thomas Graf7e1e7762014-08-02 11:47:44 +0200329{
330 return 1;
331}
Thomas Graf88d6ed12015-01-02 23:00:16 +0100332
333static inline int lockdep_rht_bucket_is_held(const struct bucket_table *tbl,
334 u32 hash)
335{
336 return 1;
337}
Thomas Graf7e1e7762014-08-02 11:47:44 +0200338#endif /* CONFIG_PROVE_LOCKING */
339
Herbert Xu488fb86e2015-03-20 21:56:59 +1100340int rhashtable_init(struct rhashtable *ht,
341 const struct rhashtable_params *params);
Thomas Graf7e1e7762014-08-02 11:47:44 +0200342
Herbert Xu3cf92222015-12-03 20:41:29 +0800343struct bucket_table *rhashtable_insert_slow(struct rhashtable *ht,
344 const void *key,
345 struct rhash_head *obj,
346 struct bucket_table *old_tbl);
347int rhashtable_insert_rehash(struct rhashtable *ht, struct bucket_table *tbl);
Thomas Graf7e1e7762014-08-02 11:47:44 +0200348
Bob Copeland8f6fd832016-03-02 10:09:19 -0500349int rhashtable_walk_init(struct rhashtable *ht, struct rhashtable_iter *iter,
350 gfp_t gfp);
Herbert Xuf2dba9c2015-02-04 07:33:23 +1100351void rhashtable_walk_exit(struct rhashtable_iter *iter);
352int rhashtable_walk_start(struct rhashtable_iter *iter) __acquires(RCU);
353void *rhashtable_walk_next(struct rhashtable_iter *iter);
354void rhashtable_walk_stop(struct rhashtable_iter *iter) __releases(RCU);
355
Thomas Graf6b6f3022015-03-24 14:18:20 +0100356void rhashtable_free_and_destroy(struct rhashtable *ht,
357 void (*free_fn)(void *ptr, void *arg),
358 void *arg);
Thomas Graf97defe12015-01-02 23:00:20 +0100359void rhashtable_destroy(struct rhashtable *ht);
Thomas Graf7e1e7762014-08-02 11:47:44 +0200360
361#define rht_dereference(p, ht) \
362 rcu_dereference_protected(p, lockdep_rht_mutex_is_held(ht))
363
364#define rht_dereference_rcu(p, ht) \
365 rcu_dereference_check(p, lockdep_rht_mutex_is_held(ht))
366
Thomas Graf88d6ed12015-01-02 23:00:16 +0100367#define rht_dereference_bucket(p, tbl, hash) \
368 rcu_dereference_protected(p, lockdep_rht_bucket_is_held(tbl, hash))
Thomas Graf7e1e7762014-08-02 11:47:44 +0200369
Thomas Graf88d6ed12015-01-02 23:00:16 +0100370#define rht_dereference_bucket_rcu(p, tbl, hash) \
371 rcu_dereference_check(p, lockdep_rht_bucket_is_held(tbl, hash))
372
373#define rht_entry(tpos, pos, member) \
374 ({ tpos = container_of(pos, typeof(*tpos), member); 1; })
375
376/**
377 * rht_for_each_continue - continue iterating over hash chain
378 * @pos: the &struct rhash_head to use as a loop cursor.
379 * @head: the previous &struct rhash_head to continue from
380 * @tbl: the &struct bucket_table
381 * @hash: the hash value / bucket index
382 */
383#define rht_for_each_continue(pos, head, tbl, hash) \
384 for (pos = rht_dereference_bucket(head, tbl, hash); \
Thomas Graff89bd6f2015-01-02 23:00:21 +0100385 !rht_is_a_nulls(pos); \
Thomas Graf88d6ed12015-01-02 23:00:16 +0100386 pos = rht_dereference_bucket((pos)->next, tbl, hash))
Thomas Graf7e1e7762014-08-02 11:47:44 +0200387
388/**
389 * rht_for_each - iterate over hash chain
Thomas Graf88d6ed12015-01-02 23:00:16 +0100390 * @pos: the &struct rhash_head to use as a loop cursor.
391 * @tbl: the &struct bucket_table
392 * @hash: the hash value / bucket index
Thomas Graf7e1e7762014-08-02 11:47:44 +0200393 */
Thomas Graf88d6ed12015-01-02 23:00:16 +0100394#define rht_for_each(pos, tbl, hash) \
395 rht_for_each_continue(pos, (tbl)->buckets[hash], tbl, hash)
396
397/**
398 * rht_for_each_entry_continue - continue iterating over hash chain
399 * @tpos: the type * to use as a loop cursor.
400 * @pos: the &struct rhash_head to use as a loop cursor.
401 * @head: the previous &struct rhash_head to continue from
402 * @tbl: the &struct bucket_table
403 * @hash: the hash value / bucket index
404 * @member: name of the &struct rhash_head within the hashable struct.
405 */
406#define rht_for_each_entry_continue(tpos, pos, head, tbl, hash, member) \
407 for (pos = rht_dereference_bucket(head, tbl, hash); \
Thomas Graff89bd6f2015-01-02 23:00:21 +0100408 (!rht_is_a_nulls(pos)) && rht_entry(tpos, pos, member); \
Thomas Graf88d6ed12015-01-02 23:00:16 +0100409 pos = rht_dereference_bucket((pos)->next, tbl, hash))
Thomas Graf7e1e7762014-08-02 11:47:44 +0200410
411/**
412 * rht_for_each_entry - iterate over hash chain of given type
Thomas Graf88d6ed12015-01-02 23:00:16 +0100413 * @tpos: the type * to use as a loop cursor.
414 * @pos: the &struct rhash_head to use as a loop cursor.
415 * @tbl: the &struct bucket_table
416 * @hash: the hash value / bucket index
417 * @member: name of the &struct rhash_head within the hashable struct.
Thomas Graf7e1e7762014-08-02 11:47:44 +0200418 */
Thomas Graf88d6ed12015-01-02 23:00:16 +0100419#define rht_for_each_entry(tpos, pos, tbl, hash, member) \
420 rht_for_each_entry_continue(tpos, pos, (tbl)->buckets[hash], \
421 tbl, hash, member)
Thomas Graf7e1e7762014-08-02 11:47:44 +0200422
423/**
424 * rht_for_each_entry_safe - safely iterate over hash chain of given type
Thomas Graf88d6ed12015-01-02 23:00:16 +0100425 * @tpos: the type * to use as a loop cursor.
426 * @pos: the &struct rhash_head to use as a loop cursor.
427 * @next: the &struct rhash_head to use as next in loop cursor.
428 * @tbl: the &struct bucket_table
429 * @hash: the hash value / bucket index
430 * @member: name of the &struct rhash_head within the hashable struct.
Thomas Graf7e1e7762014-08-02 11:47:44 +0200431 *
432 * This hash chain list-traversal primitive allows for the looped code to
433 * remove the loop cursor from the list.
434 */
Thomas Graf88d6ed12015-01-02 23:00:16 +0100435#define rht_for_each_entry_safe(tpos, pos, next, tbl, hash, member) \
436 for (pos = rht_dereference_bucket((tbl)->buckets[hash], tbl, hash), \
Thomas Graff89bd6f2015-01-02 23:00:21 +0100437 next = !rht_is_a_nulls(pos) ? \
438 rht_dereference_bucket(pos->next, tbl, hash) : NULL; \
439 (!rht_is_a_nulls(pos)) && rht_entry(tpos, pos, member); \
Patrick McHardy607954b2015-01-21 11:12:13 +0000440 pos = next, \
441 next = !rht_is_a_nulls(pos) ? \
442 rht_dereference_bucket(pos->next, tbl, hash) : NULL)
Thomas Graf88d6ed12015-01-02 23:00:16 +0100443
444/**
445 * rht_for_each_rcu_continue - continue iterating over rcu hash chain
446 * @pos: the &struct rhash_head to use as a loop cursor.
447 * @head: the previous &struct rhash_head to continue from
448 * @tbl: the &struct bucket_table
449 * @hash: the hash value / bucket index
450 *
451 * This hash chain list-traversal primitive may safely run concurrently with
452 * the _rcu mutation primitives such as rhashtable_insert() as long as the
453 * traversal is guarded by rcu_read_lock().
454 */
455#define rht_for_each_rcu_continue(pos, head, tbl, hash) \
456 for (({barrier(); }), \
457 pos = rht_dereference_bucket_rcu(head, tbl, hash); \
Thomas Graff89bd6f2015-01-02 23:00:21 +0100458 !rht_is_a_nulls(pos); \
Thomas Graf88d6ed12015-01-02 23:00:16 +0100459 pos = rcu_dereference_raw(pos->next))
Thomas Graf7e1e7762014-08-02 11:47:44 +0200460
461/**
462 * rht_for_each_rcu - iterate over rcu hash chain
Thomas Graf88d6ed12015-01-02 23:00:16 +0100463 * @pos: the &struct rhash_head to use as a loop cursor.
464 * @tbl: the &struct bucket_table
465 * @hash: the hash value / bucket index
Thomas Graf7e1e7762014-08-02 11:47:44 +0200466 *
467 * This hash chain list-traversal primitive may safely run concurrently with
Thomas Graf88d6ed12015-01-02 23:00:16 +0100468 * the _rcu mutation primitives such as rhashtable_insert() as long as the
Thomas Graf7e1e7762014-08-02 11:47:44 +0200469 * traversal is guarded by rcu_read_lock().
470 */
Thomas Graf88d6ed12015-01-02 23:00:16 +0100471#define rht_for_each_rcu(pos, tbl, hash) \
472 rht_for_each_rcu_continue(pos, (tbl)->buckets[hash], tbl, hash)
473
474/**
475 * rht_for_each_entry_rcu_continue - continue iterating over rcu hash chain
476 * @tpos: the type * to use as a loop cursor.
477 * @pos: the &struct rhash_head to use as a loop cursor.
478 * @head: the previous &struct rhash_head to continue from
479 * @tbl: the &struct bucket_table
480 * @hash: the hash value / bucket index
481 * @member: name of the &struct rhash_head within the hashable struct.
482 *
483 * This hash chain list-traversal primitive may safely run concurrently with
484 * the _rcu mutation primitives such as rhashtable_insert() as long as the
485 * traversal is guarded by rcu_read_lock().
486 */
487#define rht_for_each_entry_rcu_continue(tpos, pos, head, tbl, hash, member) \
488 for (({barrier(); }), \
489 pos = rht_dereference_bucket_rcu(head, tbl, hash); \
Thomas Graff89bd6f2015-01-02 23:00:21 +0100490 (!rht_is_a_nulls(pos)) && rht_entry(tpos, pos, member); \
Thomas Graf88d6ed12015-01-02 23:00:16 +0100491 pos = rht_dereference_bucket_rcu(pos->next, tbl, hash))
Thomas Graf7e1e7762014-08-02 11:47:44 +0200492
493/**
494 * rht_for_each_entry_rcu - iterate over rcu hash chain of given type
Thomas Graf88d6ed12015-01-02 23:00:16 +0100495 * @tpos: the type * to use as a loop cursor.
496 * @pos: the &struct rhash_head to use as a loop cursor.
497 * @tbl: the &struct bucket_table
498 * @hash: the hash value / bucket index
499 * @member: name of the &struct rhash_head within the hashable struct.
Thomas Graf7e1e7762014-08-02 11:47:44 +0200500 *
501 * This hash chain list-traversal primitive may safely run concurrently with
Thomas Graf88d6ed12015-01-02 23:00:16 +0100502 * the _rcu mutation primitives such as rhashtable_insert() as long as the
Thomas Graf7e1e7762014-08-02 11:47:44 +0200503 * traversal is guarded by rcu_read_lock().
504 */
Thomas Graf88d6ed12015-01-02 23:00:16 +0100505#define rht_for_each_entry_rcu(tpos, pos, tbl, hash, member) \
506 rht_for_each_entry_rcu_continue(tpos, pos, (tbl)->buckets[hash],\
507 tbl, hash, member)
Thomas Graf7e1e7762014-08-02 11:47:44 +0200508
Herbert Xu02fd97c2015-03-20 21:57:00 +1100509static inline int rhashtable_compare(struct rhashtable_compare_arg *arg,
510 const void *obj)
511{
512 struct rhashtable *ht = arg->ht;
513 const char *ptr = obj;
514
515 return memcmp(ptr + ht->p.key_offset, arg->key, ht->p.key_len);
516}
517
518/**
519 * rhashtable_lookup_fast - search hash table, inlined version
520 * @ht: hash table
521 * @key: the pointer to the key
522 * @params: hash table parameters
523 *
524 * Computes the hash value for the key and traverses the bucket chain looking
525 * for a entry with an identical key. The first matching entry is returned.
526 *
527 * Returns the first entry on which the compare function returned true.
528 */
529static inline void *rhashtable_lookup_fast(
530 struct rhashtable *ht, const void *key,
531 const struct rhashtable_params params)
532{
533 struct rhashtable_compare_arg arg = {
534 .ht = ht,
535 .key = key,
536 };
537 const struct bucket_table *tbl;
538 struct rhash_head *he;
Thomas Graf299e5c32015-03-24 14:18:17 +0100539 unsigned int hash;
Herbert Xu02fd97c2015-03-20 21:57:00 +1100540
541 rcu_read_lock();
542
543 tbl = rht_dereference_rcu(ht->tbl, ht);
544restart:
545 hash = rht_key_hashfn(ht, tbl, key, params);
546 rht_for_each_rcu(he, tbl, hash) {
547 if (params.obj_cmpfn ?
548 params.obj_cmpfn(&arg, rht_obj(ht, he)) :
549 rhashtable_compare(&arg, rht_obj(ht, he)))
550 continue;
551 rcu_read_unlock();
552 return rht_obj(ht, he);
553 }
554
555 /* Ensure we see any new tables. */
556 smp_rmb();
557
558 tbl = rht_dereference_rcu(tbl->future_tbl, ht);
559 if (unlikely(tbl))
560 goto restart;
561 rcu_read_unlock();
562
563 return NULL;
564}
565
Thomas Grafac833bd2015-03-24 14:18:18 +0100566/* Internal function, please use rhashtable_insert_fast() instead */
Herbert Xu02fd97c2015-03-20 21:57:00 +1100567static inline int __rhashtable_insert_fast(
568 struct rhashtable *ht, const void *key, struct rhash_head *obj,
569 const struct rhashtable_params params)
570{
571 struct rhashtable_compare_arg arg = {
572 .ht = ht,
573 .key = key,
574 };
Herbert Xu02fd97c2015-03-20 21:57:00 +1100575 struct bucket_table *tbl, *new_tbl;
576 struct rhash_head *head;
577 spinlock_t *lock;
Thomas Graf299e5c32015-03-24 14:18:17 +0100578 unsigned int elasticity;
579 unsigned int hash;
Herbert Xuccd57b12015-03-24 00:50:28 +1100580 int err;
Herbert Xu02fd97c2015-03-20 21:57:00 +1100581
Herbert Xuccd57b12015-03-24 00:50:28 +1100582restart:
Herbert Xu02fd97c2015-03-20 21:57:00 +1100583 rcu_read_lock();
584
585 tbl = rht_dereference_rcu(ht->tbl, ht);
Herbert Xu02fd97c2015-03-20 21:57:00 +1100586
Herbert Xub8244782015-03-24 00:50:26 +1100587 /* All insertions must grab the oldest table containing
588 * the hashed bucket that is yet to be rehashed.
Herbert Xu02fd97c2015-03-20 21:57:00 +1100589 */
Herbert Xub8244782015-03-24 00:50:26 +1100590 for (;;) {
591 hash = rht_head_hashfn(ht, tbl, obj, params);
592 lock = rht_bucket_lock(tbl, hash);
593 spin_lock_bh(lock);
594
595 if (tbl->rehash <= hash)
596 break;
597
598 spin_unlock_bh(lock);
599 tbl = rht_dereference_rcu(tbl->future_tbl, ht);
600 }
601
Herbert Xu02fd97c2015-03-20 21:57:00 +1100602 new_tbl = rht_dereference_rcu(tbl->future_tbl, ht);
603 if (unlikely(new_tbl)) {
Herbert Xu3cf92222015-12-03 20:41:29 +0800604 tbl = rhashtable_insert_slow(ht, key, obj, new_tbl);
605 if (!IS_ERR_OR_NULL(tbl))
Herbert Xuccd57b12015-03-24 00:50:28 +1100606 goto slow_path;
Herbert Xu3cf92222015-12-03 20:41:29 +0800607
608 err = PTR_ERR(tbl);
Herbert Xu02fd97c2015-03-20 21:57:00 +1100609 goto out;
610 }
611
Herbert Xu07ee0722015-05-15 11:30:47 +0800612 err = -E2BIG;
613 if (unlikely(rht_grow_above_max(ht, tbl)))
614 goto out;
615
Herbert Xuccd57b12015-03-24 00:50:28 +1100616 if (unlikely(rht_grow_above_100(ht, tbl))) {
617slow_path:
618 spin_unlock_bh(lock);
Herbert Xu3cf92222015-12-03 20:41:29 +0800619 err = rhashtable_insert_rehash(ht, tbl);
Thomas Graf58be8a52015-03-24 14:18:16 +0100620 rcu_read_unlock();
Herbert Xuccd57b12015-03-24 00:50:28 +1100621 if (err)
622 return err;
Herbert Xu02fd97c2015-03-20 21:57:00 +1100623
Herbert Xuccd57b12015-03-24 00:50:28 +1100624 goto restart;
625 }
626
627 err = -EEXIST;
628 elasticity = ht->elasticity;
Herbert Xu02fd97c2015-03-20 21:57:00 +1100629 rht_for_each(head, tbl, hash) {
Herbert Xuccd57b12015-03-24 00:50:28 +1100630 if (key &&
631 unlikely(!(params.obj_cmpfn ?
Herbert Xu02fd97c2015-03-20 21:57:00 +1100632 params.obj_cmpfn(&arg, rht_obj(ht, head)) :
633 rhashtable_compare(&arg, rht_obj(ht, head)))))
634 goto out;
Herbert Xuccd57b12015-03-24 00:50:28 +1100635 if (!--elasticity)
636 goto slow_path;
Herbert Xu02fd97c2015-03-20 21:57:00 +1100637 }
638
Herbert Xu02fd97c2015-03-20 21:57:00 +1100639 err = 0;
640
641 head = rht_dereference_bucket(tbl->buckets[hash], tbl, hash);
642
643 RCU_INIT_POINTER(obj->next, head);
644
645 rcu_assign_pointer(tbl->buckets[hash], obj);
646
647 atomic_inc(&ht->nelems);
648 if (rht_grow_above_75(ht, tbl))
649 schedule_work(&ht->run_work);
650
651out:
652 spin_unlock_bh(lock);
653 rcu_read_unlock();
654
655 return err;
656}
657
658/**
659 * rhashtable_insert_fast - insert object into hash table
660 * @ht: hash table
661 * @obj: pointer to hash head inside object
662 * @params: hash table parameters
663 *
664 * Will take a per bucket spinlock to protect against mutual mutations
665 * on the same bucket. Multiple insertions may occur in parallel unless
666 * they map to the same bucket lock.
667 *
668 * It is safe to call this function from atomic context.
669 *
670 * Will trigger an automatic deferred table resizing if the size grows
671 * beyond the watermark indicated by grow_decision() which can be passed
672 * to rhashtable_init().
673 */
674static inline int rhashtable_insert_fast(
675 struct rhashtable *ht, struct rhash_head *obj,
676 const struct rhashtable_params params)
677{
678 return __rhashtable_insert_fast(ht, NULL, obj, params);
679}
680
681/**
682 * rhashtable_lookup_insert_fast - lookup and insert object into hash table
683 * @ht: hash table
684 * @obj: pointer to hash head inside object
685 * @params: hash table parameters
686 *
687 * Locks down the bucket chain in both the old and new table if a resize
688 * is in progress to ensure that writers can't remove from the old table
689 * and can't insert to the new table during the atomic operation of search
690 * and insertion. Searches for duplicates in both the old and new table if
691 * a resize is in progress.
692 *
693 * This lookup function may only be used for fixed key hash table (key_len
694 * parameter set). It will BUG() if used inappropriately.
695 *
696 * It is safe to call this function from atomic context.
697 *
698 * Will trigger an automatic deferred table resizing if the size grows
699 * beyond the watermark indicated by grow_decision() which can be passed
700 * to rhashtable_init().
701 */
702static inline int rhashtable_lookup_insert_fast(
703 struct rhashtable *ht, struct rhash_head *obj,
704 const struct rhashtable_params params)
705{
706 const char *key = rht_obj(ht, obj);
707
708 BUG_ON(ht->p.obj_hashfn);
709
710 return __rhashtable_insert_fast(ht, key + ht->p.key_offset, obj,
711 params);
712}
713
714/**
715 * rhashtable_lookup_insert_key - search and insert object to hash table
716 * with explicit key
717 * @ht: hash table
718 * @key: key
719 * @obj: pointer to hash head inside object
720 * @params: hash table parameters
721 *
722 * Locks down the bucket chain in both the old and new table if a resize
723 * is in progress to ensure that writers can't remove from the old table
724 * and can't insert to the new table during the atomic operation of search
725 * and insertion. Searches for duplicates in both the old and new table if
726 * a resize is in progress.
727 *
728 * Lookups may occur in parallel with hashtable mutations and resizing.
729 *
730 * Will trigger an automatic deferred table resizing if the size grows
731 * beyond the watermark indicated by grow_decision() which can be passed
732 * to rhashtable_init().
733 *
734 * Returns zero on success.
735 */
736static inline int rhashtable_lookup_insert_key(
737 struct rhashtable *ht, const void *key, struct rhash_head *obj,
738 const struct rhashtable_params params)
739{
740 BUG_ON(!ht->p.obj_hashfn || !key);
741
742 return __rhashtable_insert_fast(ht, key, obj, params);
743}
744
Thomas Grafac833bd2015-03-24 14:18:18 +0100745/* Internal function, please use rhashtable_remove_fast() instead */
Herbert Xu02fd97c2015-03-20 21:57:00 +1100746static inline int __rhashtable_remove_fast(
747 struct rhashtable *ht, struct bucket_table *tbl,
748 struct rhash_head *obj, const struct rhashtable_params params)
749{
750 struct rhash_head __rcu **pprev;
751 struct rhash_head *he;
752 spinlock_t * lock;
Thomas Graf299e5c32015-03-24 14:18:17 +0100753 unsigned int hash;
Herbert Xu02fd97c2015-03-20 21:57:00 +1100754 int err = -ENOENT;
755
756 hash = rht_head_hashfn(ht, tbl, obj, params);
757 lock = rht_bucket_lock(tbl, hash);
758
759 spin_lock_bh(lock);
760
761 pprev = &tbl->buckets[hash];
762 rht_for_each(he, tbl, hash) {
763 if (he != obj) {
764 pprev = &he->next;
765 continue;
766 }
767
768 rcu_assign_pointer(*pprev, obj->next);
769 err = 0;
770 break;
771 }
772
773 spin_unlock_bh(lock);
774
775 return err;
776}
777
778/**
779 * rhashtable_remove_fast - remove object from hash table
780 * @ht: hash table
781 * @obj: pointer to hash head inside object
782 * @params: hash table parameters
783 *
784 * Since the hash chain is single linked, the removal operation needs to
785 * walk the bucket chain upon removal. The removal operation is thus
786 * considerable slow if the hash table is not correctly sized.
787 *
788 * Will automatically shrink the table via rhashtable_expand() if the
789 * shrink_decision function specified at rhashtable_init() returns true.
790 *
791 * Returns zero on success, -ENOENT if the entry could not be found.
792 */
793static inline int rhashtable_remove_fast(
794 struct rhashtable *ht, struct rhash_head *obj,
795 const struct rhashtable_params params)
796{
797 struct bucket_table *tbl;
798 int err;
799
800 rcu_read_lock();
801
802 tbl = rht_dereference_rcu(ht->tbl, ht);
803
804 /* Because we have already taken (and released) the bucket
805 * lock in old_tbl, if we find that future_tbl is not yet
806 * visible then that guarantees the entry to still be in
807 * the old tbl if it exists.
808 */
809 while ((err = __rhashtable_remove_fast(ht, tbl, obj, params)) &&
810 (tbl = rht_dereference_rcu(tbl->future_tbl, ht)))
811 ;
812
813 if (err)
814 goto out;
815
816 atomic_dec(&ht->nelems);
Thomas Grafb5e2c152015-03-24 20:42:19 +0000817 if (unlikely(ht->p.automatic_shrinking &&
818 rht_shrink_below_30(ht, tbl)))
Herbert Xu02fd97c2015-03-20 21:57:00 +1100819 schedule_work(&ht->run_work);
820
821out:
822 rcu_read_unlock();
823
824 return err;
825}
826
Tom Herbert3502cad2015-12-15 15:41:36 -0800827/* Internal function, please use rhashtable_replace_fast() instead */
828static inline int __rhashtable_replace_fast(
829 struct rhashtable *ht, struct bucket_table *tbl,
830 struct rhash_head *obj_old, struct rhash_head *obj_new,
831 const struct rhashtable_params params)
832{
833 struct rhash_head __rcu **pprev;
834 struct rhash_head *he;
835 spinlock_t *lock;
836 unsigned int hash;
837 int err = -ENOENT;
838
839 /* Minimally, the old and new objects must have same hash
840 * (which should mean identifiers are the same).
841 */
842 hash = rht_head_hashfn(ht, tbl, obj_old, params);
843 if (hash != rht_head_hashfn(ht, tbl, obj_new, params))
844 return -EINVAL;
845
846 lock = rht_bucket_lock(tbl, hash);
847
848 spin_lock_bh(lock);
849
850 pprev = &tbl->buckets[hash];
851 rht_for_each(he, tbl, hash) {
852 if (he != obj_old) {
853 pprev = &he->next;
854 continue;
855 }
856
857 rcu_assign_pointer(obj_new->next, obj_old->next);
858 rcu_assign_pointer(*pprev, obj_new);
859 err = 0;
860 break;
861 }
862
863 spin_unlock_bh(lock);
864
865 return err;
866}
867
868/**
869 * rhashtable_replace_fast - replace an object in hash table
870 * @ht: hash table
871 * @obj_old: pointer to hash head inside object being replaced
872 * @obj_new: pointer to hash head inside object which is new
873 * @params: hash table parameters
874 *
875 * Replacing an object doesn't affect the number of elements in the hash table
876 * or bucket, so we don't need to worry about shrinking or expanding the
877 * table here.
878 *
879 * Returns zero on success, -ENOENT if the entry could not be found,
880 * -EINVAL if hash is not the same for the old and new objects.
881 */
882static inline int rhashtable_replace_fast(
883 struct rhashtable *ht, struct rhash_head *obj_old,
884 struct rhash_head *obj_new,
885 const struct rhashtable_params params)
886{
887 struct bucket_table *tbl;
888 int err;
889
890 rcu_read_lock();
891
892 tbl = rht_dereference_rcu(ht->tbl, ht);
893
894 /* Because we have already taken (and released) the bucket
895 * lock in old_tbl, if we find that future_tbl is not yet
896 * visible then that guarantees the entry to still be in
897 * the old tbl if it exists.
898 */
899 while ((err = __rhashtable_replace_fast(ht, tbl, obj_old,
900 obj_new, params)) &&
901 (tbl = rht_dereference_rcu(tbl->future_tbl, ht)))
902 ;
903
904 rcu_read_unlock();
905
906 return err;
907}
908
Thomas Graf7e1e7762014-08-02 11:47:44 +0200909#endif /* _LINUX_RHASHTABLE_H */