1 #ifndef _LINUX_RING_BUFFER_FRONTEND_INTERNAL_H
2 #define _LINUX_RING_BUFFER_FRONTEND_INTERNAL_H
5 * linux/ringbuffer/frontend_internal.h
7 * (C) Copyright 2005-2010 - Mathieu Desnoyers <mathieu.desnoyers@efficios.com>
9 * Ring Buffer Library Synchronization Header (internal helpers).
12 * Mathieu Desnoyers <mathieu.desnoyers@efficios.com>
14 * See ring_buffer_frontend.c for more information on wait-free algorithms.
16 * Dual LGPL v2.1/GPL v2 license.
19 #include "../../wrapper/ringbuffer/config.h"
20 #include "../../wrapper/ringbuffer/backend_types.h"
21 #include "../../wrapper/ringbuffer/frontend_types.h"
22 #include "../../lib/prio_heap/lttng_prio_heap.h" /* For per-CPU read-side iterator */
24 /* Buffer offset macros */
26 /* buf_trunc mask selects only the buffer number. */
28 unsigned long buf_trunc(unsigned long offset
, struct channel
*chan
)
30 return offset
& ~(chan
->backend
.buf_size
- 1);
34 /* Select the buffer number value (counter). */
36 unsigned long buf_trunc_val(unsigned long offset
, struct channel
*chan
)
38 return buf_trunc(offset
, chan
) >> chan
->backend
.buf_size_order
;
41 /* buf_offset mask selects only the offset within the current buffer. */
43 unsigned long buf_offset(unsigned long offset
, struct channel
*chan
)
45 return offset
& (chan
->backend
.buf_size
- 1);
48 /* subbuf_offset mask selects the offset within the current subbuffer. */
50 unsigned long subbuf_offset(unsigned long offset
, struct channel
*chan
)
52 return offset
& (chan
->backend
.subbuf_size
- 1);
55 /* subbuf_trunc mask selects the subbuffer number. */
57 unsigned long subbuf_trunc(unsigned long offset
, struct channel
*chan
)
59 return offset
& ~(chan
->backend
.subbuf_size
- 1);
62 /* subbuf_align aligns the offset to the next subbuffer. */
64 unsigned long subbuf_align(unsigned long offset
, struct channel
*chan
)
66 return (offset
+ chan
->backend
.subbuf_size
)
67 & ~(chan
->backend
.subbuf_size
- 1);
70 /* subbuf_index returns the index of the current subbuffer within the buffer. */
72 unsigned long subbuf_index(unsigned long offset
, struct channel
*chan
)
74 return buf_offset(offset
, chan
) >> chan
->backend
.subbuf_size_order
;
78 * Last TSC comparison functions. Check if the current TSC overflows tsc_bits
79 * bits from the last TSC read. When overflows are detected, the full 64-bit
80 * timestamp counter should be written in the record header. Reads and writes
81 * last_tsc atomically.
84 #if (BITS_PER_LONG == 32)
86 void save_last_tsc(const struct lib_ring_buffer_config
*config
,
87 struct lib_ring_buffer
*buf
, u64 tsc
)
89 if (config
->tsc_bits
== 0 || config
->tsc_bits
== 64)
93 * Ensure the compiler performs this update in a single instruction.
95 v_set(config
, &buf
->last_tsc
, (unsigned long)(tsc
>> config
->tsc_bits
));
99 int last_tsc_overflow(const struct lib_ring_buffer_config
*config
,
100 struct lib_ring_buffer
*buf
, u64 tsc
)
102 unsigned long tsc_shifted
;
104 if (config
->tsc_bits
== 0 || config
->tsc_bits
== 64)
107 tsc_shifted
= (unsigned long)(tsc
>> config
->tsc_bits
);
108 if (unlikely(tsc_shifted
109 - (unsigned long)v_read(config
, &buf
->last_tsc
)))
116 void save_last_tsc(const struct lib_ring_buffer_config
*config
,
117 struct lib_ring_buffer
*buf
, u64 tsc
)
119 if (config
->tsc_bits
== 0 || config
->tsc_bits
== 64)
122 v_set(config
, &buf
->last_tsc
, (unsigned long)tsc
);
126 int last_tsc_overflow(const struct lib_ring_buffer_config
*config
,
127 struct lib_ring_buffer
*buf
, u64 tsc
)
129 if (config
->tsc_bits
== 0 || config
->tsc_bits
== 64)
132 if (unlikely((tsc
- v_read(config
, &buf
->last_tsc
))
133 >> config
->tsc_bits
))
141 int lib_ring_buffer_reserve_slow(struct lib_ring_buffer_ctx
*ctx
);
144 void lib_ring_buffer_switch_slow(struct lib_ring_buffer
*buf
,
145 enum switch_mode mode
);
147 /* Buffer write helpers */
150 void lib_ring_buffer_reserve_push_reader(struct lib_ring_buffer
*buf
,
151 struct channel
*chan
,
152 unsigned long offset
)
154 unsigned long consumed_old
, consumed_new
;
157 consumed_old
= atomic_long_read(&buf
->consumed
);
159 * If buffer is in overwrite mode, push the reader consumed
160 * count if the write position has reached it and we are not
161 * at the first iteration (don't push the reader farther than
162 * the writer). This operation can be done concurrently by many
163 * writers in the same buffer, the writer being at the farthest
164 * write position sub-buffer index in the buffer being the one
165 * which will win this loop.
167 if (unlikely(subbuf_trunc(offset
, chan
)
168 - subbuf_trunc(consumed_old
, chan
)
169 >= chan
->backend
.buf_size
))
170 consumed_new
= subbuf_align(consumed_old
, chan
);
173 } while (unlikely(atomic_long_cmpxchg(&buf
->consumed
, consumed_old
,
174 consumed_new
) != consumed_old
));
178 void lib_ring_buffer_vmcore_check_deliver(const struct lib_ring_buffer_config
*config
,
179 struct lib_ring_buffer
*buf
,
180 unsigned long commit_count
,
183 if (config
->oops
== RING_BUFFER_OOPS_CONSISTENCY
)
184 v_set(config
, &buf
->commit_hot
[idx
].seq
, commit_count
);
188 int lib_ring_buffer_poll_deliver(const struct lib_ring_buffer_config
*config
,
189 struct lib_ring_buffer
*buf
,
190 struct channel
*chan
)
192 unsigned long consumed_old
, consumed_idx
, commit_count
, write_offset
;
194 consumed_old
= atomic_long_read(&buf
->consumed
);
195 consumed_idx
= subbuf_index(consumed_old
, chan
);
196 commit_count
= v_read(config
, &buf
->commit_cold
[consumed_idx
].cc_sb
);
198 * No memory barrier here, since we are only interested
199 * in a statistically correct polling result. The next poll will
200 * get the data is we are racing. The mb() that ensures correct
201 * memory order is in get_subbuf.
203 write_offset
= v_read(config
, &buf
->offset
);
206 * Check that the subbuffer we are trying to consume has been
207 * already fully committed.
210 if (((commit_count
- chan
->backend
.subbuf_size
)
211 & chan
->commit_count_mask
)
212 - (buf_trunc(consumed_old
, chan
)
213 >> chan
->backend
.num_subbuf_order
)
218 * Check that we are not about to read the same subbuffer in
219 * which the writer head is.
221 if (subbuf_trunc(write_offset
, chan
) - subbuf_trunc(consumed_old
, chan
)
230 int lib_ring_buffer_pending_data(const struct lib_ring_buffer_config
*config
,
231 struct lib_ring_buffer
*buf
,
232 struct channel
*chan
)
234 return !!subbuf_offset(v_read(config
, &buf
->offset
), chan
);
238 unsigned long lib_ring_buffer_get_data_size(const struct lib_ring_buffer_config
*config
,
239 struct lib_ring_buffer
*buf
,
242 return subbuffer_get_data_size(config
, &buf
->backend
, idx
);
246 * Check if all space reservation in a buffer have been committed. This helps
247 * knowing if an execution context is nested (for per-cpu buffers only).
248 * This is a very specific ftrace use-case, so we keep this as "internal" API.
251 int lib_ring_buffer_reserve_committed(const struct lib_ring_buffer_config
*config
,
252 struct lib_ring_buffer
*buf
,
253 struct channel
*chan
)
255 unsigned long offset
, idx
, commit_count
;
257 CHAN_WARN_ON(chan
, config
->alloc
!= RING_BUFFER_ALLOC_PER_CPU
);
258 CHAN_WARN_ON(chan
, config
->sync
!= RING_BUFFER_SYNC_PER_CPU
);
261 * Read offset and commit count in a loop so they are both read
262 * atomically wrt interrupts. By deal with interrupt concurrency by
263 * restarting both reads if the offset has been pushed. Note that given
264 * we only have to deal with interrupt concurrency here, an interrupt
265 * modifying the commit count will also modify "offset", so it is safe
266 * to only check for offset modifications.
269 offset
= v_read(config
, &buf
->offset
);
270 idx
= subbuf_index(offset
, chan
);
271 commit_count
= v_read(config
, &buf
->commit_hot
[idx
].cc
);
272 } while (offset
!= v_read(config
, &buf
->offset
));
274 return ((buf_trunc(offset
, chan
) >> chan
->backend
.num_subbuf_order
)
275 - (commit_count
& chan
->commit_count_mask
) == 0);
279 void lib_ring_buffer_check_deliver(const struct lib_ring_buffer_config
*config
,
280 struct lib_ring_buffer
*buf
,
281 struct channel
*chan
,
282 unsigned long offset
,
283 unsigned long commit_count
,
286 unsigned long old_commit_count
= commit_count
287 - chan
->backend
.subbuf_size
;
290 /* Check if all commits have been done */
291 if (unlikely((buf_trunc(offset
, chan
) >> chan
->backend
.num_subbuf_order
)
292 - (old_commit_count
& chan
->commit_count_mask
) == 0)) {
294 * If we succeeded at updating cc_sb below, we are the subbuffer
295 * writer delivering the subbuffer. Deals with concurrent
296 * updates of the "cc" value without adding a add_return atomic
297 * operation to the fast path.
299 * We are doing the delivery in two steps:
300 * - First, we cmpxchg() cc_sb to the new value
301 * old_commit_count + 1. This ensures that we are the only
302 * subbuffer user successfully filling the subbuffer, but we
303 * do _not_ set the cc_sb value to "commit_count" yet.
304 * Therefore, other writers that would wrap around the ring
305 * buffer and try to start writing to our subbuffer would
306 * have to drop records, because it would appear as
308 * We therefore have exclusive access to the subbuffer control
309 * structures. This mutual exclusion with other writers is
310 * crucially important to perform record overruns count in
311 * flight recorder mode locklessly.
312 * - When we are ready to release the subbuffer (either for
313 * reading or for overrun by other writers), we simply set the
314 * cc_sb value to "commit_count" and perform delivery.
316 * The subbuffer size is least 2 bytes (minimum size: 1 page).
317 * This guarantees that old_commit_count + 1 != commit_count.
319 if (likely(v_cmpxchg(config
, &buf
->commit_cold
[idx
].cc_sb
,
320 old_commit_count
, old_commit_count
+ 1)
321 == old_commit_count
)) {
323 * Start of exclusive subbuffer access. We are
324 * guaranteed to be the last writer in this subbuffer
325 * and any other writer trying to access this subbuffer
326 * in this state is required to drop records.
328 tsc
= config
->cb
.ring_buffer_clock_read(chan
);
330 subbuffer_get_records_count(config
,
332 &buf
->records_count
);
334 subbuffer_count_records_overrun(config
,
337 &buf
->records_overrun
);
338 config
->cb
.buffer_end(buf
, tsc
, idx
,
339 lib_ring_buffer_get_data_size(config
,
344 * Set noref flag and offset for this subbuffer id.
345 * Contains a memory barrier that ensures counter stores
346 * are ordered before set noref and offset.
348 lib_ring_buffer_set_noref_offset(config
, &buf
->backend
, idx
,
349 buf_trunc_val(offset
, chan
));
352 * Order set_noref and record counter updates before the
353 * end of subbuffer exclusive access. Orders with
354 * respect to writers coming into the subbuffer after
355 * wrap around, and also order wrt concurrent readers.
358 /* End of exclusive subbuffer access */
359 v_set(config
, &buf
->commit_cold
[idx
].cc_sb
,
361 lib_ring_buffer_vmcore_check_deliver(config
, buf
,
365 * RING_BUFFER_WAKEUP_BY_WRITER wakeup is not lock-free.
367 if (config
->wakeup
== RING_BUFFER_WAKEUP_BY_WRITER
368 && atomic_long_read(&buf
->active_readers
)
369 && lib_ring_buffer_poll_deliver(config
, buf
, chan
)) {
370 wake_up_interruptible(&buf
->read_wait
);
371 wake_up_interruptible(&chan
->read_wait
);
379 * lib_ring_buffer_write_commit_counter
381 * For flight recording. must be called after commit.
382 * This function increments the subbuffer's commit_seq counter each time the
383 * commit count reaches back the reserve offset (modulo subbuffer size). It is
384 * useful for crash dump.
387 void lib_ring_buffer_write_commit_counter(const struct lib_ring_buffer_config
*config
,
388 struct lib_ring_buffer
*buf
,
389 struct channel
*chan
,
391 unsigned long buf_offset
,
392 unsigned long commit_count
,
395 unsigned long offset
, commit_seq_old
;
397 if (config
->oops
!= RING_BUFFER_OOPS_CONSISTENCY
)
400 offset
= buf_offset
+ slot_size
;
403 * subbuf_offset includes commit_count_mask. We can simply
404 * compare the offsets within the subbuffer without caring about
405 * buffer full/empty mismatch because offset is never zero here
406 * (subbuffer header and record headers have non-zero length).
408 if (unlikely(subbuf_offset(offset
- commit_count
, chan
)))
411 commit_seq_old
= v_read(config
, &buf
->commit_hot
[idx
].seq
);
412 while ((long) (commit_seq_old
- commit_count
) < 0)
413 commit_seq_old
= v_cmpxchg(config
, &buf
->commit_hot
[idx
].seq
,
414 commit_seq_old
, commit_count
);
417 extern int lib_ring_buffer_create(struct lib_ring_buffer
*buf
,
418 struct channel_backend
*chanb
, int cpu
);
419 extern void lib_ring_buffer_free(struct lib_ring_buffer
*buf
);
421 /* Keep track of trap nesting inside ring buffer code */
422 DECLARE_PER_CPU(unsigned int, lib_ring_buffer_nesting
);
424 #endif /* _LINUX_RING_BUFFER_FRONTEND_INTERNAL_H */