Every language
12 implementations, copy-ready. One at a time with syntax highlighting, or all inline.
JSJavaScript
class TokenBucket {
constructor(capacity, ratePerSec) {
this.capacity = capacity;
this.ratePerSec = ratePerSec;
this.tokens = capacity; // start full — the burst is allowed
this.lastRefill = Date.now();
}
// false = exhausted → the caller answers 429 + Retry-After:
take(n = 1) {
const now = Date.now();
// lazy refill — elapsed time IS the timer, no setInterval:
this.tokens = Math.min(
this.capacity,
this.tokens + ((now - this.lastRefill) / 1000) * this.ratePerSec,
);
this.lastRefill = now;
if (this.tokens < n) return false;
this.tokens -= n;
return true;
}
}The refill is computed from elapsed time on EVERY take — lazy, so there is NO setInterval, no background timer, no drift when idle. Date.now() is wall time; on Node, process.hrtime.bigint() is the monotonic strict choice. In a cluster this in-memory bucket is per-instance — each server grants the full limit.
TSTypeScript
type Clock = () => number; // ms; inject a fake in tests, never sleep
export class TokenBucket {
private tokens: number;
private lastRefill: number;
constructor(
readonly capacity: number,
readonly ratePerSec: number,
private clock: Clock = () => Date.now(),
) {
this.tokens = capacity; // full bucket = the allowed burst
this.lastRefill = this.clock();
}
private refill(): void {
const now = this.clock();
this.tokens = Math.min(
this.capacity,
this.tokens + ((now - this.lastRefill) / 1000) * this.ratePerSec,
);
this.lastRefill = now;
}
take(n = 1): boolean {
this.refill();
if (this.tokens < n) return false;
this.tokens -= n;
return true;
}
// seconds until n tokens exist — the Retry-After header value:
retryAfterSec(n = 1): number {
this.refill();
return Math.ceil((n - this.tokens) / this.ratePerSec);
}
}take(n?: number): boolean is the whole contract; the injected Clock makes the elapsed-time refill testable without real waiting. The exhausted case is a 429 with Retry-After (retryAfterSec computes it), not a silent drop — that decision belongs in the caller, which is why take returns a boolean instead of throwing.
GoGo
import (
"sync"
"time"
)
type TokenBucket struct {
mu sync.Mutex
tokens float64
lastRefill time.Time
capacity float64
ratePerSec float64
}
func NewTokenBucket(capacity, ratePerSec float64) *TokenBucket {
return &TokenBucket{
tokens: capacity, // full — the burst is allowed
lastRefill: time.Now(),
capacity: capacity,
ratePerSec: ratePerSec,
}
}
// Allow reports whether one token exists. Exhausted = false →
// the handler answers 429 with Retry-After, never a silent drop.
func (b *TokenBucket) Allow() bool {
b.mu.Lock()
defer b.mu.Unlock()
now := time.Now() // the lazy refill — no Ticker, no goroutine
elapsed := now.Sub(b.lastRefill).Seconds()
b.tokens = min(b.capacity, b.tokens+elapsed*b.ratePerSec)
b.lastRefill = now
if b.tokens < 1 {
return false
}
b.tokens--
return true
}The mutex is load-bearing: the lazy refill reads and writes tokens+lastRefill as a pair. time.Now() monotonic reading (its embedded clock) is what makes the elapsed-time math safe. For production, golang.org/x/time/rate is the limiter — rate.NewLimiter(rate.Limit(r), burst) with AllowN/WaitN, reservation-aware and battle-tested.
RsRust
use std::sync::Mutex;
use std::time::Instant;
pub struct TokenBucket {
capacity: f64,
rate_per_sec: f64,
state: Mutex<State>, // tokens + timestamp travel as a pair
}
struct State {
tokens: f64,
last_refill: Instant,
}
impl TokenBucket {
pub fn new(capacity: f64, rate_per_sec: f64) -> Self {
Self {
capacity,
rate_per_sec,
state: Mutex::new(State {
tokens: capacity, // full — the burst is allowed
last_refill: Instant::now(),
}),
}
}
/// false = exhausted → answer 429 with Retry-After, not a silent drop.
pub fn take(&self, n: f64) -> bool {
let mut s = self.state.lock().unwrap();
let elapsed = s.last_refill.elapsed().as_secs_f64(); // lazy refill
s.tokens = (s.tokens + elapsed * self.rate_per_sec)
.min(self.capacity);
s.last_refill = Instant::now();
if s.tokens < n {
return false;
}
s.tokens -= n;
true
}
}Instant is monotonic — immune to the wall-clock steps that would corrupt elapsed-time math (SystemTime is not). Atomics only work if the whole state fits in one word; the tokens+timestamp pair needs the Mutex. The governor crate is the ecosystem answer — GCRA-based, with ready middleware for axum/actix.
PHPPHP
// PHP is shared-nothing: a property/static bucket resets EVERY request.
// The state must live in APCu (one host) or Redis (any fleet size):
function allow(string $key, int $capacity, float $ratePerSec): bool
{
$now = microtime(true); // float seconds
[$tokens, $last] = apcu_fetch($key) ?: [$capacity, $now];
// lazy refill from elapsed time — no timer, no daemon:
$tokens = min($capacity, $tokens + ($now - $last) * $ratePerSec);
$ok = $tokens >= 1.0;
if ($ok) {
$tokens -= 1.0;
}
apcu_store($key, [$tokens, $now]);
return $ok; // false → 429 + Retry-After header, not a silent drop
}A per-request (in-memory) bucket in PHP grants capacity × server count — each worker starts full. APCu fixes one host; Redis fixes the fleet, and its EVAL/Lua does the read-modify-write atomically (the fetch/store pair above races under concurrency). microtime(true) is the lazy refill's clock — elapsed time replaces any timer.
PyPython
import time
class TokenBucket:
def __init__(self, capacity: float, rate_per_sec: float):
self.capacity = capacity
self.rate_per_sec = rate_per_sec
self.tokens = capacity # full — burst allowed
self.last_refill = time.monotonic()
def take(self, n: float = 1) -> bool:
now = time.monotonic()
# lazy refill — elapsed time IS the timer:
self.tokens = min(
self.capacity,
self.tokens + (now - self.last_refill) * self.rate_per_sec,
)
self.last_refill = now
if self.tokens < n:
return False # exhausted → 429 + Retry-After, not a silent drop
self.tokens -= n
return Truetime.monotonic(), never time.time() — a backwards wall-clock step would freeze the refill math. Under threads, take() needs a threading.Lock around the read-modify-write. The limits package (Redis-backed) and aiolimiter (async) ship this pattern ready-made — a Redis bucket serializes the same math in Lua.
C#C#
using System.Diagnostics;
using System.Threading;
public sealed class TokenBucket(int capacity, double ratePerSec)
{
private readonly object _gate = new();
private double _tokens = capacity; // full — burst allowed
private long _lastRefillTicks = Stopwatch.GetTimestamp();
/// false = exhausted → answer 429 + Retry-After, never drop silently.
public bool Take(int n = 1)
{
lock (_gate)
{
long now = Stopwatch.GetTimestamp(); // monotonic
double elapsed = (now - _lastRefillTicks)
/ (double)Stopwatch.Frequency; // lazy refill
_tokens = Math.Min(capacity, _tokens + elapsed * ratePerSec);
_lastRefillTicks = now;
if (_tokens < n) return false;
_tokens -= n;
return true;
}
}
}SemaphoreSlim is the trap — it throttles concurrency, not rate (no refill over time). Stopwatch.GetTimestamp()/Frequency is the monotonic-clock lazy refill; DateTime.Now steps and corrupts the math. Since .NET 7 the stdlib answer IS System.Threading.RateLimiting — TokenBucketRateLimiter, wired into ASP.NET Core via RateLimiterMiddleware; use it before hand-rolling.
JvJava
import java.util.concurrent.locks.ReentrantLock;
public final class TokenBucket {
private final double capacity, ratePerSec;
private double tokens, lastRefill; // seconds on the monotonic clock
private final ReentrantLock lock = new ReentrantLock();
public TokenBucket(double capacity, double ratePerSec) {
this.capacity = capacity;
this.ratePerSec = ratePerSec;
this.tokens = capacity; // full — burst allowed
this.lastRefill = System.nanoTime() / 1e9;
}
/** false = exhausted → respond 429 + Retry-After, never drop silently. */
public boolean take(int n) {
lock.lock();
try {
double now = System.nanoTime() / 1e9; // lazy refill, no scheduler
tokens = Math.min(capacity,
tokens + (now - lastRefill) * ratePerSec);
lastRefill = now;
if (tokens < n) return false;
tokens -= n;
return true;
} finally {
lock.unlock();
}
}
}Semaphore is NOT a rate limiter — the common wrong import: it caps concurrency and never refills over time, so a 5-permit Semaphore is a 5-burst forever after, not 5-per-second. System.nanoTime() is the monotonic clock (currentTimeMillis steps with NTP). Bucket4j is the production library, with Redis/Hazelcast backends for clustered bucket state.
SwSwift
actor TokenBucket {
private let capacity: Double
private let ratePerSec: Double
private var tokens: Double
private var lastRefill: Double // seconds, monotonic
init(capacity: Double, ratePerSec: Double) {
self.capacity = capacity
self.ratePerSec = ratePerSec
self.tokens = capacity // full — burst allowed
self.lastRefill = Double(DispatchTime.now().uptimeNanoseconds)
/ 1_000_000_000
}
/// false = exhausted → respond 429 with Retry-After, not a silent drop.
func take(_ n: Double = 1) -> Bool {
let now = Double(DispatchTime.now().uptimeNanoseconds) / 1_000_000_000
tokens = min(capacity, tokens + (now - lastRefill) * ratePerSec)
lastRefill = now
if tokens < n { return false }
tokens -= n
return true
}
}An actor needs no lock at all — Swift concurrency serializes every method call on the actor's executor, so take() is race-free by construction; just never hand out the var. DispatchTime.uptimeNanoseconds is the monotonic clock (Date is wall time and steps). Callers write await bucket.take() — and blocking a server thread on Task { } wrappers defeats the point.
KtKotlin
import kotlinx.coroutines.sync.Mutex
import kotlinx.coroutines.sync.withLock
class TokenBucket(
private val capacity: Double,
private val ratePerSec: Double,
) {
private val mutex = Mutex() // carries tokens + timestamp as a pair
private var tokens = capacity // full — burst allowed
private var lastRefill = System.nanoTime() / 1e9
/** false = exhausted → 429 + Retry-After, not a silent drop. */
suspend fun take(n: Int = 1): Boolean = mutex.withLock {
val now = System.nanoTime() / 1e9
tokens = minOf(capacity, tokens + (now - lastRefill) * ratePerSec)
lastRefill = now
if (tokens < n) return@withLock false
tokens -= n
true
}
}kotlinx.coroutines.sync.Mutex, not synchronized — the suspend-friendly lock does not park the thread while waiting. There is no atomic double, and two separate atomics (tokens + timestamp) can interleave mid-refill, so one Mutex owns both fields. Bucket4j is the production JVM answer; its async API bridges to coroutines by awaiting the returned CompletableFuture.
RbRuby
class TokenBucket
def initialize(capacity:, rate_per_sec:)
@capacity = capacity
@rate = rate_per_sec
@tokens = capacity # full — burst allowed
@last = Process.clock_gettime(Process::CLOCK_MONOTONIC)
@mu = Mutex.new
end
# false = exhausted → 429 + Retry-After, not a silent drop
def take(n = 1)
@mu.synchronize do
now = Process.clock_gettime(Process::CLOCK_MONOTONIC)
@tokens = [@capacity, @tokens + (now - @last) * @rate].min
@last = now
next false if @tokens < n
@tokens -= n
true
end
end
endProcess.clock_gettime(Process::CLOCK_MONOTONIC) — Time.now can step under NTP and corrupt an elapsed-time refill. The Mutex matters even on MRI: the GVL does not make the read-refill-write sequence atomic. For the HTTP layer, Rack::Attack (Redis-backed) is the drop-in; per-process buckets multiply the limit by your puma worker count.
ZigZig
const std = @import("std");
pub const TokenBucket = struct {
capacity: f64,
rate_per_ms: f64,
tokens: f64,
last_ms: i64,
mu: std.Thread.Mutex = .{},
pub fn init(capacity: f64, rate_per_sec: f64) TokenBucket {
return .{
.capacity = capacity,
.rate_per_ms = rate_per_sec / 1000.0,
.tokens = capacity, // full — burst allowed
.last_ms = std.time.milliTimestamp(),
};
}
/// false = exhausted → answer 429 + Retry-After, never drop silently.
pub fn take(self: *TokenBucket, n: f64) bool {
self.mu.lock();
defer self.mu.unlock();
const now = std.time.milliTimestamp();
const elapsed: f64 = @floatFromInt(now - self.last_ms);
self.tokens = @min(self.capacity,
self.tokens + elapsed * self.rate_per_ms); // lazy refill
self.last_ms = now;
if (self.tokens < n) return false;
self.tokens -= n;
return true;
}
};The mutex exists for multi-threaded callers — the refill is a read-modify-write over two fields, not one atomic swap; plain atomics suffice only for single-counter buckets that fit in a word. std.time.milliTimestamp() is wall-clock ms — adequate because the math just needs a consistently increasing counter, but std.time.Timer is the strictly monotonic choice.