Caching keeps the result of slow work so the next request is fast. The common pattern is cache-aside: look in the cache, and on a miss load from the database and store the result with a time-to-live (TTL). Two details matter: when many requests miss at the same moment, load once and share the promise; and when the data changes, delete the cached entry so nobody sees stale results for a whole TTL. An in-memory Map works for one server process; with several servers, use a shared store such as Redis.
HTTP has caching built in. Cache-Control: private, max-age=30 lets the browser reuse a response for 30 seconds without asking. Express adds an ETag (a fingerprint of the body) to JSON responses; when the browser revalidates with If-None-Match, an unchanged response becomes a tiny 304 Not Modified.
Rate limiting caps how often one client can call you, which protects against password guessing, scraping and runaway scripts. A fixed window counts requests per key (IP address or user id) per time window and answers 429 Too Many Requests with a Retry-After header when the count is exceeded. Use a strict limit on /login and a generous one elsewhere. Behind a proxy or load balancer, set app.set("trust proxy", 1) so req.ip is the real client address. Passing the clock in as now() makes both the cache and the limiter easy to test.
type Clock = () => number;
interface Entry<V> {
value: V;
expiresAt: number;
}
// A small in-memory cache with a time-to-live. For several servers, use a shared cache such as Redis.
export class TtlCache<V> {
private entries = new Map<string, Entry<V>>();
private loading = new Map<string, Promise<V>>();
constructor(private ttlMs: number, private now: Clock = Date.now) {}
get(key: string): V | undefined {
const entry = this.entries.get(key);
if (!entry) return undefined;
if (entry.expiresAt <= this.now()) {
this.entries.delete(key);
return undefined;
}
return entry.value;
}
set(key: string, value: V): void {
this.entries.set(key, { value, expiresAt: this.now() + this.ttlMs });
}
delete(key: string): void {
this.entries.delete(key);
}
// Cache-aside: return the cached value, or load it ONCE even if many requests ask at the same time.
async getOrLoad(key: string, load: () => Promise<V>): Promise<V> {
const cached = this.get(key);
if (cached !== undefined) return cached;
const inFlight = this.loading.get(key);
if (inFlight) return inFlight;
const promise = load()
.then((value) => {
this.set(key, value);
return value;
})
.finally(() => this.loading.delete(key));
this.loading.set(key, promise);
return promise;
}
}import type { NextFunction, Request, Response } from "express";
interface Options {
windowMs: number;
max: number;
key?: (req: Request) => string;
now?: () => number;
}
// Fixed window: at most `max` requests per key in each window of `windowMs`.
export function rateLimit({ windowMs, max, key = (req) => req.ip ?? "unknown", now = Date.now }: Options) {
const windows = new Map<string, { count: number; resetAt: number }>();
return (req: Request, res: Response, next: NextFunction) => {
const k = key(req);
const t = now();
let w = windows.get(k);
if (!w || w.resetAt <= t) {
w = { count: 0, resetAt: t + windowMs };
windows.set(k, w);
}
w.count++;
const resetSeconds = Math.ceil((w.resetAt - t) / 1000);
res.set("RateLimit-Limit", String(max));
res.set("RateLimit-Remaining", String(Math.max(0, max - w.count)));
res.set("RateLimit-Reset", String(resetSeconds));
if (w.count > max) {
res.set("Retry-After", String(resetSeconds));
return res.status(429).json({ error: "Too many requests, slow down" });
}
next();
};
}import express from "express";
import { TtlCache } from "./cache";
import { rateLimit } from "./rateLimit";
export interface Ranking {
position: number;
name: string;
average: number;
}
// `loadRanking` stands in for a slow database query.
export function createApp(loadRanking: (form: number) => Promise<Ranking[]>) {
const app = express();
app.use(express.json());
const rankings = new TtlCache<Ranking[]>(60_000);
app.use("/login", rateLimit({ windowMs: 15 * 60_000, max: 5 })); // slow down password guessing
app.use(rateLimit({ windowMs: 60_000, max: 100 })); // general limit for everything
app.get("/forms/:form/ranking", async (req, res) => {
const form = Number(req.params.form);
const ranking = await rankings.getOrLoad(`ranking:${form}`, () => loadRanking(form));
// Browsers may reuse this response for 30 s; Express adds an ETag so later requests can get 304.
res.set("Cache-Control", "private, max-age=30");
res.json(ranking);
});
app.post("/forms/:form/results", (req, res) => {
// ...save the result to the database, then drop the stale cached ranking
rankings.delete(`ranking:${Number(req.params.form)}`);
res.status(201).json({ saved: true });
});
app.post("/login", (_req, res) => {
res.status(401).json({ error: "Wrong email or password" });
});
return app;
}import { after, before, describe, it } from "node:test";
import assert from "node:assert/strict";
import type { AddressInfo } from "node:net";
import type { Server } from "node:http";
import { createApp, type Ranking } from "../src/app";
import { TtlCache } from "../src/cache";
describe("TtlCache", () => {
it("expires entries after the TTL", () => {
let now = 0;
const cache = new TtlCache<string>(1000, () => now);
cache.set("a", "x");
now = 999;
assert.equal(cache.get("a"), "x");
now = 1000;
assert.equal(cache.get("a"), undefined);
});
it("loads once for concurrent requests", async () => {
const cache = new TtlCache<number>(1000);
let calls = 0;
const load = async () => {
calls++;
return 42;
};
const values = await Promise.all([cache.getOrLoad("k", load), cache.getOrLoad("k", load), cache.getOrLoad("k", load)]);
assert.deepEqual(values, [42, 42, 42]);
assert.equal(calls, 1);
});
});
describe("HTTP caching and limits", () => {
let server: Server;
let base: string;
let queries = 0;
const ranking: Ranking[] = [{ position: 1, name: "Amina Hassan", average: 83.5 }];
before(async () => {
server = createApp(async () => {
queries++;
return ranking;
}).listen(0);
await new Promise((resolve) => server.once("listening", resolve));
base = `http://localhost:${(server.address() as AddressInfo).port}`;
});
after(() => server.close());
it("serves repeat requests from the cache and answers 304 for a matching ETag", async () => {
const first = await fetch(`${base}/forms/4/ranking`);
const etag = first.headers.get("etag")!;
await fetch(`${base}/forms/4/ranking`);
assert.equal(queries, 1);
// What a browser sends when its cached copy is older than max-age:
const again = await fetch(`${base}/forms/4/ranking`, { headers: { "If-None-Match": etag, "Cache-Control": "max-age=0" } });
assert.equal(again.status, 304);
await fetch(`${base}/forms/4/results`, { method: "POST" }); // invalidates
await fetch(`${base}/forms/4/ranking`);
assert.equal(queries, 2);
});
it("returns 429 with Retry-After after 5 login attempts", async () => {
const statuses: number[] = [];
let last: Response | undefined;
for (let i = 0; i < 6; i++) {
last = await fetch(`${base}/login`, { method: "POST" });
statuses.push(last.status);
}
assert.deepEqual(statuses, [401, 401, 401, 401, 401, 429]);
assert.equal(last!.headers.get("retry-after"), "900");
});
});Key points
- Cache-aside with a TTL; share in-flight loads and invalidate when data changes.
Cache-Controland ETags let browsers skip or shrink repeat requests.- Rate-limit per key with 429 and
Retry-After; be strict on login. Inject the clock for tests.
Exercise
Implement a token-bucket limiter: each key has a bucket of capacity tokens that refills at refillPerSecond, and each request spends one. Return whether the request is allowed and how long to wait, wrap it in middleware that limits per user id (or IP when logged out), and test bursts, refills, the capacity cap and separate keys with a fake clock.
Show solution
Try the exercise yourself first — then compare your approach with this one.
Instead of storing a timer per bucket, each bucket remembers when it was last updated and adds elapsed × refillPerSecond tokens when it is next used, capped at the capacity. That makes the limiter cheap and lets a fake clock drive the tests. A burst of capacity requests is allowed after a quiet period, then requests are spaced by the refill rate, which avoids the fixed window's double burst at the window edge.
import type { NextFunction, Request, Response } from "express";
interface Bucket {
tokens: number;
updatedAt: number;
}
// Each key gets a bucket of `capacity` tokens that refills steadily. A request spends one token.
// Unlike a fixed window, this allows short bursts but no "double burst" at a window boundary.
export class TokenBucket {
private buckets = new Map<string, Bucket>();
constructor(
private capacity: number,
private refillPerSecond: number,
private now: () => number = Date.now,
) {}
take(key: string): { allowed: boolean; remaining: number; retryAfterMs: number } {
const t = this.now();
const b = this.buckets.get(key) ?? { tokens: this.capacity, updatedAt: t };
const elapsed = (t - b.updatedAt) / 1000;
b.tokens = Math.min(this.capacity, b.tokens + elapsed * this.refillPerSecond);
b.updatedAt = t;
this.buckets.set(key, b);
if (b.tokens >= 1) {
b.tokens -= 1;
return { allowed: true, remaining: Math.floor(b.tokens), retryAfterMs: 0 };
}
const retryAfterMs = Math.ceil(((1 - b.tokens) / this.refillPerSecond) * 1000);
return { allowed: false, remaining: 0, retryAfterMs };
}
}
// Limit per logged-in user when there is one, otherwise per IP address.
export function tokenBucketLimit(bucket: TokenBucket) {
return (req: Request, res: Response, next: NextFunction) => {
const userId = (req as Request & { user?: { id: number } }).user?.id;
const result = bucket.take(userId !== undefined ? `user:${userId}` : `ip:${req.ip}`);
res.set("RateLimit-Remaining", String(result.remaining));
if (result.allowed) return next();
res.set("Retry-After", String(Math.ceil(result.retryAfterMs / 1000)));
res.status(429).json({ error: "Too many requests, slow down" });
};
}import { describe, it } from "node:test";
import assert from "node:assert/strict";
import { TokenBucket } from "../src/tokenBucket";
describe("TokenBucket", () => {
it("allows a burst up to capacity, then refuses", () => {
const bucket = new TokenBucket(3, 1, () => 0);
const results = [1, 2, 3, 4].map(() => bucket.take("amina").allowed);
assert.deepEqual(results, [true, true, true, false]);
});
it("refills over time and says when to retry", () => {
let now = 0;
const bucket = new TokenBucket(2, 0.5, () => now); // one token every 2 seconds
bucket.take("juma");
bucket.take("juma");
assert.equal(bucket.take("juma").retryAfterMs, 2000);
now = 2000;
assert.equal(bucket.take("juma").allowed, true);
assert.equal(bucket.take("juma").allowed, false);
});
it("never stores more than its capacity", () => {
let now = 0;
const bucket = new TokenBucket(2, 1, () => now);
now = 60_000; // a long quiet minute
const results = [1, 2, 3].map(() => bucket.take("neema").allowed);
assert.deepEqual(results, [true, true, false]);
});
it("keeps separate buckets per key", () => {
const bucket = new TokenBucket(1, 1, () => 0);
assert.equal(bucket.take("user:1").allowed, true);
assert.equal(bucket.take("user:2").allowed, true);
assert.equal(bucket.take("user:1").allowed, false);
});
});