Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -886,6 +886,7 @@ Notes:
- Continues on errors: failed pages are logged and the copy proceeds.
- Exclude patterns use simple globbing: `*` matches any sequence, `?` matches any single character, and special regex characters are treated literally.
- Large trees may take time; the CLI applies a small delay between sibling page creations to avoid rate limits (configurable via `--delay-ms`).
- If the server answers `429` (for example a Data Center instance limited to a few requests per second), the CLI paces its requests: a positive `Retry-After` makes every request wait, and requests are spaced further apart until the server accepts them, then speeded up again. Traversals such as `copy-tree`, `export` and `children --recursive` therefore finish under a low shared limit instead of failing after a few retries, and a server that keeps refusing still fails after about as long as before. Without throttling nothing is slowed down. A `503` on a read is retried per request, as before.
- Root title suffix defaults to ` (Copy)`; override with `--copy-suffix`. Child pages keep their original titles.
- Use `--fail-on-error` to exit non-zero if any page fails to copy.

Expand Down
89 changes: 73 additions & 16 deletions lib/confluence-client.js
Original file line number Diff line number Diff line change
Expand Up @@ -9,12 +9,21 @@ const { decodeHTML } = require('entities');
const MacroConverter = require('./macro-converter');
const { resolvePlantumlFormat } = require('./plantuml-format');
const { htmlToMarkdown, NAMED_ENTITIES } = require('./html-to-markdown');
const { RequestGate } = require('./request-gate');
const { isJsonMode } = require('./output');

const WRITE_FORMATS = ['auto', 'storage', 'html', 'markdown'];
const PAGE_LINK_LOOKUP_CONCURRENCY = 10;
const RETRY_SAFE_METHODS = new Set(['get', 'head', 'options']);
const MAX_RETRY_DELAY_MS = 60000;
// Throttled responses that are not evidence against a request (it was part of
// a burst that was already in flight, or the server kept serving others) do
// not use up its retries; this caps how many of those it may ride out, as
// many as its own retries, so a path that is always refused fails after the
// same number of attempts as before.
// New 429s without a Retry-After, beyond what one request would retry, before
// the gate gives up (see RequestGate).
const EXTRA_FRESH_THROTTLES = 4;

async function returnNullForNotFound(request) {
try {
Expand Down Expand Up @@ -155,8 +164,19 @@ class ConfluenceClient {
this.maxRetries = 3;
this.retryBaseDelayMs = 1000;

// Shared across every request this client makes, so a throttled response
// slows them all down instead of each retrying in lockstep.
this.gate = new RequestGate();
this.client.interceptors.request.use(async config => {
await this.gate.acquire(config);
return config;
});

this.client.interceptors.response.use(
response => response,
response => {
this.gate.reportSuccess();
return response;
},
async error => {
const requestConfig = error.config;
const status = error.response?.status;
Expand All @@ -172,10 +192,41 @@ class ConfluenceClient {
typeof requestConfig.data?.pipe !== 'function'
) {
const attempt = requestConfig.__retryCount || 0;
if (attempt < this.maxRetries) {
requestConfig.__retryCount = attempt + 1;
await this.sleep(this.retryDelayMs(error.response, attempt));
return this.client.request(requestConfig);
const free = requestConfig.__freeRetries || 0;
if (attempt < this.maxRetries && free < this.maxRetries) {
if (status !== 429) {
requestConfig.__retryCount = attempt + 1;
await this.sleep(this.retryDelayMs(error.response, attempt));
return this.client.request(requestConfig);
}
// A 429 is shared with every other request of this client: the
// gate paces their starts and honors a positive Retry-After for
// all of them. The backoff grows with new 429s seen by the whole
// client but never past what one request would reach on its
// own. A rejection from a burst this request was already part of,
// or while the server was still serving others, waits like the
// rest but keeps its retries, and once the gate opens its circuit
// (a run of new 429s with no success) nothing more is retried.
const level = this.gate.level;
const pauseMs = this.retryAfterMs(error.response);
const { retry, fresh, progressed, until } = this.gate.reportThrottle(requestConfig, {
pauseMs,
// A server that names the wait is believed after as many
// failures as one request would have given up on. Without one
// the spacing needs a few doublings to find the server's pace.
maxFresh: this.maxRetries + (pauseMs > 0 ? 1 : EXTRA_FRESH_THROTTLES)
});
if (retry) {
if (fresh && !progressed) {
requestConfig.__retryCount = attempt + 1;
} else {
requestConfig.__freeRetries = free + 1;
}
const exponent = Math.min(level, Math.max(0, this.maxRetries - 1));
await this.sleep(this.retryDelayMs(error.response, exponent));
this.gate.expire(until);
return this.client.request(requestConfig);
}
}
}
if (error.response?.status === 401) {
Expand Down Expand Up @@ -2522,20 +2573,26 @@ class ConfluenceClient {
* up all attempts within milliseconds.
*/
retryDelayMs(response, attempt) {
const retryAfter = response?.headers?.['retry-after'];
if (retryAfter !== undefined && retryAfter !== null && retryAfter !== '') {
const seconds = Number(retryAfter);
const waitMs = Number.isFinite(seconds)
? seconds * 1000
: Date.parse(retryAfter) - Date.now();
// NaN (unparsable) and non-positive waits fall through to the backoff.
if (waitMs > 0) {
return Math.min(waitMs, MAX_RETRY_DELAY_MS);
}
}
const retryAfterMs = this.retryAfterMs(response);
if (retryAfterMs > 0) return retryAfterMs;
return Math.min(this.retryBaseDelayMs * 2 ** attempt, MAX_RETRY_DELAY_MS);
}

/**
* The wait a response asks for with Retry-After (seconds or HTTP-date),
* capped at the maximum delay; 0 when it names no usable wait.
*/
retryAfterMs(response) {
const retryAfter = response?.headers?.['retry-after'];
if (retryAfter === undefined || retryAfter === null || retryAfter === '') return 0;
const seconds = Number(retryAfter);
const waitMs = Number.isFinite(seconds)
? seconds * 1000
: Date.parse(retryAfter) - Date.now();
// NaN (unparsable) and non-positive waits name no wait.
return waitMs > 0 ? Math.min(waitMs, MAX_RETRY_DELAY_MS) : 0;
}

/**
* Accumulate results across start/limit pages until the server stops
* returning a next link or `maxResults` is reached (null = unlimited).
Expand Down
146 changes: 146 additions & 0 deletions lib/request-gate.js
Original file line number Diff line number Diff line change
@@ -0,0 +1,146 @@
// Coordinates the requests of one client that share a rate-limited server.
//
// Each request used to back off on its own, so a burst of N concurrent
// requests that hit a low shared limit (Confluence Data Center can allow as
// little as 3 requests per second) was rejected N times over, retried in
// lockstep, rejected again, and ran out of retries even though a patient
// client would have finished. Capping concurrency does not fix that: a fast
// server answers at once, so even one request at a time arrives faster than
// the limit. The gate paces request *starts* instead, and shares what it
// learns across every request the client makes. It only ever sees 429
// responses; other retryable statuses keep their independent backoff.
//
// - A positive Retry-After pauses every request, not just the one that saw
// it. Without one (Data Center answers 429 with `Retry-After: 0`) there is
// nothing to wait for, so only the spacing below changes.
// - A new 429 doubles the minimum spacing between request starts, and each
// success shrinks it again. A rejection from a burst that was already in
// flight when the spacing last changed is stale: it neither widens the
// spacing again nor uses up the request's own retries.
// - After `maxFresh` new 429s with no success in between, the gate opens a
// circuit: nothing more is retried or delayed until a request succeeds, so
// a server that keeps refusing fails as fast as a single request would.
// - Without throttling there is no pause and no spacing: starts are released
// back to back, so concurrency and throughput are untouched and the gate
// only adds a microtask to each request.

// Spacing applied after the first throttle, its ceiling, and the factor each
// success applies to it. Below MIN_INTERVAL_MS the spacing is dropped. The
// decay is deliberately memoryless: a 429 that has nothing to do with the
// request rate (a flaky proxy) is forgotten within a handful of successes
// instead of leaving the client throttled, and against a real limit the
// spacing settles around the server's pace, crossing it now and then.
const INITIAL_INTERVAL_MS = 100;
const MAX_INTERVAL_MS = 2000;
const SUCCESS_DECAY = 0.9;
const MIN_INTERVAL_MS = 25;

// The gate waits with its own timer, not ConfluenceClient.sleep, which a
// caller may stub to skip the per-request backoff.
const defaultSleep = (ms) => new Promise((resolve) => setTimeout(resolve, ms));

class RequestGate {
constructor({ now = () => Date.now(), sleep = defaultSleep } = {}) {
this.now = now;
this.sleep = sleep;
this.intervalMs = 0;
this.lastStart = -Infinity;
// Requests start one at a time, in arrival order.
this.tail = Promise.resolve();
this.pauseUntil = 0;
// New 429s since the last success; scales a request's own backoff and
// trips the circuit.
this.level = 0;
this.open = false;
// Bumped whenever the spacing widens. A request stamped with an older
// epoch was sent before the change that its rejection would cause.
this.epoch = 0;
// Successful responses so far; a request that sees this grow while it is
// throttled knows the server is still making progress for others.
this.successes = 0;
}

// Wait for this request's turn to start, then for any shared pause and the
// spacing since the previous start. `config` is stamped so a rejection can
// later be told apart as new or stale.
async acquire(config) {
const previous = this.tail;
let done;
this.tail = new Promise((resolve) => { done = resolve; });
try {
await previous;
// Track the time waited until rather than re-reading the clock, so the
// loop ends even if `sleep` returns early; it only repeats when the
// pause was extended while this request slept.
let startAt = this.now();
while (!this.open) {
const target = Math.max(this.pauseUntil, this.lastStart + this.intervalMs);
if (target <= startAt) break;
await this.sleep(target - startAt);
startAt = target;
}
this.lastStart = Math.max(startAt, this.now());
config.__gateEpoch = this.epoch;
if (config.__gateSuccesses === undefined) config.__gateSuccesses = this.successes;
} finally {
done();
}
}

// Record a 429 that may be retried. `pauseMs` is a positive Retry-After
// (0 if there was none) and `maxFresh` how many new 429s in a row are
// allowed before the circuit opens. Returns `retry` (false once the circuit
// is open), `fresh` (a new 429 rather than one from a burst already in
// flight), `progressed` (some request has succeeded since this one first
// started, so the rejection is not evidence against it) and `until` (the
// shared pause it set, for `expire`). Only a fresh rejection with no
// progress should count against the request's own retries.
reportThrottle(config, { pauseMs = 0, maxFresh = Infinity } = {}) {
const fresh = config.__gateEpoch === this.epoch;
const progressed = this.successes > (config.__gateSuccesses ?? this.successes);
let until = 0;
if (fresh) {
this.epoch++;
this.level++;
this.intervalMs = this.intervalMs === 0
? INITIAL_INTERVAL_MS
: Math.min(this.intervalMs * 2, MAX_INTERVAL_MS);
if (this.level >= maxFresh) {
this.open = true;
}
}
// Nothing waits on a pause once the circuit is open, and one set now
// would come back to life when a later success closes it.
if (pauseMs > 0 && !this.open) {
until = Math.max(this.pauseUntil, this.now() + pauseMs);
this.pauseUntil = until;
}
return { retry: !this.open, fresh, progressed, until };
}

// The caller that set the pause has waited it out itself; clear it unless
// someone has extended it since.
expire(until) {
if (until && this.pauseUntil === until) this.pauseUntil = 0;
}

reportSuccess() {
this.successes++;
this.level = 0;
if (this.open) {
this.open = false;
this.pauseUntil = 0;
}
if (this.intervalMs === 0) return;
const next = this.intervalMs * SUCCESS_DECAY;
this.intervalMs = next < MIN_INTERVAL_MS ? 0 : next;
}
}

module.exports = {
RequestGate,
INITIAL_INTERVAL_MS,
MAX_INTERVAL_MS,
MIN_INTERVAL_MS,
SUCCESS_DECAY
};
Loading
Loading