From 59630b9d4af6bb049034010ca26d24ef0d26ed6e Mon Sep 17 00:00:00 2001 From: Austin Burdine Date: Fri, 2 Oct 2026 11:16:22 -0400 Subject: [PATCH 1/6] Added freeze() and cached root URL parsing to url-utils Ghost's sitemap build calls url-utils several times per post. Profiling a 300k-post site showed repeated getter calls (Ghost's getSubdir parses the configured url on every call) and repeated `new URL()` parsing of the same handful of root URLs as a large part of the cost. - added `freeze()`/`unfreeze()`/`isFrozen` and a `frozen` constructor option that snapshot `getSiteUrl`/`getSubdir`/`getAdminUrl`. Opt-in; unfrozen behaviour is unchanged - added `parseRootUrl`, a bounded cache of parsed root URLs, used wherever a root/base URL was parsed with `new URL()` - cached the subdirectory regex in `deduplicateSubdirectory` Both caches are pure functions of their input string so can't go stale. Co-Authored-By: Claude Opus 5.5 --- packages/url-utils/README.md | 16 +++++ packages/url-utils/src/UrlUtils.ts | 65 +++++++++++++++++ .../src/utils/absolute-to-relative.ts | 5 +- .../src/utils/deduplicate-subdirectory.ts | 27 +++++--- packages/url-utils/src/utils/index.ts | 3 + .../url-utils/src/utils/parse-root-url.ts | 55 +++++++++++++++ .../plaintext-absolute-to-transform-ready.ts | 4 +- .../src/utils/relative-to-absolute.ts | 3 +- .../src/utils/relative-to-transform-ready.ts | 3 +- .../src/utils/strip-subdirectory-from-path.ts | 6 +- .../src/utils/transform-ready-to-relative.ts | 4 +- .../url-utils/test/unit/url-utils.test.js | 69 +++++++++++++++++++ .../utils/deduplicate-subdirectory.test.js | 8 +++ .../test/unit/utils/parse-root-url.test.js | 50 ++++++++++++++ 14 files changed, 299 insertions(+), 19 deletions(-) create mode 100644 packages/url-utils/src/utils/parse-root-url.ts create mode 100644 packages/url-utils/test/unit/utils/parse-root-url.test.js diff --git a/packages/url-utils/README.md b/packages/url-utils/README.md index ec889aa48..95f7554e7 100644 --- a/packages/url-utils/README.md +++ b/packages/url-utils/README.md @@ -11,6 +11,22 @@ or ## Usage +### Freezing URLs + +When the site, subdirectory and admin URLs never change at runtime (e.g. in production), mark them as frozen so url-utils can skip calling the URL getters: + +```js +const urlUtils = new UrlUtils({getSiteUrl, getSubdir, getAdminUrl, frozen: true}); + +// or at any time after creation +urlUtils.freeze(); + +// restore the original getters (e.g. after changing config in tests) +urlUtils.unfreeze(); +``` + +While frozen, `getSiteUrl()`, `getSubdir()` and `getAdminUrl()` return the values captured at freeze time. Calling `freeze()` again re-captures the current values. + ## Develop diff --git a/packages/url-utils/src/UrlUtils.ts b/packages/url-utils/src/UrlUtils.ts index d607fabc3..eeb947cfc 100644 --- a/packages/url-utils/src/UrlUtils.ts +++ b/packages/url-utils/src/UrlUtils.ts @@ -60,6 +60,13 @@ interface UrlUtilsOptions { media?: string | null; }; cardTransformers?: MobiledocCardTransformer[]; + frozen?: boolean; +} + +interface UrlGetters { + getSubdir: () => string; + getSiteUrl: () => string; + getAdminUrl: () => string; } // similar to Object.assign but will not override defaults if a source value is undefined @@ -78,6 +85,7 @@ export default class UrlUtils { public getSubdir: () => string; public getSiteUrl: () => string; public getAdminUrl: () => string; + private _unfrozenGetters: UrlGetters | null = null; /** * Initialization method to pass in URL configurations @@ -96,6 +104,7 @@ export default class UrlUtils { * @param {string} [options.assetBaseUrls.image] image asset CDN base URL * @param {string} [options.assetBaseUrls.files] files asset CDN base URL * @param {string} [options.assetBaseUrls.media] media asset CDN base URL + * @param {boolean} [options.frozen=false] freeze url getters on creation, see `freeze()` */ constructor(options: UrlUtilsOptions = {}) { const defaultOptions: UrlUtilsConfig = { @@ -120,6 +129,62 @@ export default class UrlUtils { this.getSubdir = options.getSubdir || (() => ''); this.getSiteUrl = options.getSiteUrl || (() => ''); this.getAdminUrl = options.getAdminUrl || (() => ''); + + if (options.frozen) { + this.freeze(); + } + } + + /** + * Mark the site, subdirectory and admin URLs as frozen: they won't change for + * the lifetime of this instance (or until `unfreeze()` is called). + * + * While frozen, `getSubdir`, `getSiteUrl` and `getAdminUrl` return a snapshot + * taken at freeze time rather than calling the configured getters. + * + * Only freeze when the underlying config is static, e.g. in production. If the + * URLs can change at runtime (tests that swap config) leave unfrozen, or call + * `unfreeze()`/`freeze()` again after changing config. + */ + freeze(): this { + if (this._unfrozenGetters) { + this.unfreeze(); + } + + const getters: UrlGetters = { + getSubdir: this.getSubdir, + getSiteUrl: this.getSiteUrl, + getAdminUrl: this.getAdminUrl + }; + + const subdir = getters.getSubdir(); + const siteUrl = getters.getSiteUrl(); + const adminUrl = getters.getAdminUrl(); + + this._unfrozenGetters = getters; + this.getSubdir = () => subdir; + this.getSiteUrl = () => siteUrl; + this.getAdminUrl = () => adminUrl; + + return this; + } + + /** + * Restore the original URL getters. + */ + unfreeze(): this { + if (this._unfrozenGetters) { + this.getSubdir = this._unfrozenGetters.getSubdir; + this.getSiteUrl = this._unfrozenGetters.getSiteUrl; + this.getAdminUrl = this._unfrozenGetters.getAdminUrl; + this._unfrozenGetters = null; + } + + return this; + } + + get isFrozen(): boolean { + return this._unfrozenGetters !== null; } private _assetOptionDefaults(): BaseUrlOptionsInput & { diff --git a/packages/url-utils/src/utils/absolute-to-relative.ts b/packages/url-utils/src/utils/absolute-to-relative.ts index 02efc1009..f0d50ebc5 100644 --- a/packages/url-utils/src/utils/absolute-to-relative.ts +++ b/packages/url-utils/src/utils/absolute-to-relative.ts @@ -1,4 +1,5 @@ import {URL} from 'url'; +import parseRootUrl, {type ParsedRootUrl} from './parse-root-url'; import stripSubdirectoryFromPath from './strip-subdirectory-from-path'; export interface AbsoluteToRelativeOptions { @@ -37,11 +38,11 @@ const absoluteToRelative = function absoluteToRelative(url: string, rootUrl?: st } let parsedUrl: URL; - let parsedRoot: URL | undefined; + let parsedRoot: ParsedRootUrl | undefined; try { parsedUrl = new URL(url, 'http://relative'); - parsedRoot = parsedUrl.origin === 'null' ? undefined : new URL(rootUrl || parsedUrl.origin); + parsedRoot = parsedUrl.origin === 'null' ? undefined : parseRootUrl(rootUrl || parsedUrl.origin); // return the url as-is if it was relative or non-http if (parsedUrl.origin === 'null' || parsedUrl.origin === 'http://relative') { diff --git a/packages/url-utils/src/utils/deduplicate-subdirectory.ts b/packages/url-utils/src/utils/deduplicate-subdirectory.ts index c6b9c6e84..0b55d716d 100644 --- a/packages/url-utils/src/utils/deduplicate-subdirectory.ts +++ b/packages/url-utils/src/utils/deduplicate-subdirectory.ts @@ -1,4 +1,6 @@ -import {URL} from 'url'; +import parseRootUrl from './parse-root-url'; + +const subdirRegexCache = new Map(); /** * Remove duplicated directories from the start of a path or url's path @@ -13,19 +15,28 @@ const deduplicateSubdirectory = function deduplicateSubdirectory(url: string, ro rootUrl = `${rootUrl}/`; } - const parsedRoot = new URL(rootUrl); + const {pathname} = parseRootUrl(rootUrl); // do nothing if rootUrl does not have a subdirectory - if (parsedRoot.pathname === '/') { + if (pathname === '/') { return url; } - const subdir = parsedRoot.pathname.replace(/(^\/|\/$)+/g, ''); - // we can have subdirs that match TLDs so we need to restrict matches to - // duplicates that start with a / or the beginning of the url - const subdirRegex = new RegExp(`(^|/)${subdir}/${subdir}(/|$)`); + let cached = subdirRegexCache.get(pathname); + + if (!cached) { + const subdir = pathname.replace(/(^\/|\/$)+/g, ''); + // we can have subdirs that match TLDs so we need to restrict matches to + // duplicates that start with a / or the beginning of the url + cached = {subdir, regex: new RegExp(`(^|/)${subdir}/${subdir}(/|$)`)}; + + if (subdirRegexCache.size >= 100) { + subdirRegexCache.clear(); + } + subdirRegexCache.set(pathname, cached); + } - return url.replace(subdirRegex, `$1${subdir}/`); + return url.replace(cached.regex, `$1${cached.subdir}/`); }; export default deduplicateSubdirectory; diff --git a/packages/url-utils/src/utils/index.ts b/packages/url-utils/src/utils/index.ts index 8498d6503..11d9d2cdc 100644 --- a/packages/url-utils/src/utils/index.ts +++ b/packages/url-utils/src/utils/index.ts @@ -23,6 +23,7 @@ import mobiledocRelativeToAbsolute from './mobiledoc-relative-to-absolute'; import mobiledocAbsoluteToTransformReady from './mobiledoc-absolute-to-transform-ready'; import mobiledocRelativeToTransformReady from './mobiledoc-relative-to-transform-ready'; import mobiledocToTransformReady from './mobiledoc-to-transform-ready'; +import parseRootUrl from './parse-root-url'; import plaintextAbsoluteToTransformReady from './plaintext-absolute-to-transform-ready'; import plaintextRelativeToTransformReady from './plaintext-relative-to-transform-ready'; import plaintextToTransformReady from './plaintext-to-transform-ready'; @@ -61,6 +62,7 @@ export { mobiledocAbsoluteToTransformReady, mobiledocRelativeToTransformReady, mobiledocToTransformReady, + parseRootUrl, plaintextAbsoluteToTransformReady, plaintextRelativeToTransformReady, plaintextToTransformReady, @@ -100,6 +102,7 @@ const utils = { mobiledocRelativeToAbsolute, mobiledocRelativeToTransformReady, mobiledocToTransformReady, + parseRootUrl, plaintextAbsoluteToTransformReady, plaintextRelativeToTransformReady, plaintextToTransformReady, diff --git a/packages/url-utils/src/utils/parse-root-url.ts b/packages/url-utils/src/utils/parse-root-url.ts new file mode 100644 index 000000000..8bf852d8d --- /dev/null +++ b/packages/url-utils/src/utils/parse-root-url.ts @@ -0,0 +1,55 @@ +import {URL} from 'url'; + +export interface ParsedRootUrl { + readonly href: string; + readonly origin: string; + readonly protocol: string; + readonly host: string; + readonly hostname: string; + readonly pathname: string; +} + +// Root URLs (site url, admin url, CDN base urls) are a tiny, near-static set of +// strings that get re-parsed on almost every url-utils call. Parsing is a pure +// function of the input string so results can be cached without any risk of +// going stale. Bounded so arbitrary input can't grow the cache unchecked. +const MAX_ENTRIES = 100; +const cache = new Map(); + +/** + * Parse a root URL, returning a cached immutable snapshot of the parts url-utils uses. + * Throws the same errors as `new URL()` for invalid input (errors are not cached). + * + * @param {string} rootUrl + * @returns {ParsedRootUrl} + */ +function parseRootUrl(rootUrl: string): ParsedRootUrl { + let parsed = cache.get(rootUrl); + + if (parsed) { + return parsed; + } + + const url = new URL(rootUrl); + parsed = Object.freeze({ + href: url.href, + origin: url.origin, + protocol: url.protocol, + host: url.host, + hostname: url.hostname, + pathname: url.pathname + }); + + if (cache.size >= MAX_ENTRIES) { + cache.clear(); + } + cache.set(rootUrl, parsed); + + return parsed; +} + +parseRootUrl.clearCache = function clearCache(): void { + cache.clear(); +}; + +export default parseRootUrl; diff --git a/packages/url-utils/src/utils/plaintext-absolute-to-transform-ready.ts b/packages/url-utils/src/utils/plaintext-absolute-to-transform-ready.ts index 39fec1d75..d09e7c029 100644 --- a/packages/url-utils/src/utils/plaintext-absolute-to-transform-ready.ts +++ b/packages/url-utils/src/utils/plaintext-absolute-to-transform-ready.ts @@ -2,7 +2,7 @@ import type {AbsoluteToTransformReadyOptionsInput, BaseUrlOptionsInput} from './ import absoluteToTransformReady from './absolute-to-transform-ready'; import buildEarlyExitMatchModule from './build-early-exit-match'; const {escapeRegExp} = buildEarlyExitMatchModule; -import {URL} from 'url'; +import parseRootUrl from './parse-root-url'; type PlaintextAbsoluteToTransformReadyOptions = AbsoluteToTransformReadyOptionsInput & BaseUrlOptionsInput; type PlaintextAbsoluteToTransformReadyOptionsInput = Partial; @@ -13,7 +13,7 @@ function buildLinkRegex(rootUrl: string, options: PlaintextAbsoluteToTransformRe .filter((value): value is string => Boolean(value)); const patterns = baseUrls.map((baseUrl: string) => { - const parsed = new URL(baseUrl); + const parsed = parseRootUrl(baseUrl); const escapedUrl = escapeRegExp(`${parsed.hostname}${parsed.pathname.replace(/\/$/, '')}`); return escapedUrl; }); diff --git a/packages/url-utils/src/utils/relative-to-absolute.ts b/packages/url-utils/src/utils/relative-to-absolute.ts index a8e14a716..6343c1836 100644 --- a/packages/url-utils/src/utils/relative-to-absolute.ts +++ b/packages/url-utils/src/utils/relative-to-absolute.ts @@ -1,6 +1,7 @@ import type {SecureOptions, SecureOptionsInput} from './types'; import {URL} from 'url'; import urlJoin from './url-join'; +import parseRootUrl from './parse-root-url'; export type RelativeToAbsoluteOptions = SecureOptions; export type RelativeToAbsoluteOptionsInput = SecureOptionsInput; @@ -92,7 +93,7 @@ const relativeToAbsolute = function relativeToAbsolute( rootUrl = `${rootUrl}/`; } - const parsedRootUrl: URL = new URL(rootUrl); + const parsedRootUrl = parseRootUrl(rootUrl); const basePath = path.startsWith('/') ? '' : (finalItemPath || ''); const fullPath = urlJoin([parsedRootUrl.pathname, basePath, path], {rootUrl}); const absoluteUrl = new URL(fullPath, rootUrl); diff --git a/packages/url-utils/src/utils/relative-to-transform-ready.ts b/packages/url-utils/src/utils/relative-to-transform-ready.ts index b1010afac..f6a8ad1f5 100644 --- a/packages/url-utils/src/utils/relative-to-transform-ready.ts +++ b/packages/url-utils/src/utils/relative-to-transform-ready.ts @@ -3,6 +3,7 @@ import type { } from './types'; import relativeToAbsolute from './relative-to-absolute'; import {URL} from 'url'; +import parseRootUrl from './parse-root-url'; export interface RelativeToTransformReadyOptions extends TransformReadyReplacementOptions { staticImageUrlPrefix: string; @@ -44,7 +45,7 @@ const relativeToTransformReady = function ( return url; } - const rootUrl: URL = new URL(root); + const rootUrl = parseRootUrl(root); const rootPathname = rootUrl.pathname.replace(/\/$/, ''); // only convert to transform-ready if root url has no subdirectory or the subdirectory matches diff --git a/packages/url-utils/src/utils/strip-subdirectory-from-path.ts b/packages/url-utils/src/utils/strip-subdirectory-from-path.ts index acec9bf00..65977d13e 100644 --- a/packages/url-utils/src/utils/strip-subdirectory-from-path.ts +++ b/packages/url-utils/src/utils/strip-subdirectory-from-path.ts @@ -1,4 +1,4 @@ -import {URL} from 'url'; +import parseRootUrl, {type ParsedRootUrl} from './parse-root-url'; /** * Removes the directory in the root url from the relative path @@ -13,10 +13,10 @@ const stripSubdirectoryFromPath = function stripSubdirectoryFromPath(path: strin rootUrl = `${rootUrl}/`; } - let parsedRoot: URL; + let parsedRoot: ParsedRootUrl; try { - parsedRoot = new URL(rootUrl); + parsedRoot = parseRootUrl(rootUrl); } catch { return path; } diff --git a/packages/url-utils/src/utils/transform-ready-to-relative.ts b/packages/url-utils/src/utils/transform-ready-to-relative.ts index 7678d57da..2de4c0dac 100644 --- a/packages/url-utils/src/utils/transform-ready-to-relative.ts +++ b/packages/url-utils/src/utils/transform-ready-to-relative.ts @@ -1,5 +1,5 @@ import type {TransformReadyReplacementOptions, TransformReadyReplacementOptionsInput} from './types'; -import {URL} from 'url'; +import parseRootUrl from './parse-root-url'; function escapeRegExp(string: string): string { return string.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); @@ -21,7 +21,7 @@ const transformReadyToRelative = function ( return str; } - const rootURL = new URL(root); + const rootURL = parseRootUrl(root); // subdir with no trailing slash because we'll always have a trailing slash after the magic string const subdir = rootURL.pathname.replace(/\/$/, ''); diff --git a/packages/url-utils/test/unit/url-utils.test.js b/packages/url-utils/test/unit/url-utils.test.js index 35ebca23a..8e6698370 100644 --- a/packages/url-utils/test/unit/url-utils.test.js +++ b/packages/url-utils/test/unit/url-utils.test.js @@ -472,6 +472,75 @@ describe('UrlUtils', function () { }); }); + describe('freeze', function () { + it('is not frozen by default', function () { + assert.equal(utils.isFrozen, false); + }); + + it('can be frozen via constructor option', function () { + const frozenUtils = new UrlUtils({getSiteUrl: () => 'http://my-ghost-blog.com/', frozen: true}); + assert.equal(frozenUtils.isFrozen, true); + }); + + it('snapshots url getters when frozen', function () { + fakeConfig.url = 'http://my-ghost-blog.com/blog/'; + fakeConfig.adminUrl = 'http://admin.ghost-blog.com'; + assert.equal(utils.freeze(), utils); + + nconf.get.resetHistory(); + fakeConfig.url = 'http://changed.com/'; + fakeConfig.adminUrl = 'http://admin.changed.com'; + + assert.equal(utils.getSiteUrl(), 'http://my-ghost-blog.com/blog/'); + assert.equal(utils.getSubdir(), '/blog'); + assert.equal(utils.getAdminUrl(), 'http://admin.ghost-blog.com/blog/'); + assert.equal(utils.urlFor('home', true), 'http://my-ghost-blog.com/blog/'); + assert.equal(utils.urlFor({relativeUrl: '/post/'}), '/blog/post/'); + assert.equal(utils.urlFor('admin', true), 'http://admin.ghost-blog.com/blog/ghost/'); + assert.equal(nconf.get.callCount, 0); + }); + + it('createUrl reflects config changes when not frozen', function () { + assert.equal(utils.createUrl('/tag/', true), 'http://my-ghost-blog.com/tag/'); + fakeConfig.url = 'http://changed.com/'; + assert.equal(utils.createUrl('/tag/', true), 'http://changed.com/tag/'); + }); + + it('restores getters on unfreeze', function () { + utils.freeze(); + assert.equal(utils.createUrl('/tag/', true), 'http://my-ghost-blog.com/tag/'); + + fakeConfig.url = 'http://changed.com/'; + assert.equal(utils.unfreeze(), utils); + + assert.equal(utils.isFrozen, false); + assert.equal(utils.getSiteUrl(), 'http://changed.com/'); + assert.equal(utils.createUrl('/tag/', true), 'http://changed.com/tag/'); + }); + + it('re-snapshots when frozen again', function () { + utils.freeze(); + assert.equal(utils.createUrl('/tag/', true), 'http://my-ghost-blog.com/tag/'); + + fakeConfig.url = 'http://changed.com/'; + utils.freeze(); + + assert.equal(utils.isFrozen, true); + assert.equal(utils.getSiteUrl(), 'http://changed.com/'); + assert.equal(utils.createUrl('/tag/', true), 'http://changed.com/tag/'); + + utils.unfreeze(); + fakeConfig.url = 'http://changed-again.com/'; + assert.equal(utils.getSiteUrl(), 'http://changed-again.com/'); + }); + + it('unfreeze is a no-op when not frozen', function () { + const getSiteUrl = utils.getSiteUrl; + utils.unfreeze(); + assert.equal(utils.getSiteUrl, getSiteUrl); + }); + }); + describe('urlFor home with trailingSlash:false', function () { it('returns home url without trailing slash', function () { const result = utils.urlFor('home', {trailingSlash: false}, true); diff --git a/packages/url-utils/test/unit/utils/deduplicate-subdirectory.test.js b/packages/url-utils/test/unit/utils/deduplicate-subdirectory.test.js index fb52f72b8..04e3bb77a 100644 --- a/packages/url-utils/test/unit/utils/deduplicate-subdirectory.test.js +++ b/packages/url-utils/test/unit/utils/deduplicate-subdirectory.test.js @@ -137,3 +137,11 @@ describe('utils: deduplicateSubdirectory()', function () { }); }); }); + +describe('utils: deduplicateSubdirectory() cache', function () { + it('keeps working after the subdirectory cache is cleared', function () { + for (let i = 0; i < 101; i++) { + deduplicateSubdirectory(`/sub${i}/sub${i}/`, `https://example.com/sub${i}/`).should.equal(`/sub${i}/`); + } + }); +}); diff --git a/packages/url-utils/test/unit/utils/parse-root-url.test.js b/packages/url-utils/test/unit/utils/parse-root-url.test.js new file mode 100644 index 000000000..fd0d0a82c --- /dev/null +++ b/packages/url-utils/test/unit/utils/parse-root-url.test.js @@ -0,0 +1,50 @@ +// Switch these lines once there are useful utils +// const testUtils = require('../../utils'); +require('../../utils'); + +const parseRootUrl = require('../../../lib/utils/parse-root-url').default; + +describe('utils: parseRootUrl()', function () { + beforeEach(function () { + parseRootUrl.clearCache(); + }); + + it('returns the parsed parts of a url', function () { + const parsed = parseRootUrl('https://example.com:2368/blog/'); + + parsed.href.should.equal('https://example.com:2368/blog/'); + parsed.origin.should.equal('https://example.com:2368'); + parsed.protocol.should.equal('https:'); + parsed.host.should.equal('example.com:2368'); + parsed.hostname.should.equal('example.com'); + parsed.pathname.should.equal('/blog/'); + }); + + it('returns the same cached object for repeated calls', function () { + const first = parseRootUrl('https://example.com/'); + const second = parseRootUrl('https://example.com/'); + + first.should.equal(second); + }); + + it('returns an immutable result', function () { + const parsed = parseRootUrl('https://example.com/'); + + Object.isFrozen(parsed).should.be.true(); + }); + + it('throws for invalid urls and does not cache them', function () { + (() => parseRootUrl('not a url')).should.throw(TypeError); + (() => parseRootUrl('not a url')).should.throw(TypeError); + }); + + it('clears the cache once it reaches its size limit', function () { + const first = parseRootUrl('https://example.com/'); + + for (let i = 0; i < 100; i++) { + parseRootUrl(`https://example.com/${i}/`); + } + + parseRootUrl('https://example.com/').should.not.equal(first); + }); +}); From c967e95c144a43bb94a14ba8e27f837c3dfb42e4 Mon Sep 17 00:00:00 2001 From: Austin Burdine Date: Fri, 2 Oct 2026 11:17:50 -0400 Subject: [PATCH 2/6] Sped up transformReadyToAbsolute Sitemaps call `transformReadyToAbsolute` twice per post (post url and feature image). Previously every call built three option objects and looked up the site url before the no-placeholder early return, then built a new RegExp and sliced the remainder of the string for every match. - `UrlUtils#transformReadyToAbsolute` returns early before computing the site url or options when there's nothing to replace, and builds its default asset options once rather than per call. Options are only merged when passed - the util now walks the string with `indexOf`, checks asset prefixes in place without slicing, and memoizes trailing-slash stripping of base urls Output is unchanged, verified by equivalence tests against the 5.3.0 implementation. Cuts `transformReadyToAbsolute` time on 300k posts from ~420ms to ~25ms. Co-Authored-By: Claude Opus 5.5 --- packages/url-utils/src/UrlUtils.ts | 64 +++++-- .../src/utils/transform-ready-to-absolute.ts | 128 +++++++++---- ...form-ready-to-absolute-equivalence.test.js | 172 ++++++++++++++++++ .../legacy/transform-ready-to-absolute.js | 73 ++++++++ 4 files changed, 384 insertions(+), 53 deletions(-) create mode 100644 packages/url-utils/test/unit/utils/transform-ready-to-absolute-equivalence.test.js create mode 100644 packages/url-utils/test/utils/legacy/transform-ready-to-absolute.js diff --git a/packages/url-utils/src/UrlUtils.ts b/packages/url-utils/src/UrlUtils.ts index eeb947cfc..76da832ef 100644 --- a/packages/url-utils/src/UrlUtils.ts +++ b/packages/url-utils/src/UrlUtils.ts @@ -14,7 +14,12 @@ import type {RelativeToAbsoluteOptionsInput} from './utils/relative-to-absolute' import type {AbsoluteToTransformReadyOptionsInput as AbsoluteToTransformReadyOptionsInputType} from './utils/absolute-to-transform-ready'; import type {RelativeToTransformReadyOptionsInput as RelativeToTransformReadyOptionsInputType} from './utils/relative-to-transform-ready'; import type {ToTransformReadyOptions} from './utils/to-transform-ready'; -import type {TransformReadyToAbsoluteOptionsInput} from './utils/transform-ready-to-absolute'; +import { + DEFAULT_OPTIONS as TRANSFORM_READY_TO_ABSOLUTE_DEFAULTS, + replaceTransformReadyPlaceholders, + type TransformReadyToAbsoluteOptions, + type TransformReadyToAbsoluteOptionsInput +} from './utils/transform-ready-to-absolute'; import type {TransformReadyReplacementOptionsInput as TransformReadyToRelativeOptionsInput} from './utils/types'; interface ExpressResponse { @@ -63,6 +68,12 @@ interface UrlUtilsOptions { frozen?: boolean; } +type AssetOptionDefaults = BaseUrlOptionsInput & { + staticImageUrlPrefix: string; + staticFilesUrlPrefix: string; + staticMediaUrlPrefix: string; +}; + interface UrlGetters { getSubdir: () => string; getSiteUrl: () => string; @@ -86,6 +97,9 @@ export default class UrlUtils { public getSiteUrl: () => string; public getAdminUrl: () => string; private _unfrozenGetters: UrlGetters | null = null; + // asset options are fixed at construction so are only built once + private readonly _assetOptionDefaults: Readonly; + private readonly _transformReadyToAbsoluteDefaults: Readonly; /** * Initialization method to pass in URL configurations @@ -126,6 +140,18 @@ export default class UrlUtils { media: assetBaseUrls.media || null }; + this._assetOptionDefaults = Object.freeze({ + staticImageUrlPrefix: this._config.staticImageUrlPrefix, + staticFilesUrlPrefix: this._config.staticFilesUrlPrefix, + staticMediaUrlPrefix: this._config.staticMediaUrlPrefix, + imageBaseUrl: this._assetBaseUrls.image, + filesBaseUrl: this._assetBaseUrls.files, + mediaBaseUrl: this._assetBaseUrls.media + }); + this._transformReadyToAbsoluteDefaults = Object.freeze( + Object.assign({}, TRANSFORM_READY_TO_ABSOLUTE_DEFAULTS, this._assetOptionDefaults) + ); + this.getSubdir = options.getSubdir || (() => ''); this.getSiteUrl = options.getSiteUrl || (() => ''); this.getAdminUrl = options.getAdminUrl || (() => ''); @@ -187,23 +213,8 @@ export default class UrlUtils { return this._unfrozenGetters !== null; } - private _assetOptionDefaults(): BaseUrlOptionsInput & { - staticImageUrlPrefix: string; - staticFilesUrlPrefix: string; - staticMediaUrlPrefix: string; - } { - return { - staticImageUrlPrefix: this._config.staticImageUrlPrefix, - staticFilesUrlPrefix: this._config.staticFilesUrlPrefix, - staticMediaUrlPrefix: this._config.staticMediaUrlPrefix, - imageBaseUrl: this._assetBaseUrls.image || null, - filesBaseUrl: this._assetBaseUrls.files || null, - mediaBaseUrl: this._assetBaseUrls.media || null - }; - } - private _buildAssetOptions(additionalDefaults: Record = {}, options?: Record): Record { - return assignOptions({}, this._assetOptionDefaults(), additionalDefaults, options || {}); + return assignOptions({}, this._assetOptionDefaults, additionalDefaults, options || {}); } getProtectedSlugs(): string[] { @@ -427,8 +438,23 @@ export default class UrlUtils { } transformReadyToAbsolute(url: string, options?: TransformReadyToAbsoluteOptionsInput): string { - const _options = this._buildAssetOptions({}, options) as TransformReadyToAbsoluteOptionsInput; - return utils.transformReadyToAbsolute(url, this.getSiteUrl(), _options); + if (options) { + const _options = this._buildAssetOptions({}, options) as TransformReadyToAbsoluteOptionsInput; + return utils.transformReadyToAbsolute(url, this.getSiteUrl(), _options); + } + + // hot path (called for every post url and image when rendering e.g. sitemaps), + // skip option merging and the site url lookup when there's nothing to replace + if (!url) { + // matches the util, which defaults a missing url to '' + return url === undefined ? '' : url; + } + + if (!url.includes(TRANSFORM_READY_TO_ABSOLUTE_DEFAULTS.replacementStr)) { + return url; + } + + return replaceTransformReadyPlaceholders(url, this.getSiteUrl(), this._transformReadyToAbsoluteDefaults); } transformReadyToRelative(url: string, options?: TransformReadyToRelativeOptionsInput): string { diff --git a/packages/url-utils/src/utils/transform-ready-to-absolute.ts b/packages/url-utils/src/utils/transform-ready-to-absolute.ts index 18ab8c486..ca865deb8 100644 --- a/packages/url-utils/src/utils/transform-ready-to-absolute.ts +++ b/packages/url-utils/src/utils/transform-ready-to-absolute.ts @@ -1,9 +1,5 @@ import type {TransformReadyReplacementOptions, BaseUrlOptions} from './types'; -function escapeRegExp(string: string): string { - return string.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); -} - export interface TransformReadyToAbsoluteOptions extends TransformReadyReplacementOptions, BaseUrlOptions { staticImageUrlPrefix: string; staticFilesUrlPrefix: string; @@ -12,45 +8,109 @@ export interface TransformReadyToAbsoluteOptions extends TransformReadyReplaceme export type TransformReadyToAbsoluteOptionsInput = Partial; -const transformReadyToAbsolute = function ( - str: string = '', +export const DEFAULT_OPTIONS: Readonly = Object.freeze({ + replacementStr: '__GHOST_URL__', + staticImageUrlPrefix: 'content/images', + staticFilesUrlPrefix: 'content/files', + staticMediaUrlPrefix: 'content/media', + imageBaseUrl: null, + filesBaseUrl: null, + mediaBaseUrl: null +}); + +// Root and CDN base URLs are a tiny, near-static set of strings, memoize their +// trailing-slash-stripped form so we don't allocate a new string per replacement. +// Bounded so arbitrary input can't grow the cache unchecked. +const MAX_STRIPPED_URL_ENTRIES = 100; +const strippedUrlCache = new Map(); + +function stripTrailingSlash(url: string): string { + let stripped = strippedUrlCache.get(url); + + if (stripped === undefined) { + stripped = url.replace(/\/$/, ''); + + if (strippedUrlCache.size >= MAX_STRIPPED_URL_ENTRIES) { + strippedUrlCache.clear(); + } + strippedUrlCache.set(url, stripped); + } + + return stripped; +} + +// true if `str` has `/${prefix}` at `pos`, without slicing or building strings. +// String() because a configured prefix can be null, which 5.3.0 matched as "/null" +function hasPrefixAt(str: string, pos: number, prefix: string): boolean { + return str[pos] === '/' && str.startsWith(String(prefix), pos + 1); +} + +// the CDN base url for the asset type that follows a placeholder ending at `pos`, +// or the root url if there's no CDN for it +function baseUrlFor(str: string, pos: number, root: string, options: TransformReadyToAbsoluteOptions): string { + if (options.mediaBaseUrl && hasPrefixAt(str, pos, options.staticMediaUrlPrefix)) { + return options.mediaBaseUrl; + } + + if (options.filesBaseUrl && hasPrefixAt(str, pos, options.staticFilesUrlPrefix)) { + return options.filesBaseUrl; + } + + if (options.imageBaseUrl && hasPrefixAt(str, pos, options.staticImageUrlPrefix)) { + return options.imageBaseUrl; + } + + return root; +} + +/** + * Replaces every `replacementStr` in `str` with the root URL, or the matching + * CDN base URL when the placeholder is followed by a static asset prefix. + * Expects fully-populated options, callers are responsible for merging defaults. + */ +export function replaceTransformReadyPlaceholders( + str: string, root: string, - _options: TransformReadyToAbsoluteOptionsInput = {} + options: TransformReadyToAbsoluteOptions ): string { - const defaultOptions: TransformReadyToAbsoluteOptions = { - replacementStr: '__GHOST_URL__', - staticImageUrlPrefix: 'content/images', - staticFilesUrlPrefix: 'content/files', - staticMediaUrlPrefix: 'content/media', - imageBaseUrl: null, - filesBaseUrl: null, - mediaBaseUrl: null - }; - const options = Object.assign({}, defaultOptions, _options); - - if (!str || str.indexOf(options.replacementStr) === -1) { - return str; + const {replacementStr} = options; + let result = ''; + + // an empty replacementStr matches at every position, like a global regex replace + if (replacementStr === '') { + for (let pos = 0; pos <= str.length; pos++) { + result += stripTrailingSlash(baseUrlFor(str, pos, root, options)) + str.slice(pos, pos + 1); + } + return result; } - const replacementRegex = new RegExp(escapeRegExp(options.replacementStr), 'g'); + let lastIndex = 0; + let pos = str.indexOf(replacementStr); - return str.replace(replacementRegex, (match: string, offset: number): string => { - const remainder = str.slice(offset + match.length); + while (pos !== -1) { + const after = pos + replacementStr.length; + result += str.slice(lastIndex, pos) + stripTrailingSlash(baseUrlFor(str, after, root, options)); + lastIndex = after; + pos = str.indexOf(replacementStr, lastIndex); + } - if (remainder.startsWith(`/${options.staticMediaUrlPrefix}`) && options.mediaBaseUrl) { - return options.mediaBaseUrl.replace(/\/$/, ''); - } + return result + str.slice(lastIndex); +} - if (remainder.startsWith(`/${options.staticFilesUrlPrefix}`) && options.filesBaseUrl) { - return options.filesBaseUrl.replace(/\/$/, ''); - } +const transformReadyToAbsolute = function ( + str: string = '', + root: string, + _options?: TransformReadyToAbsoluteOptionsInput +): string { + if (!str) { + return str; + } - if (remainder.startsWith(`/${options.staticImageUrlPrefix}`) && options.imageBaseUrl) { - return options.imageBaseUrl.replace(/\/$/, ''); - } + const options: TransformReadyToAbsoluteOptions = _options + ? Object.assign({}, DEFAULT_OPTIONS, _options) + : DEFAULT_OPTIONS; - return root.replace(/\/$/, ''); - }); + return replaceTransformReadyPlaceholders(str, root, options); }; export default transformReadyToAbsolute; diff --git a/packages/url-utils/test/unit/utils/transform-ready-to-absolute-equivalence.test.js b/packages/url-utils/test/unit/utils/transform-ready-to-absolute-equivalence.test.js new file mode 100644 index 000000000..540b788e1 --- /dev/null +++ b/packages/url-utils/test/unit/utils/transform-ready-to-absolute-equivalence.test.js @@ -0,0 +1,172 @@ +// Switch these lines once there are useful utils +// const testUtils = require('./utils'); +require('../../utils'); + +const UrlUtils = require('../../../lib/UrlUtils').default; +const transformReadyToAbsolute = require('../../../lib/utils/transform-ready-to-absolute').default; +const legacy = require('../../utils/legacy/transform-ready-to-absolute'); + +const inputs = [ + undefined, + null, + '', + 'no placeholder here', + 'https://example.com/content/images/a.jpg', + '__GHOST_URL__', + '__GHOST_URL__/', + '__GHOST_URL__/my-post/', + '__GHOST_URL__/content/images/a.jpg', + '__GHOST_URL__/content/images', + '__GHOST_URL__content/images/a.jpg', + '__GHOST_URL__/content/imagesX/a.jpg', + '__GHOST_URL__/content/image/a.jpg', + '__GHOST_URL__/content/files/doc.pdf', + '__GHOST_URL__/content/filesX/doc.pdf', + '__GHOST_URL__/content/media/video.mp4', + '__GHOST_URL__/content/mediaX/video.mp4', + '__GHOST_URL__/cdn/a.jpg', + '__GHOST_URL__/custom/images/a.jpg', + '', + '__GHOST_URL__/content/media/v.mp4 __GHOST_URL__/content/files/f.pdf __GHOST_URL__/content/images/i.png __GHOST_URL__/', + '__GHOST_URL____GHOST_URL__/content/images/z.png', + '__GHOST_URL___GHOST_URL__/content/images/z.png', + 'text __GHOST_URL__', + '__CUSTOM__/content/images/a.jpg and __GHOST_URL__/content/images/a.jpg', + 'emoji 😀__GHOST_URL__/😀' +]; + +const roots = [ + 'https://example.com', + 'https://example.com/', + 'https://example.com/subdir/', + 'https://example.com//' +]; + +const assetBaseUrlVariants = [ + {}, + {image: 'https://images.cdn.com/'}, + {image: 'https://images.cdn.com', files: 'https://files.cdn.com/', media: 'https://media.cdn.com/subdir/'}, + {files: 'https://files.cdn.com'}, + {media: 'https://media.cdn.com'} +]; + +const callOptionVariants = [ + undefined, + {}, + {replacementStr: undefined}, + {replacementStr: '__CUSTOM__'}, + {staticImageUrlPrefix: 'custom/images'}, + {imageBaseUrl: 'https://override.cdn.com/'}, + {imageBaseUrl: null}, + {filesBaseUrl: 'https://override-files.cdn.com', mediaBaseUrl: 'https://override-media.cdn.com/'}, + {staticImageUrlPrefix: 'content/images', staticFilesUrlPrefix: 'content/files', staticMediaUrlPrefix: 'content/media'}, + {staticImageUrlPrefix: ''} +]; + +function assetOptionsFor(assetBaseUrls) { + return { + imageBaseUrl: assetBaseUrls.image || null, + filesBaseUrl: assetBaseUrls.files || null, + mediaBaseUrl: assetBaseUrls.media || null + }; +} + +describe('utils: transformReadyToAbsolute() equivalence with 5.3.0', function () { + it('produces identical output for all inputs and options', function () { + let checked = 0; + + for (const root of roots) { + for (const assetBaseUrls of assetBaseUrlVariants) { + for (const callOptions of callOptionVariants) { + const options = callOptions === undefined && Object.keys(assetBaseUrls).length === 0 + ? undefined + : Object.assign({}, assetOptionsFor(assetBaseUrls), callOptions); + + for (const input of inputs) { + const expected = legacy.transformReadyToAbsolute(input, root, options); + const actual = transformReadyToAbsolute(input, root, options); + (actual === expected).should.equal(true, `input: ${input}, root: ${root}, options: ${JSON.stringify(options)}\nexpected: ${expected}\nactual: ${actual}`); + checked += 1; + } + } + } + } + + checked.should.be.above(0); + }); + + it('produces identical output for an empty replacementStr', function () { + for (const input of ['abc', '😀/content/images/x', '/content/images/a.jpg']) { + const options = {replacementStr: '', imageBaseUrl: 'https://cdn.com/'}; + transformReadyToAbsolute(input, 'https://example.com/', options) + .should.equal(legacy.transformReadyToAbsolute(input, 'https://example.com/', options)); + } + }); + + it('keeps working after the base url cache is cleared', function () { + for (let i = 0; i < 101; i++) { + transformReadyToAbsolute('__GHOST_URL__/a/', `https://example${i}.com/`) + .should.equal(`https://example${i}.com/a/`); + } + }); + + it('produces identical output when options are null', function () { + transformReadyToAbsolute('__GHOST_URL__/a/', 'https://example.com/', null) + .should.equal(legacy.transformReadyToAbsolute('__GHOST_URL__/a/', 'https://example.com/', null)); + }); +}); + +describe('UrlUtils: transformReadyToAbsolute() equivalence with 5.3.0', function () { + it('produces identical output for all inputs and options', function () { + for (const siteUrl of roots) { + for (const assetBaseUrls of assetBaseUrlVariants) { + for (const frozen of [false, true]) { + const urlUtils = new UrlUtils({getSiteUrl: () => siteUrl, assetBaseUrls, frozen}); + const config = {siteUrl, assetBaseUrls}; + + for (const callOptions of callOptionVariants) { + for (const input of inputs) { + const expected = legacy.urlUtilsTransformReadyToAbsolute(config, input, callOptions); + const actual = urlUtils.transformReadyToAbsolute(input, callOptions); + (actual === expected).should.equal(true, `input: ${input}, site: ${siteUrl}, assets: ${JSON.stringify(assetBaseUrls)}, options: ${JSON.stringify(callOptions)}\nexpected: ${expected}\nactual: ${actual}`); + } + } + } + } + } + }); + + it('produces identical output with custom static prefixes', function () { + const config = { + siteUrl: 'https://example.com/', + staticImageUrlPrefix: 'assets/img', + staticFilesUrlPrefix: 'assets/files', + staticMediaUrlPrefix: 'assets/media', + assetBaseUrls: {image: 'https://img.cdn/', files: 'https://files.cdn/', media: 'https://media.cdn/'} + }; + const urlUtils = new UrlUtils(Object.assign({getSiteUrl: () => config.siteUrl}, config)); + + for (const input of inputs) { + for (const callOptions of callOptionVariants) { + const expected = legacy.urlUtilsTransformReadyToAbsolute(config, input, callOptions); + (urlUtils.transformReadyToAbsolute(input, callOptions) === expected).should.equal(true); + } + } + + urlUtils.transformReadyToAbsolute('__GHOST_URL__/assets/img/a.png').should.equal('https://img.cdn/assets/img/a.png'); + }); + + it('does not look up the site url when there is nothing to replace', function () { + let calls = 0; + const urlUtils = new UrlUtils({getSiteUrl: () => { + calls += 1; + return 'https://example.com/'; + }}); + + urlUtils.transformReadyToAbsolute('https://example.com/a/').should.equal('https://example.com/a/'); + calls.should.equal(0); + + urlUtils.transformReadyToAbsolute('__GHOST_URL__/a/').should.equal('https://example.com/a/'); + calls.should.equal(1); + }); +}); diff --git a/packages/url-utils/test/utils/legacy/transform-ready-to-absolute.js b/packages/url-utils/test/utils/legacy/transform-ready-to-absolute.js new file mode 100644 index 000000000..a31700ea5 --- /dev/null +++ b/packages/url-utils/test/utils/legacy/transform-ready-to-absolute.js @@ -0,0 +1,73 @@ +// Verbatim copy of the @tryghost/url-utils@5.3.0 implementation, used to check +// the optimised implementation produces identical output. Not shipped. + +function escapeRegExp(string) { + return string.replace(/[.*+?^${}()|[\]\\]/g, '\\$&'); +} + +const transformReadyToAbsolute = function (str = '', root, _options = {}) { + const defaultOptions = { + replacementStr: '__GHOST_URL__', + staticImageUrlPrefix: 'content/images', + staticFilesUrlPrefix: 'content/files', + staticMediaUrlPrefix: 'content/media', + imageBaseUrl: null, + filesBaseUrl: null, + mediaBaseUrl: null + }; + const options = Object.assign({}, defaultOptions, _options); + + if (!str || str.indexOf(options.replacementStr) === -1) { + return str; + } + + const replacementRegex = new RegExp(escapeRegExp(options.replacementStr), 'g'); + + return str.replace(replacementRegex, (match, offset) => { + const remainder = str.slice(offset + match.length); + + if (remainder.startsWith(`/${options.staticMediaUrlPrefix}`) && options.mediaBaseUrl) { + return options.mediaBaseUrl.replace(/\/$/, ''); + } + + if (remainder.startsWith(`/${options.staticFilesUrlPrefix}`) && options.filesBaseUrl) { + return options.filesBaseUrl.replace(/\/$/, ''); + } + + if (remainder.startsWith(`/${options.staticImageUrlPrefix}`) && options.imageBaseUrl) { + return options.imageBaseUrl.replace(/\/$/, ''); + } + + return root.replace(/\/$/, ''); + }); +}; + +// similar to Object.assign but will not override defaults if a source value is undefined +function assignOptions(target, ...sources) { + const options = sources.map((x) => { + return Object.entries(x) + .filter(([, value]) => value !== undefined) + .reduce((obj, [key, value]) => (obj[key] = value, obj), {}); + }); + return Object.assign(target, ...options); +} + +// UrlUtils#transformReadyToAbsolute from 5.3.0, `config` mirrors the instance's +// static prefixes and asset base urls +function urlUtilsTransformReadyToAbsolute(config, url, options) { + const assetDefaults = { + staticImageUrlPrefix: config.staticImageUrlPrefix || 'content/images', + staticFilesUrlPrefix: config.staticFilesUrlPrefix || 'content/files', + staticMediaUrlPrefix: config.staticMediaUrlPrefix || 'content/media', + imageBaseUrl: (config.assetBaseUrls && config.assetBaseUrls.image) || null, + filesBaseUrl: (config.assetBaseUrls && config.assetBaseUrls.files) || null, + mediaBaseUrl: (config.assetBaseUrls && config.assetBaseUrls.media) || null + }; + const _options = assignOptions({}, assetDefaults, {}, options || {}); + return transformReadyToAbsolute(url, config.siteUrl, _options); +} + +module.exports = { + transformReadyToAbsolute, + urlUtilsTransformReadyToAbsolute +}; From f9d6673c3af78679aa41c647f92c970c88c0e6ad Mon Sep 17 00:00:00 2001 From: Austin Burdine Date: Fri, 2 Oct 2026 11:29:33 -0400 Subject: [PATCH 3/6] Computed permalink dates lazily without moment `replacePermalink` created a moment-timezone instance for every call, even for the default `/:slug/` permalink which has no date tokens (~160ms per 300k posts in sitemap builds). - date parts are now only computed when `:year`, `:month` or `:day` appear - they're formatted with a cached `Intl.DateTimeFormat('en-CA')` per timezone for Dates, timestamps and ISO 8601 strings with an explicit offset - other inputs (strings without an offset, which moment parses in the site timezone, invalid dates, years outside 1900-9999 and timezones Intl doesn't support) still go through moment-timezone so their output is unchanged. The dependency stays for that fallback Verified identical to the 5.3.0 output for every Intl timezone, for dates in every year from 1900 to 2100 including date-line and DST edges. Co-Authored-By: Claude Opus 5.5 --- .../url-utils/src/utils/replace-permalink.ts | 169 ++++++++++++++---- .../replace-permalink-equivalence.test.js | 151 ++++++++++++++++ .../test/utils/legacy/replace-permalink.js | 50 ++++++ 3 files changed, 338 insertions(+), 32 deletions(-) create mode 100644 packages/url-utils/test/unit/utils/replace-permalink-equivalence.test.js create mode 100644 packages/url-utils/test/utils/legacy/replace-permalink.js diff --git a/packages/url-utils/src/utils/replace-permalink.ts b/packages/url-utils/src/utils/replace-permalink.ts index b8fbe6389..8b5e21f86 100644 --- a/packages/url-utils/src/utils/replace-permalink.ts +++ b/packages/url-utils/src/utils/replace-permalink.ts @@ -12,49 +12,154 @@ interface PermalinkResource { id: string; } +interface DateParts { + year: string; + month: string; + day: string; +} + +// ISO 8601 date-times with an explicit offset describe an exact instant, so they +// parse identically with `Date.parse` and moment. Anything else (no offset, +// other formats) is parsed by moment in the site timezone and goes the slow path. +const ISO_WITH_OFFSET_REGEX = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(:\d{2}(\.\d{1,3})?)?(Z|[+-]\d{2}:\d{2})$/; + +// Intl uses the Julian calendar before 1582 and moment pads/formats years outside +// 4 digits differently, only use Intl for timestamps comfortably inside 1900-9999 +const MIN_FAST_TIMESTAMP = Date.UTC(1900, 0, 2); +const MAX_FAST_TIMESTAMP = Date.UTC(9999, 11, 30); + +// Timezones are a tiny set (usually one per site), bounded so arbitrary input +// can't grow the cache unchecked. `null` marks a timezone Intl doesn't support. +const MAX_FORMATTER_ENTRIES = 100; +const formatterCache = new Map(); + +function getFormatter(timezone: string): Intl.DateTimeFormat | null { + let formatter = formatterCache.get(timezone); + + if (formatter === undefined) { + try { + // en-CA formats as YYYY-MM-DD + formatter = new Intl.DateTimeFormat('en-CA', { + timeZone: timezone, + year: 'numeric', + month: '2-digit', + day: '2-digit' + }); + } catch { + formatter = null; + } + + if (formatterCache.size >= MAX_FORMATTER_ENTRIES) { + formatterCache.clear(); + } + formatterCache.set(timezone, formatter); + } + + return formatter; +} + +function getTimestamp(date: unknown): number | null { + if (date instanceof Date) { + return date.getTime(); + } + + if (typeof date === 'number') { + return date; + } + + if (typeof date === 'string' && ISO_WITH_OFFSET_REGEX.test(date)) { + return Date.parse(date); + } + + return null; +} + +function getDatePartsFast(date: unknown, timezone: string): DateParts | null { + if (typeof timezone !== 'string') { + return null; + } + + const timestamp = getTimestamp(date); + + if (timestamp === null || !(timestamp >= MIN_FAST_TIMESTAMP && timestamp <= MAX_FAST_TIMESTAMP)) { + return null; + } + + const formatter = getFormatter(timezone); + + if (!formatter) { + return null; + } + + const formatted = formatter.format(timestamp); + + // guard against ICU/CLDR changes to the en-CA date format + if (formatted.length !== 10 || formatted.charCodeAt(4) !== 45 || formatted.charCodeAt(7) !== 45) { + return null; + } + + return { + year: formatted.slice(0, 4), + month: formatted.slice(5, 7), + day: formatted.slice(8, 10) + }; +} + +function getDateParts(date: unknown, timezone: string): DateParts { + const fastParts = getDatePartsFast(date, timezone); + + if (fastParts) { + return fastParts; + } + + // fall back to moment-timezone for inputs Intl can't handle identically + const publishedAtMoment = moment.tz(date as moment.MomentInput, timezone); + + return { + year: publishedAtMoment.format('YYYY'), + month: publishedAtMoment.format('MM'), + day: publishedAtMoment.format('DD') + }; +} + /** * creates the url path for a post based on blog timezone and permalink pattern */ function replacePermalink(permalink: string, resource: PermalinkResource, timezone: string = 'UTC'): string { const primaryTagFallback = 'all'; - const publishedAtMoment = moment.tz(resource.published_at || Date.now(), timezone); - const permalinkLookUp: Record string> = { - year: function () { - return publishedAtMoment.format('YYYY'); - }, - month: function () { - return publishedAtMoment.format('MM'); - }, - day: function () { - return publishedAtMoment.format('DD'); - }, - author: function () { - return resource.primary_author?.slug ?? 'undefined'; - }, - primary_author: function () { - return resource.primary_author ? resource.primary_author.slug : primaryTagFallback; - }, - primary_tag: function () { - return resource.primary_tag ? resource.primary_tag.slug : primaryTagFallback; - }, - slug: function () { - return resource.slug; - }, - id: function () { - return resource.id; + let dateParts: DateParts | undefined; + + // date parts are only computed if the permalink contains a date token + const getPublishedDateParts = function (): DateParts { + if (!dateParts) { + dateParts = getDateParts(resource.published_at || Date.now(), timezone); } + return dateParts; }; // replace tags like :slug or :year with actual values - const permalinkKeys = Object.keys(permalinkLookUp); return permalink.replace(/(:[a-z_]+)/g, function (match: string): string { - const key = match.slice(1); - if (permalinkKeys.includes(key)) { - // Known route segment - use the lookup function - return permalinkLookUp[key](); + switch (match) { + case ':year': + return getPublishedDateParts().year; + case ':month': + return getPublishedDateParts().month; + case ':day': + return getPublishedDateParts().day; + case ':author': + return resource.primary_author?.slug ?? 'undefined'; + case ':primary_author': + return resource.primary_author ? resource.primary_author.slug : primaryTagFallback; + case ':primary_tag': + return resource.primary_tag ? resource.primary_tag.slug : primaryTagFallback; + case ':slug': + return resource.slug; + case ':id': + return resource.id; + default: + // Unknown route segment - return 'undefined' string + return 'undefined'; } - // Unknown route segment - return 'undefined' string - return 'undefined'; }); } diff --git a/packages/url-utils/test/unit/utils/replace-permalink-equivalence.test.js b/packages/url-utils/test/unit/utils/replace-permalink-equivalence.test.js new file mode 100644 index 000000000..edab6cbc1 --- /dev/null +++ b/packages/url-utils/test/unit/utils/replace-permalink-equivalence.test.js @@ -0,0 +1,151 @@ +// Switch these lines once there are useful utils +// const testUtils = require('./utils'); +require('../../utils'); + +const sinon = require('sinon'); +const replacePermalink = require('../../../lib/utils/replace-permalink').default; +const legacy = require('../../utils/legacy/replace-permalink'); + +const permalinks = [ + '/:slug/', + '/:year/:month/:day/:slug/', + '/:year/:id/', + '/:year/:month/:slug/', + '/:primary_tag/:slug/', + '/:primary_author/:slug/', + '/:author/:slug/', + '/:unknown/:slug/', + '/:constructor/:toString/:slug/', + '/blog/:slug:id/', + '/no-tokens/', + '/:Year/:slug/', + '' +]; + +const timezones = [ + undefined, + 'UTC', + 'Europe/Berlin', + 'America/Los_Angeles', + 'Pacific/Kiritimati', + 'Pacific/Pago_Pago', + 'Pacific/Apia', + 'Asia/Kolkata', + 'Australia/Lord_Howe', + 'Etc/GMT+12', + 'US/Pacific', + 'europe/berlin' +]; + +const publishedAts = [ + // date-line edges, Kiritimati is UTC+14 and Pago Pago UTC-11 + new Date('2016-05-18T06:30:00.000Z'), + new Date('2016-05-17T09:59:59.999Z'), + new Date('2016-05-17T10:00:00.000Z'), + new Date('2016-05-17T11:00:00.000Z'), + new Date('2016-05-17T12:00:00.000Z'), + new Date('2016-05-17T23:59:59.999Z'), + new Date('2016-12-31T23:30:00.000Z'), + new Date('2017-01-01T00:30:00.000Z'), + // DST transitions + new Date('2021-03-28T00:30:00.000Z'), + new Date('2021-03-28T01:30:00.000Z'), + new Date('2021-11-07T07:30:00.000Z'), + new Date('2021-11-07T08:30:00.000Z'), + // Apia skipped 30th Dec 2011, Kiritimati skipped 31st Dec 1994 + new Date('2011-12-29T10:30:00.000Z'), + new Date('2011-12-30T10:30:00.000Z'), + new Date('1994-12-30T10:30:00.000Z'), + new Date('1994-12-31T10:30:00.000Z'), + // outside the Intl fast path range + new Date('1899-12-31T23:30:00.000Z'), + new Date('1066-10-14T12:00:00.000Z'), + new Date('9999-12-31T12:00:00.000Z'), + new Date('invalid'), + 1463553000000, + 1463553000000.7, + NaN, + 0, + '2016-05-17T23:30:00.000Z', + '2016-05-17T23:30:00Z', + '2016-05-17T23:30Z', + '2016-05-17T23:30:00.000+02:00', + '2016-05-17T23:30:00.000-11:00', + '2016-05-17T23:30:00.123456Z', + '2016-05-17T23:30:00', + '2016-05-17 23:30:00', + '2016-05-17', + 'May 17, 2016 23:30', + 'not a date', + null, + undefined +]; + +function resourceFor(publishedAt) { + return { + id: '5ca5b2b8a7f5e6001ec4e1f0', + slug: 'short-and-sweet', + published_at: publishedAt, + primary_tag: {slug: 'news'}, + primary_author: {slug: 'joe'} + }; +} + +describe('utils: replacePermalink() equivalence with 5.3.0', function () { + let clock; + + beforeEach(function () { + // moment logs a deprecation warning for non-ISO date strings + sinon.stub(console, 'warn'); + clock = sinon.useFakeTimers(new Date('2016-05-17T23:30:00.000Z')); + }); + + afterEach(function () { + clock.restore(); + sinon.restore(); + }); + + it('produces identical output for all permalinks, dates and timezones', function () { + for (const timezone of timezones) { + for (const publishedAt of publishedAts) { + const resource = resourceFor(publishedAt); + + for (const permalink of permalinks) { + const expected = legacy.replacePermalink(permalink, resource, timezone); + const actual = replacePermalink(permalink, resource, timezone); + actual.should.equal(expected, `permalink: ${permalink}, published_at: ${publishedAt}, timezone: ${timezone}`); + } + } + } + }); + + it('produces identical output for missing primary tag and author', function () { + const resource = {id: '1', slug: 'slug', published_at: new Date('2016-05-17T23:30:00.000Z')}; + + for (const permalink of permalinks) { + replacePermalink(permalink, resource, 'Europe/Berlin') + .should.equal(legacy.replacePermalink(permalink, resource, 'Europe/Berlin')); + } + }); + + it('produces identical output for timezones Intl does not support', function () { + // moment-timezone logs an error for unknown timezones + sinon.stub(console, 'error'); + const resource = resourceFor(new Date('2016-05-17T23:30:00.000Z')); + + for (const timezone of ['Not/A_Zone', null]) { + replacePermalink('/:year/:month/:day/:slug/', resource, timezone) + .should.equal(legacy.replacePermalink('/:year/:month/:day/:slug/', resource, timezone)); + } + }); + + it('keeps working after the formatter cache is cleared', function () { + const resource = resourceFor(new Date('2016-05-17T23:30:00.000Z')); + + for (let i = 0; i < 101; i++) { + const timezone = `Etc/GMT${i % 2 ? '+' : '-'}${i % 12}`; + replacePermalink('/:year/:month/:day/', resource, timezone) + .should.equal(legacy.replacePermalink('/:year/:month/:day/', resource, timezone)); + } + }); +}); diff --git a/packages/url-utils/test/utils/legacy/replace-permalink.js b/packages/url-utils/test/utils/legacy/replace-permalink.js new file mode 100644 index 000000000..d2364965f --- /dev/null +++ b/packages/url-utils/test/utils/legacy/replace-permalink.js @@ -0,0 +1,50 @@ +// Verbatim copy of the @tryghost/url-utils@5.3.0 implementation, used to check +// the optimised implementation produces identical output. Not shipped. +const moment = require('moment-timezone'); + +function replacePermalink(permalink, resource, timezone = 'UTC') { + const primaryTagFallback = 'all'; + const publishedAtMoment = moment.tz(resource.published_at || Date.now(), timezone); + const permalinkLookUp = { + year: function () { + return publishedAtMoment.format('YYYY'); + }, + month: function () { + return publishedAtMoment.format('MM'); + }, + day: function () { + return publishedAtMoment.format('DD'); + }, + author: function () { + return resource.primary_author?.slug ?? 'undefined'; + }, + primary_author: function () { + return resource.primary_author ? resource.primary_author.slug : primaryTagFallback; + }, + primary_tag: function () { + return resource.primary_tag ? resource.primary_tag.slug : primaryTagFallback; + }, + slug: function () { + return resource.slug; + }, + id: function () { + return resource.id; + } + }; + + // replace tags like :slug or :year with actual values + const permalinkKeys = Object.keys(permalinkLookUp); + return permalink.replace(/(:[a-z_]+)/g, function (match) { + const key = match.slice(1); + if (permalinkKeys.includes(key)) { + // Known route segment - use the lookup function + return permalinkLookUp[key](); + } + // Unknown route segment - return 'undefined' string + return 'undefined'; + }); +} + +module.exports = { + replacePermalink +}; From d145bfa60848e23b63e19bb22cc5a4c6fda67a25 Mon Sep 17 00:00:00 2001 From: Austin Burdine Date: Fri, 2 Oct 2026 11:54:49 -0400 Subject: [PATCH 4/6] Parsed permalink patterns once in replacePermalink `replacePermalink` ran its token regex and allocated a replace callback on every call. Sites only have a handful of permalink patterns, so each pattern is now split into literal and token parts once (bounded cache) and calls just concatenate the parts. 300k posts: `/:slug/` 32ms -> 4ms, `/:primary_tag/:slug/` 50ms -> 7ms, `/:year/:month/:day/:slug/` 185ms -> 109ms. Output is unchanged. Co-Authored-By: Claude Opus 5.5 --- .../url-utils/src/utils/replace-permalink.ts | 126 ++++++++++++++---- .../replace-permalink-equivalence.test.js | 8 ++ 2 files changed, 106 insertions(+), 28 deletions(-) diff --git a/packages/url-utils/src/utils/replace-permalink.ts b/packages/url-utils/src/utils/replace-permalink.ts index 8b5e21f86..6fcc20ae9 100644 --- a/packages/url-utils/src/utils/replace-permalink.ts +++ b/packages/url-utils/src/utils/replace-permalink.ts @@ -122,45 +122,115 @@ function getDateParts(date: unknown, timezone: string): DateParts { }; } +const TOKEN_YEAR = 0; +const TOKEN_MONTH = 1; +const TOKEN_DAY = 2; +const TOKEN_AUTHOR = 3; +const TOKEN_PRIMARY_AUTHOR = 4; +const TOKEN_PRIMARY_TAG = 5; +const TOKEN_SLUG = 6; +const TOKEN_ID = 7; + +const TOKENS = new Map([ + [':year', TOKEN_YEAR], + [':month', TOKEN_MONTH], + [':day', TOKEN_DAY], + [':author', TOKEN_AUTHOR], + [':primary_author', TOKEN_PRIMARY_AUTHOR], + [':primary_tag', TOKEN_PRIMARY_TAG], + [':slug', TOKEN_SLUG], + [':id', TOKEN_ID] +]); + +// literal strings are output as-is, numbers are TOKEN_* values +type PermalinkPart = string | number; + +// Permalink patterns are a tiny set (one per collection/route), so parse each +// once rather than running the token regex on every call. Bounded so arbitrary +// input can't grow the cache unchecked. +const MAX_PERMALINK_ENTRIES = 100; +const permalinkCache = new Map(); + +function compilePermalink(permalink: string): PermalinkPart[] { + let parts = permalinkCache.get(permalink); + + if (parts) { + return parts; + } + + parts = []; + let lastIndex = 0; + + for (const match of permalink.matchAll(/:[a-z_]+/g)) { + if (match.index > lastIndex) { + parts.push(permalink.slice(lastIndex, match.index)); + } + // unknown route segments are output as the string 'undefined' + parts.push(TOKENS.get(match[0]) ?? 'undefined'); + lastIndex = match.index + match[0].length; + } + + if (lastIndex < permalink.length) { + parts.push(permalink.slice(lastIndex)); + } + + if (permalinkCache.size >= MAX_PERMALINK_ENTRIES) { + permalinkCache.clear(); + } + permalinkCache.set(permalink, parts); + + return parts; +} + /** * creates the url path for a post based on blog timezone and permalink pattern */ function replacePermalink(permalink: string, resource: PermalinkResource, timezone: string = 'UTC'): string { const primaryTagFallback = 'all'; + const parts = compilePermalink(permalink); + // date parts are only computed if the permalink contains a date token let dateParts: DateParts | undefined; + let result = ''; - // date parts are only computed if the permalink contains a date token - const getPublishedDateParts = function (): DateParts { - if (!dateParts) { + for (const part of parts) { + if (typeof part === 'string') { + result += part; + continue; + } + + if (part <= TOKEN_DAY && !dateParts) { dateParts = getDateParts(resource.published_at || Date.now(), timezone); } - return dateParts; - }; - // replace tags like :slug or :year with actual values - return permalink.replace(/(:[a-z_]+)/g, function (match: string): string { - switch (match) { - case ':year': - return getPublishedDateParts().year; - case ':month': - return getPublishedDateParts().month; - case ':day': - return getPublishedDateParts().day; - case ':author': - return resource.primary_author?.slug ?? 'undefined'; - case ':primary_author': - return resource.primary_author ? resource.primary_author.slug : primaryTagFallback; - case ':primary_tag': - return resource.primary_tag ? resource.primary_tag.slug : primaryTagFallback; - case ':slug': - return resource.slug; - case ':id': - return resource.id; - default: - // Unknown route segment - return 'undefined' string - return 'undefined'; + switch (part) { + case TOKEN_YEAR: + result += dateParts!.year; + break; + case TOKEN_MONTH: + result += dateParts!.month; + break; + case TOKEN_DAY: + result += dateParts!.day; + break; + case TOKEN_AUTHOR: + result += resource.primary_author?.slug ?? 'undefined'; + break; + case TOKEN_PRIMARY_AUTHOR: + result += resource.primary_author ? resource.primary_author.slug : primaryTagFallback; + break; + case TOKEN_PRIMARY_TAG: + result += resource.primary_tag ? resource.primary_tag.slug : primaryTagFallback; + break; + case TOKEN_SLUG: + result += resource.slug; + break; + case TOKEN_ID: + result += resource.id; + break; } - }); + } + + return result; } export default replacePermalink; diff --git a/packages/url-utils/test/unit/utils/replace-permalink-equivalence.test.js b/packages/url-utils/test/unit/utils/replace-permalink-equivalence.test.js index edab6cbc1..a9a751bca 100644 --- a/packages/url-utils/test/unit/utils/replace-permalink-equivalence.test.js +++ b/packages/url-utils/test/unit/utils/replace-permalink-equivalence.test.js @@ -139,6 +139,14 @@ describe('utils: replacePermalink() equivalence with 5.3.0', function () { } }); + it('keeps working after the permalink cache is cleared', function () { + const resource = resourceFor(new Date('2016-05-17T23:30:00.000Z')); + + for (let i = 0; i < 101; i++) { + replacePermalink(`/${i}/:year/:slug/`, resource).should.equal(`/${i}/2016/short-and-sweet/`); + } + }); + it('keeps working after the formatter cache is cleared', function () { const resource = resourceFor(new Date('2016-05-17T23:30:00.000Z')); From 979bc107d5bdc20331b0a6784ad31dd9018df5f4 Mon Sep 17 00:00:00 2001 From: Austin Burdine Date: Fri, 2 Oct 2026 12:02:27 -0400 Subject: [PATCH 5/6] Added shared lru-cache memoize helper for url-utils caches The root URL, subdirectory regex and trailing-slash caches each had their own Map with clear-when-full eviction. They now share a small `memoize` helper backed by lru-cache, so frequently used entries are never evicted by a burst of unique inputs. Errors thrown by the memoized function are not cached. `parseRootUrl.clearCache()` is now `parseRootUrl.clear()`. Co-Authored-By: Claude Opus 5.5 --- packages/url-utils/package.json | 1 + .../src/utils/deduplicate-subdirectory.ts | 24 ++++----- packages/url-utils/src/utils/memoize.ts | 36 +++++++++++++ .../url-utils/src/utils/parse-root-url.ts | 36 ++++--------- .../src/utils/transform-ready-to-absolute.ts | 24 ++------- .../url-utils/test/unit/utils/memoize.test.js | 54 +++++++++++++++++++ .../test/unit/utils/parse-root-url.test.js | 4 +- 7 files changed, 116 insertions(+), 63 deletions(-) create mode 100644 packages/url-utils/src/utils/memoize.ts create mode 100644 packages/url-utils/test/unit/utils/memoize.test.js diff --git a/packages/url-utils/package.json b/packages/url-utils/package.json index 04b59b0c0..b6985d7b8 100644 --- a/packages/url-utils/package.json +++ b/packages/url-utils/package.json @@ -39,6 +39,7 @@ "dependencies": { "cheerio": "1.2.0", "lodash": "4.18.1", + "lru-cache": "^11.0.0", "moment": "2.31.0", "moment-timezone": "0.6.4", "remark": "11.0.2", diff --git a/packages/url-utils/src/utils/deduplicate-subdirectory.ts b/packages/url-utils/src/utils/deduplicate-subdirectory.ts index 0b55d716d..425f04205 100644 --- a/packages/url-utils/src/utils/deduplicate-subdirectory.ts +++ b/packages/url-utils/src/utils/deduplicate-subdirectory.ts @@ -1,6 +1,12 @@ +import memoize from './memoize'; import parseRootUrl from './parse-root-url'; -const subdirRegexCache = new Map(); +const buildSubdirRegex = memoize(function buildSubdirRegex(pathname: string): {subdir: string; regex: RegExp} { + const subdir = pathname.replace(/(^\/|\/$)+/g, ''); + // we can have subdirs that match TLDs so we need to restrict matches to + // duplicates that start with a / or the beginning of the url + return {subdir, regex: new RegExp(`(^|/)${subdir}/${subdir}(/|$)`)}; +}); /** * Remove duplicated directories from the start of a path or url's path @@ -22,21 +28,9 @@ const deduplicateSubdirectory = function deduplicateSubdirectory(url: string, ro return url; } - let cached = subdirRegexCache.get(pathname); + const {subdir, regex} = buildSubdirRegex(pathname); - if (!cached) { - const subdir = pathname.replace(/(^\/|\/$)+/g, ''); - // we can have subdirs that match TLDs so we need to restrict matches to - // duplicates that start with a / or the beginning of the url - cached = {subdir, regex: new RegExp(`(^|/)${subdir}/${subdir}(/|$)`)}; - - if (subdirRegexCache.size >= 100) { - subdirRegexCache.clear(); - } - subdirRegexCache.set(pathname, cached); - } - - return url.replace(cached.regex, `$1${cached.subdir}/`); + return url.replace(regex, `$1${subdir}/`); }; export default deduplicateSubdirectory; diff --git a/packages/url-utils/src/utils/memoize.ts b/packages/url-utils/src/utils/memoize.ts new file mode 100644 index 000000000..b749a490d --- /dev/null +++ b/packages/url-utils/src/utils/memoize.ts @@ -0,0 +1,36 @@ +import {LRUCache} from 'lru-cache'; + +export interface Memoized { + (key: string): V; + clear(): void; +} + +/** + * Memoize a pure single-argument function keyed on its string input, keeping + * the `max` most recently used results so arbitrary input can't grow the cache + * unchecked. Errors thrown by `fn` are not cached. + * + * @param {Function} fn function to memoize, must be a pure function of its input + * @param {number} [max=100] maximum number of cached results + * @returns {Function} memoized function with a `clear()` method + */ +export default function memoize(fn: (key: string) => V, max: number = 100): Memoized { + const cache = new LRUCache({max}); + + const memoized = function (key: string): V { + let value = cache.get(key); + + if (value === undefined) { + value = fn(key); + cache.set(key, value); + } + + return value; + }; + + memoized.clear = function clear(): void { + cache.clear(); + }; + + return memoized; +} diff --git a/packages/url-utils/src/utils/parse-root-url.ts b/packages/url-utils/src/utils/parse-root-url.ts index 8bf852d8d..1617f0ebb 100644 --- a/packages/url-utils/src/utils/parse-root-url.ts +++ b/packages/url-utils/src/utils/parse-root-url.ts @@ -1,4 +1,5 @@ import {URL} from 'url'; +import memoize from './memoize'; export interface ParsedRootUrl { readonly href: string; @@ -9,29 +10,21 @@ export interface ParsedRootUrl { readonly pathname: string; } -// Root URLs (site url, admin url, CDN base urls) are a tiny, near-static set of -// strings that get re-parsed on almost every url-utils call. Parsing is a pure -// function of the input string so results can be cached without any risk of -// going stale. Bounded so arbitrary input can't grow the cache unchecked. -const MAX_ENTRIES = 100; -const cache = new Map(); - /** * Parse a root URL, returning a cached immutable snapshot of the parts url-utils uses. * Throws the same errors as `new URL()` for invalid input (errors are not cached). * + * Root URLs (site url, admin url, CDN base urls) are a tiny, near-static set of + * strings that get re-parsed on almost every url-utils call. Parsing is a pure + * function of the input string so results can't go stale. + * * @param {string} rootUrl * @returns {ParsedRootUrl} */ -function parseRootUrl(rootUrl: string): ParsedRootUrl { - let parsed = cache.get(rootUrl); - - if (parsed) { - return parsed; - } - +const parseRootUrl = memoize(function parseRootUrl(rootUrl: string): ParsedRootUrl { const url = new URL(rootUrl); - parsed = Object.freeze({ + + return Object.freeze({ href: url.href, origin: url.origin, protocol: url.protocol, @@ -39,17 +32,6 @@ function parseRootUrl(rootUrl: string): ParsedRootUrl { hostname: url.hostname, pathname: url.pathname }); - - if (cache.size >= MAX_ENTRIES) { - cache.clear(); - } - cache.set(rootUrl, parsed); - - return parsed; -} - -parseRootUrl.clearCache = function clearCache(): void { - cache.clear(); -}; +}); export default parseRootUrl; diff --git a/packages/url-utils/src/utils/transform-ready-to-absolute.ts b/packages/url-utils/src/utils/transform-ready-to-absolute.ts index ca865deb8..71f2a4ce8 100644 --- a/packages/url-utils/src/utils/transform-ready-to-absolute.ts +++ b/packages/url-utils/src/utils/transform-ready-to-absolute.ts @@ -1,4 +1,5 @@ import type {TransformReadyReplacementOptions, BaseUrlOptions} from './types'; +import memoize from './memoize'; export interface TransformReadyToAbsoluteOptions extends TransformReadyReplacementOptions, BaseUrlOptions { staticImageUrlPrefix: string; @@ -19,25 +20,10 @@ export const DEFAULT_OPTIONS: Readonly = Object }); // Root and CDN base URLs are a tiny, near-static set of strings, memoize their -// trailing-slash-stripped form so we don't allocate a new string per replacement. -// Bounded so arbitrary input can't grow the cache unchecked. -const MAX_STRIPPED_URL_ENTRIES = 100; -const strippedUrlCache = new Map(); - -function stripTrailingSlash(url: string): string { - let stripped = strippedUrlCache.get(url); - - if (stripped === undefined) { - stripped = url.replace(/\/$/, ''); - - if (strippedUrlCache.size >= MAX_STRIPPED_URL_ENTRIES) { - strippedUrlCache.clear(); - } - strippedUrlCache.set(url, stripped); - } - - return stripped; -} +// trailing-slash-stripped form so we don't allocate a new string per replacement +const stripTrailingSlash = memoize(function stripTrailingSlash(url: string): string { + return url.replace(/\/$/, ''); +}); // true if `str` has `/${prefix}` at `pos`, without slicing or building strings. // String() because a configured prefix can be null, which 5.3.0 matched as "/null" diff --git a/packages/url-utils/test/unit/utils/memoize.test.js b/packages/url-utils/test/unit/utils/memoize.test.js new file mode 100644 index 000000000..7fb160fa1 --- /dev/null +++ b/packages/url-utils/test/unit/utils/memoize.test.js @@ -0,0 +1,54 @@ +// Switch these lines once there are useful utils +// const testUtils = require('./utils'); +require('../../utils'); + +const sinon = require('sinon'); +const memoize = require('../../../lib/utils/memoize').default; + +describe('utils: memoize()', function () { + it('returns cached results for repeated keys', function () { + const fn = sinon.spy(key => ({key})); + const memoized = memoize(fn); + + const first = memoized('a'); + memoized('a').should.equal(first); + memoized('b').should.not.equal(first); + fn.callCount.should.equal(2); + }); + + it('does not cache thrown errors', function () { + const fn = sinon.stub().throws(new TypeError('nope')); + const memoized = memoize(fn); + + (() => memoized('a')).should.throw(TypeError); + (() => memoized('a')).should.throw(TypeError); + fn.callCount.should.equal(2); + }); + + it('evicts the least recently used entry once full', function () { + const fn = sinon.spy(key => `${key}!`); + const memoized = memoize(fn, 2); + + memoized('a'); + memoized('b'); + memoized('a'); + memoized('c'); + fn.callCount.should.equal(3); + + memoized('a'); + fn.callCount.should.equal(3); + + memoized('b'); + fn.callCount.should.equal(4); + }); + + it('can be cleared', function () { + const fn = sinon.spy(key => `${key}!`); + const memoized = memoize(fn); + + memoized('a'); + memoized.clear(); + memoized('a'); + fn.callCount.should.equal(2); + }); +}); diff --git a/packages/url-utils/test/unit/utils/parse-root-url.test.js b/packages/url-utils/test/unit/utils/parse-root-url.test.js index fd0d0a82c..49a4a7079 100644 --- a/packages/url-utils/test/unit/utils/parse-root-url.test.js +++ b/packages/url-utils/test/unit/utils/parse-root-url.test.js @@ -6,7 +6,7 @@ const parseRootUrl = require('../../../lib/utils/parse-root-url').default; describe('utils: parseRootUrl()', function () { beforeEach(function () { - parseRootUrl.clearCache(); + parseRootUrl.clear(); }); it('returns the parsed parts of a url', function () { @@ -38,7 +38,7 @@ describe('utils: parseRootUrl()', function () { (() => parseRootUrl('not a url')).should.throw(TypeError); }); - it('clears the cache once it reaches its size limit', function () { + it('evicts the least recently used entry once full', function () { const first = parseRootUrl('https://example.com/'); for (let i = 0; i < 100; i++) { From be8abbe8c0216a901170fbfed53244cb07a9f0c8 Mon Sep 17 00:00:00 2001 From: Austin Burdine Date: Fri, 2 Oct 2026 12:02:27 -0400 Subject: [PATCH 6/6] Simplified replacePermalink date parts - only Dates and timestamps use the cached `Intl.DateTimeFormat`, all strings go through moment. Drops the ISO-with-offset regex; Ghost passes `published_at` as a Date - the formatter and compiled permalink caches use the shared memoize helper, unsupported timezones throw and fall back to moment rather than being cached as `null` - dropped the en-CA output guard, the equivalence tests catch any change to the format - the Intl path is only used for timezones moment knows (Intl accepts some, e.g. `+01:00`, that moment treats as UTC) and when the active moment locale doesn't rewrite digits (e.g. `ar`), so output is unchanged for both Output is unchanged, still identical to 5.3.0 for every Intl timezone and every year from 1900 to 2100, and for every moment locale. Co-Authored-By: Claude Opus 5.5 --- .../url-utils/src/utils/replace-permalink.ts | 232 +++++++----------- .../replace-permalink-equivalence.test.js | 33 ++- 2 files changed, 116 insertions(+), 149 deletions(-) diff --git a/packages/url-utils/src/utils/replace-permalink.ts b/packages/url-utils/src/utils/replace-permalink.ts index 6fcc20ae9..f44078a9c 100644 --- a/packages/url-utils/src/utils/replace-permalink.ts +++ b/packages/url-utils/src/utils/replace-permalink.ts @@ -1,4 +1,5 @@ import * as moment from 'moment-timezone'; +import memoize from './memoize'; interface PermalinkResource { published_at?: string | number | Date | null; @@ -18,147 +19,92 @@ interface DateParts { day: string; } -// ISO 8601 date-times with an explicit offset describe an exact instant, so they -// parse identically with `Date.parse` and moment. Anything else (no offset, -// other formats) is parsed by moment in the site timezone and goes the slow path. -const ISO_WITH_OFFSET_REGEX = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(:\d{2}(\.\d{1,3})?)?(Z|[+-]\d{2}:\d{2})$/; - // Intl uses the Julian calendar before 1582 and moment pads/formats years outside // 4 digits differently, only use Intl for timestamps comfortably inside 1900-9999 -const MIN_FAST_TIMESTAMP = Date.UTC(1900, 0, 2); -const MAX_FAST_TIMESTAMP = Date.UTC(9999, 11, 30); +const MIN_TIMESTAMP = Date.UTC(1900, 0, 2); +const MAX_TIMESTAMP = Date.UTC(9999, 11, 30); + +// moment's default postformat, which leaves formatted digits as-is. Locales like +// 'ar' override it to output non-Latin digits +const DEFAULT_POSTFORMAT = moment.localeData('en').postformat; + +// en-CA formats as YYYY-MM-DD. Throws for timezones moment doesn't know (it falls +// back to UTC for those, while Intl accepts some, e.g. '+01:00') or Intl doesn't support +const getFormatter = memoize(function getFormatter(timezone: string): Intl.DateTimeFormat { + if (!moment.tz.zone(timezone)) { + throw new RangeError(`Unknown timezone: ${timezone}`); + } -// Timezones are a tiny set (usually one per site), bounded so arbitrary input -// can't grow the cache unchecked. `null` marks a timezone Intl doesn't support. -const MAX_FORMATTER_ENTRIES = 100; -const formatterCache = new Map(); + return new Intl.DateTimeFormat('en-CA', { + timeZone: timezone, + year: 'numeric', + month: '2-digit', + day: '2-digit' + }); +}); -function getFormatter(timezone: string): Intl.DateTimeFormat | null { - let formatter = formatterCache.get(timezone); +function getDateParts(date: unknown, timezone: string): DateParts { + const timestamp = date instanceof Date ? date.getTime() : date; + const isIntlTimestamp = typeof timestamp === 'number' && timestamp >= MIN_TIMESTAMP && timestamp <= MAX_TIMESTAMP; + // the active moment locale is global, so check it on every call + const isDefaultLocale = moment.localeData().postformat === DEFAULT_POSTFORMAT; - if (formatter === undefined) { + if (isIntlTimestamp && isDefaultLocale) { try { - // en-CA formats as YYYY-MM-DD - formatter = new Intl.DateTimeFormat('en-CA', { - timeZone: timezone, - year: 'numeric', - month: '2-digit', - day: '2-digit' - }); + const formatted = getFormatter(timezone).format(timestamp); + return {year: formatted.slice(0, 4), month: formatted.slice(5, 7), day: formatted.slice(8, 10)}; } catch { - formatter = null; + // unknown or unsupported timezone, use moment below } - - if (formatterCache.size >= MAX_FORMATTER_ENTRIES) { - formatterCache.clear(); - } - formatterCache.set(timezone, formatter); } - return formatter; -} - -function getTimestamp(date: unknown): number | null { - if (date instanceof Date) { - return date.getTime(); - } - - if (typeof date === 'number') { - return date; - } - - if (typeof date === 'string' && ISO_WITH_OFFSET_REGEX.test(date)) { - return Date.parse(date); - } - - return null; -} - -function getDatePartsFast(date: unknown, timezone: string): DateParts | null { - if (typeof timezone !== 'string') { - return null; - } - - const timestamp = getTimestamp(date); - - if (timestamp === null || !(timestamp >= MIN_FAST_TIMESTAMP && timestamp <= MAX_FAST_TIMESTAMP)) { - return null; - } - - const formatter = getFormatter(timezone); - - if (!formatter) { - return null; - } - - const formatted = formatter.format(timestamp); - - // guard against ICU/CLDR changes to the en-CA date format - if (formatted.length !== 10 || formatted.charCodeAt(4) !== 45 || formatted.charCodeAt(7) !== 45) { - return null; - } + // moment handles everything Intl can't match exactly: strings (parsed in the + // site timezone), invalid dates, years outside 1900-9999, unknown timezones and + // locales that rewrite digits + const publishedAt = moment.tz(date as moment.MomentInput, timezone); return { - year: formatted.slice(0, 4), - month: formatted.slice(5, 7), - day: formatted.slice(8, 10) + year: publishedAt.format('YYYY'), + month: publishedAt.format('MM'), + day: publishedAt.format('DD') }; } -function getDateParts(date: unknown, timezone: string): DateParts { - const fastParts = getDatePartsFast(date, timezone); - - if (fastParts) { - return fastParts; - } - - // fall back to moment-timezone for inputs Intl can't handle identically - const publishedAtMoment = moment.tz(date as moment.MomentInput, timezone); - - return { - year: publishedAtMoment.format('YYYY'), - month: publishedAtMoment.format('MM'), - day: publishedAtMoment.format('DD') - }; +function getPublishedDateParts(resource: PermalinkResource, timezone: string): DateParts { + return getDateParts(resource.published_at || Date.now(), timezone); } -const TOKEN_YEAR = 0; -const TOKEN_MONTH = 1; -const TOKEN_DAY = 2; -const TOKEN_AUTHOR = 3; -const TOKEN_PRIMARY_AUTHOR = 4; -const TOKEN_PRIMARY_TAG = 5; -const TOKEN_SLUG = 6; -const TOKEN_ID = 7; - -const TOKENS = new Map([ - [':year', TOKEN_YEAR], - [':month', TOKEN_MONTH], - [':day', TOKEN_DAY], - [':author', TOKEN_AUTHOR], - [':primary_author', TOKEN_PRIMARY_AUTHOR], - [':primary_tag', TOKEN_PRIMARY_TAG], - [':slug', TOKEN_SLUG], - [':id', TOKEN_ID] +const Token = { + Year: 0, + Month: 1, + Day: 2, + Author: 3, + PrimaryAuthor: 4, + PrimaryTag: 5, + Slug: 6, + Id: 7 +} as const; + +type Token = typeof Token[keyof typeof Token]; + +const TOKENS = new Map([ + [':year', Token.Year], + [':month', Token.Month], + [':day', Token.Day], + [':author', Token.Author], + [':primary_author', Token.PrimaryAuthor], + [':primary_tag', Token.PrimaryTag], + [':slug', Token.Slug], + [':id', Token.Id] ]); -// literal strings are output as-is, numbers are TOKEN_* values -type PermalinkPart = string | number; +// literal strings are output as-is +type PermalinkPart = string | Token; // Permalink patterns are a tiny set (one per collection/route), so parse each -// once rather than running the token regex on every call. Bounded so arbitrary -// input can't grow the cache unchecked. -const MAX_PERMALINK_ENTRIES = 100; -const permalinkCache = new Map(); - -function compilePermalink(permalink: string): PermalinkPart[] { - let parts = permalinkCache.get(permalink); - - if (parts) { - return parts; - } - - parts = []; +// once rather than running the token regex on every call +const compilePermalink = memoize(function compilePermalink(permalink: string): PermalinkPart[] { + const parts: PermalinkPart[] = []; let lastIndex = 0; for (const match of permalink.matchAll(/:[a-z_]+/g)) { @@ -174,57 +120,51 @@ function compilePermalink(permalink: string): PermalinkPart[] { parts.push(permalink.slice(lastIndex)); } - if (permalinkCache.size >= MAX_PERMALINK_ENTRIES) { - permalinkCache.clear(); - } - permalinkCache.set(permalink, parts); - return parts; -} +}); /** * creates the url path for a post based on blog timezone and permalink pattern */ function replacePermalink(permalink: string, resource: PermalinkResource, timezone: string = 'UTC'): string { - const primaryTagFallback = 'all'; - const parts = compilePermalink(permalink); + // used in place of a missing primary tag or author + const fallbackSlug = 'all'; // date parts are only computed if the permalink contains a date token let dateParts: DateParts | undefined; let result = ''; - for (const part of parts) { + for (const part of compilePermalink(permalink)) { if (typeof part === 'string') { result += part; continue; } - if (part <= TOKEN_DAY && !dateParts) { - dateParts = getDateParts(resource.published_at || Date.now(), timezone); - } - switch (part) { - case TOKEN_YEAR: - result += dateParts!.year; + case Token.Year: + dateParts ??= getPublishedDateParts(resource, timezone); + result += dateParts.year; break; - case TOKEN_MONTH: - result += dateParts!.month; + case Token.Month: + dateParts ??= getPublishedDateParts(resource, timezone); + result += dateParts.month; break; - case TOKEN_DAY: - result += dateParts!.day; + case Token.Day: + dateParts ??= getPublishedDateParts(resource, timezone); + result += dateParts.day; break; - case TOKEN_AUTHOR: + case Token.Author: result += resource.primary_author?.slug ?? 'undefined'; break; - case TOKEN_PRIMARY_AUTHOR: - result += resource.primary_author ? resource.primary_author.slug : primaryTagFallback; + case Token.PrimaryAuthor: + result += resource.primary_author ? resource.primary_author.slug : fallbackSlug; break; - case TOKEN_PRIMARY_TAG: - result += resource.primary_tag ? resource.primary_tag.slug : primaryTagFallback; + case Token.PrimaryTag: + result += resource.primary_tag ? resource.primary_tag.slug : fallbackSlug; break; - case TOKEN_SLUG: + case Token.Slug: result += resource.slug; break; - case TOKEN_ID: + case Token.Id: result += resource.id; break; } diff --git a/packages/url-utils/test/unit/utils/replace-permalink-equivalence.test.js b/packages/url-utils/test/unit/utils/replace-permalink-equivalence.test.js index a9a751bca..52d8e7cce 100644 --- a/packages/url-utils/test/unit/utils/replace-permalink-equivalence.test.js +++ b/packages/url-utils/test/unit/utils/replace-permalink-equivalence.test.js @@ -3,6 +3,7 @@ require('../../utils'); const sinon = require('sinon'); +const moment = require('moment-timezone'); const replacePermalink = require('../../../lib/utils/replace-permalink').default; const legacy = require('../../utils/legacy/replace-permalink'); @@ -34,6 +35,9 @@ const timezones = [ 'Australia/Lord_Howe', 'Etc/GMT+12', 'US/Pacific', + // Intl accepts offsets but moment doesn't know them and falls back to UTC + '+01:00', + '-05:30', 'europe/berlin' ]; @@ -61,6 +65,9 @@ const publishedAts = [ new Date('1899-12-31T23:30:00.000Z'), new Date('1066-10-14T12:00:00.000Z'), new Date('9999-12-31T12:00:00.000Z'), + new Date('+010000-01-01T12:00:00.000Z'), + new Date('-000001-06-15T12:00:00.000Z'), + new Date('0999-06-15T12:00:00.000Z'), new Date('invalid'), 1463553000000, 1463553000000.7, @@ -95,12 +102,15 @@ describe('utils: replacePermalink() equivalence with 5.3.0', function () { let clock; beforeEach(function () { - // moment logs a deprecation warning for non-ISO date strings + // moment logs a deprecation warning for non-ISO date strings, and an + // error for timezones it doesn't know sinon.stub(console, 'warn'); + sinon.stub(console, 'error'); clock = sinon.useFakeTimers(new Date('2016-05-17T23:30:00.000Z')); }); afterEach(function () { + moment.locale('en'); clock.restore(); sinon.restore(); }); @@ -119,6 +129,25 @@ describe('utils: replacePermalink() equivalence with 5.3.0', function () { } }); + it('produces identical output for moment locales that rewrite digits', function () { + for (const locale of ['ar', 'hi', 'fa', 'en-gb']) { + moment.locale(locale); + moment.locale().should.equal(locale); + + for (const timezone of ['UTC', 'Pacific/Kiritimati']) { + for (const publishedAt of [new Date('2016-05-17T23:30:00.000Z'), 1463553000000, '2016-05-17T23:30:00.000Z']) { + const resource = resourceFor(publishedAt); + replacePermalink('/:year/:month/:day/:slug/', resource, timezone) + .should.equal(legacy.replacePermalink('/:year/:month/:day/:slug/', resource, timezone), `locale: ${locale}`); + } + } + } + + moment.locale('ar'); + replacePermalink('/:year/:slug/', resourceFor(new Date('2016-05-17T23:30:00.000Z'))) + .should.equal('/٢٠١٦/short-and-sweet/'); + }); + it('produces identical output for missing primary tag and author', function () { const resource = {id: '1', slug: 'slug', published_at: new Date('2016-05-17T23:30:00.000Z')}; @@ -129,8 +158,6 @@ describe('utils: replacePermalink() equivalence with 5.3.0', function () { }); it('produces identical output for timezones Intl does not support', function () { - // moment-timezone logs an error for unknown timezones - sinon.stub(console, 'error'); const resource = resourceFor(new Date('2016-05-17T23:30:00.000Z')); for (const timezone of ['Not/A_Zone', null]) {