Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions docs/index.html
Original file line number Diff line number Diff line change
Expand Up @@ -152,6 +152,7 @@ <h4>
</footer>
</div>
<script src="./js/jquery-2.1.3.min.js"></script>
<script src="https://cdn.jsdelivr.net/npm/dompurify@3.3.1/dist/purify.min.js"></script>
<script src="./js/search.js"></script>
<script src="./js/bootstrap.min.js"></script>
<script>
Expand Down
72 changes: 57 additions & 15 deletions docs/js/search.js
Original file line number Diff line number Diff line change
Expand Up @@ -11,32 +11,69 @@ var REF_STATUSES = {
, "REC": "W3C Recommendation"
};

function safeSetText(element, text) {
element.textContent = text;
return element;
}

function highlight(txt, searchString) {
var regexp = new RegExp("(<[^>]+>)|(" + searchString + ")", "gi");
// Escape the search string for use in regex to prevent injection
var escapedSearchString = searchString.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
var regexp = new RegExp("(<[^>]+>)|(" + escapedSearchString + ")", "gi");
var flag = false;
return (txt || "").replace(regexp, function wrap(_, tag, txt) {
if (tag == "<code>") flag = true;
if (tag == "</code>") flag = false;
if (tag) return tag;
if (flag) return txt;
return "<mark>" + txt + "</mark>";
return "<mark>" + DOMPurify.sanitize(txt) + "</mark>";
});
Comment thread
marcoscaceres marked this conversation as resolved.
}

function safeUrl(url) {
if (!url || typeof url !== 'string') return '';
// Block javascript: and data: URIs for security
if (url.toLowerCase().startsWith('javascript:') || url.toLowerCase().startsWith('data:')) {
return '';
}
return url;
}

function createSafeLink(href, text) {
if (!href) return DOMPurify.sanitize('<cite>' + (text || '') + '</cite>');

// Use DOM API to safely create link with proper attribute encoding
var a = document.createElement('a');
a.href = href; // Browser automatically encodes href attribute properly
a.innerHTML = DOMPurify.sanitize('<cite>' + (text || '') + '</cite>');
return a.outerHTML;
}

function stringifyRef(ref) {
var output = "";
var parts = [];

if (ref.authors && ref.authors.length) {
output += ref.authors.join("; ");
if (ref.etAl) output += " et al";
output += ". ";
var authors = DOMPurify.sanitize(ref.authors.join("; "));
if (ref.etAl) authors += " et al";
parts.push(authors + ". ");
}
var safeHref = safeUrl(ref.href);
var safeTitle = ref.title || '';
parts.push(createSafeLink(safeHref, safeTitle) + ". ");

if (ref.date) parts.push(DOMPurify.sanitize(ref.date) + ". ");
if (ref.status) parts.push(DOMPurify.sanitize(REF_STATUSES[ref.status] || ref.status) + ". ");

if (safeHref) {
parts.push('URL:&nbsp;' + createSafeLink(safeHref, safeHref));
}
if (ref.href) output += '<a href="' + ref.href + '"><cite>' + ref.title + "</cite></a>. ";
else output += '<cite>' + ref.title + '</cite>. ';
if (ref.date) output += ref.date + ". ";
if (ref.status) output += (REF_STATUSES[ref.status] || ref.status) + ". ";
if (ref.href) output += 'URL:&nbsp;<a href="' + ref.href + '">' + ref.href + "</a>";
if (ref.edDraft) output += ' ED:&nbsp;<a href="' + ref.edDraft + '">' + ref.edDraft + "</a>";
return "<div>" + output + "</div>";

var safeEdDraft = safeUrl(ref.edDraft);
if (safeEdDraft) {
parts.push(' ED:&nbsp;' + createSafeLink(safeEdDraft, safeEdDraft));
}

return "<div>" + parts.join('') + "</div>";
};

function pluralize (count, sing, plur) {
Expand Down Expand Up @@ -76,6 +113,7 @@ function prettifyApiOutput(obj) {
}

function msg(query, count) {
// Return plain text - escaping will be handled by safeSetText when displayed
if (count) {
return 'We found ' + pluralize(count, 'result', 'results') + ' for your search for "' + query + '".';
}
Expand Down Expand Up @@ -121,8 +159,12 @@ function setup($root) {
});

function update(query, output) {
$results.html(highlight(output.html, query));
$status.text(msg(query, output.count));
// Use DOMPurify to sanitize HTML before injection
var sanitizedHtml = DOMPurify.sanitize(highlight(output.html, query));
$results.html(sanitizedHtml);
// Use safe text setting for status message
var statusElement = $status[0];
safeSetText(statusElement, msg(query, output.count));
$search.select();
$('[data-toggle="tooltip"]').tooltip();
}
Expand Down
27 changes: 25 additions & 2 deletions index.js
Original file line number Diff line number Diff line change
@@ -1,6 +1,8 @@
var t0 = Date.now();

var bibref = require('./lib/bibref');
var sanitizeHtml = require('sanitize-html');
var helmet = require('helmet');

var app = module.exports = require("express")();

Expand All @@ -10,6 +12,18 @@ if (process.env.NODE_ENV == "dev" || process.env.NODE_ENV == "development") {
errorhandlerOptions.showStack = true;
}
app.enable("etag");

app.use(helmet({
contentSecurityPolicy: {
directives: {
defaultSrc: ["'self'"],
scriptSrc: ["'self'"],
objectSrc: ["'none'"]
}
},
crossOriginEmbedderPolicy: false
}));

var bannedIPs = [
// Palo Alto Networks bot
"34.96.130.0/24", "34.77.162.0/24", "34.86.35.0/24"
Expand All @@ -35,7 +49,10 @@ app.get('/bibrefs', function (req, res, next) {

// search
app.get('/search-refs', function (req, res, next) {
var q = (req.query["q"] || "").toLowerCase();
var q = sanitizeHtml(req.query["q"] || "", {
allowedTags: [],
allowedAttributes: {}
}).toLowerCase();
Comment on lines +52 to +55

Copilot AI Jan 19, 2026

Copy link

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The XSS protection added in this PR lacks test coverage. Consider adding tests that verify the sanitization of malicious input such as script tags, event handlers, and javascript: URLs in both the search-refs and reverse-lookup endpoints. This is especially important given that the PR specifically mentions validation against XSS payloads.

Copilot uses AI. Check for mistakes.
if (q) {
var obj = {};
var current, shortname;
Expand Down Expand Up @@ -95,7 +112,13 @@ app.get('/reverse-lookup', function (req, res, next) {
var refs,
urls = req.query["urls"];
if (urls) {
refs = bibref.reverseLookup(urls.split(","));
var sanitizedUrls = urls.split(",").map(function(url) {
return sanitizeHtml(url.trim(), {
allowedTags: [],
allowedAttributes: {}
});
});
refs = bibref.reverseLookup(sanitizedUrls);
Comment thread
marcoscaceres marked this conversation as resolved.
res.status(200).jsonp(refs);
} else {
res.status(400).jsonp({ message: "Missing urls parameter" });
Expand Down
31 changes: 26 additions & 5 deletions lib/bibref.js
Original file line number Diff line number Diff line change
@@ -1,6 +1,5 @@
var fs = require('fs'),
path = require('path'),
url = require('url'),
formatDate = require('./format-date');

var PROPS = [
Expand Down Expand Up @@ -115,10 +114,32 @@ function normalizeUrl(u) {
// Likewise, we consider trailing slashes optional and remove them
// when present (it's easier to remove a trailing slash than to add
// one where missing).
u = url.parse(u);
u = (u.hostname || "").replace(/^www\./, "") + (u.pathname || "").replace(/\/$/, "") + (u.search || "") + (u.hash || "");
// Case insensitive match.
return u.toLowerCase();
try {
// Validate that input is a string and looks like an absolute URL
if (!u || typeof u !== 'string') {
return '';
}

// If no protocol, default to https:// for security
var urlToParse = u;
if (!/^https?:\/\//.test(u)) {
urlToParse = 'https://' + u;
}

var urlObj = new URL(urlToParse);

// Block dangerous protocols
if (urlObj.protocol === 'javascript:' || urlObj.protocol === 'data:') {
return '';
}

u = (urlObj.hostname || "").replace(/^www\./, "") + (urlObj.pathname || "").replace(/\/$/, "") + (urlObj.search || "") + (urlObj.hash || "");
// Case insensitive match.
return u.toLowerCase();
} catch (e) {
// For truly malformed URLs, return empty string instead of unvalidated input
return '';
}
}

function prepareReverseLookup(refs) {
Expand Down
Loading
Loading