|
|
@@ -1,19 +1,17 @@
|
|
|
import { v4 as uuid } from 'uuid'
|
|
|
+import { fetchRange } from './features/pdf-preview/util/pdf-caching'
|
|
|
const OError = require('@overleaf/o-error')
|
|
|
|
|
|
// VERSION should get incremented when making changes to caching behavior or
|
|
|
// adjusting metrics collection.
|
|
|
// Keep in sync with PdfJsMetrics.
|
|
|
-const VERSION = 2
|
|
|
+const VERSION = 3
|
|
|
|
|
|
const CLEAR_CACHE_REQUEST_MATCHER = /^\/project\/[0-9a-f]{24}\/output$/
|
|
|
const COMPILE_REQUEST_MATCHER = /^\/project\/[0-9a-f]{24}\/compile$/
|
|
|
const PDF_REQUEST_MATCHER =
|
|
|
/^(\/zone\/.)?(\/project\/[0-9a-f]{24}\/.*\/output.pdf)$/
|
|
|
const PDF_JS_CHUNK_SIZE = 128 * 1024
|
|
|
-const MAX_SUBREQUEST_COUNT = 4
|
|
|
-const MAX_SUBREQUEST_BYTES = 4 * PDF_JS_CHUNK_SIZE
|
|
|
-const INCREMENTAL_CACHE_SIZE = 1000
|
|
|
|
|
|
// Each compile request defines a context (essentially the specific pdf file for
|
|
|
// that compile), requests for that pdf file can use the hashes in the compile
|
|
|
@@ -101,57 +99,6 @@ function expirePdfContexts() {
|
|
|
})
|
|
|
}
|
|
|
|
|
|
-/**
|
|
|
- *
|
|
|
- * @param {Object} metrics
|
|
|
- * @param {number} size
|
|
|
- * @param {number} cachedCount
|
|
|
- * @param {number} cachedBytes
|
|
|
- * @param {number} fetchedCount
|
|
|
- * @param {number} fetchedBytes
|
|
|
- */
|
|
|
-function trackDownloadStats(
|
|
|
- metrics,
|
|
|
- { size, cachedCount, cachedBytes, fetchedCount, fetchedBytes }
|
|
|
-) {
|
|
|
- metrics.cachedCount += cachedCount
|
|
|
- metrics.cachedBytes += cachedBytes
|
|
|
- metrics.fetchedCount += fetchedCount
|
|
|
- metrics.fetchedBytes += fetchedBytes
|
|
|
- metrics.requestedCount++
|
|
|
- metrics.requestedBytes += size
|
|
|
-}
|
|
|
-
|
|
|
-/**
|
|
|
- * @param {Object} metrics
|
|
|
- * @param {boolean} sizeDiffers
|
|
|
- * @param {boolean} mismatch
|
|
|
- * @param {boolean} success
|
|
|
- */
|
|
|
-function trackChunkVerify(metrics, { sizeDiffers, mismatch, success }) {
|
|
|
- if (sizeDiffers) {
|
|
|
- metrics.chunkVerifySizeDiffers |= 0
|
|
|
- metrics.chunkVerifySizeDiffers += 1
|
|
|
- }
|
|
|
- if (mismatch) {
|
|
|
- metrics.chunkVerifyMismatch |= 0
|
|
|
- metrics.chunkVerifyMismatch += 1
|
|
|
- }
|
|
|
- if (success) {
|
|
|
- metrics.chunkVerifySuccess |= 0
|
|
|
- metrics.chunkVerifySuccess += 1
|
|
|
- }
|
|
|
-}
|
|
|
-
|
|
|
-/**
|
|
|
- * @param {Array} chunks
|
|
|
- */
|
|
|
-function countBytes(chunks) {
|
|
|
- return chunks.reduce((totalBytes, chunk) => {
|
|
|
- return totalBytes + (chunk.end - chunk.start)
|
|
|
- }, 0)
|
|
|
-}
|
|
|
-
|
|
|
/**
|
|
|
* @param {FetchEvent} event
|
|
|
*/
|
|
|
@@ -266,7 +213,6 @@ function processPdfRequest(
|
|
|
return event.respondWith(response)
|
|
|
}
|
|
|
|
|
|
- const verifyChunks = event.request.url.includes('verify_chunks=true')
|
|
|
const rangeHeader =
|
|
|
event.request.headers.get('Range') || `bytes=0-${file.size - 1}`
|
|
|
const [start, last] = rangeHeader
|
|
|
@@ -275,307 +221,35 @@ function processPdfRequest(
|
|
|
.map(i => parseInt(i, 10))
|
|
|
const end = last + 1
|
|
|
|
|
|
- // Check that handling the range request won't trigger excessive subrequests,
|
|
|
- // (to avoid unwanted latency compared to the original request).
|
|
|
- const { chunks, newChunks } = cutRequestAmplification(
|
|
|
- getMatchingChunks(file.ranges, start, end),
|
|
|
- cached,
|
|
|
- metrics
|
|
|
- )
|
|
|
- const dynamicChunks = getInterleavingDynamicChunks(chunks, start, end)
|
|
|
- const chunksSize = countBytes(newChunks)
|
|
|
- const size = end - start
|
|
|
-
|
|
|
- if (chunks.length === 0 && dynamicChunks.length === 1) {
|
|
|
- // fall back to the original range request when no chunks are cached.
|
|
|
- trackDownloadStats(metrics, {
|
|
|
- size,
|
|
|
- cachedCount: 0,
|
|
|
- cachedBytes: 0,
|
|
|
- fetchedCount: 1,
|
|
|
- fetchedBytes: size,
|
|
|
- })
|
|
|
- return
|
|
|
- }
|
|
|
- if (
|
|
|
- chunksSize > MAX_SUBREQUEST_BYTES &&
|
|
|
- !(dynamicChunks.length === 0 && newChunks.length <= 1)
|
|
|
- ) {
|
|
|
- // fall back to the original range request when a very large amount of
|
|
|
- // object data would be requested, unless it is the only object in the
|
|
|
- // request or everything is already cached.
|
|
|
- metrics.tooLargeOverheadCount++
|
|
|
- trackDownloadStats(metrics, {
|
|
|
- size,
|
|
|
- cachedCount: 0,
|
|
|
- cachedBytes: 0,
|
|
|
- fetchedCount: 1,
|
|
|
- fetchedBytes: size,
|
|
|
+ return event.respondWith(
|
|
|
+ fetchRange({
|
|
|
+ url: event.request.url,
|
|
|
+ start,
|
|
|
+ end,
|
|
|
+ file,
|
|
|
+ pdfCreatedAt,
|
|
|
+ metrics,
|
|
|
+ cached,
|
|
|
})
|
|
|
- return
|
|
|
- }
|
|
|
-
|
|
|
- // URL prefix is /project/:id/user/:id/build/... or /project/:id/build/...
|
|
|
- // for authenticated and unauthenticated users respectively.
|
|
|
- const perUserPrefix = file.url.slice(0, file.url.indexOf('/build/'))
|
|
|
- const byteRanges = dynamicChunks
|
|
|
- .map(chunk => `${chunk.start}-${chunk.end - 1}`)
|
|
|
- .join(',')
|
|
|
- const coalescedDynamicChunks = []
|
|
|
- switch (dynamicChunks.length) {
|
|
|
- case 0:
|
|
|
- break
|
|
|
- case 1:
|
|
|
- coalescedDynamicChunks.push({
|
|
|
- chunk: dynamicChunks[0],
|
|
|
- url: event.request.url,
|
|
|
- init: { headers: { Range: `bytes=${byteRanges}` } },
|
|
|
- })
|
|
|
- break
|
|
|
- default:
|
|
|
- coalescedDynamicChunks.push({
|
|
|
- chunk: dynamicChunks,
|
|
|
- url: event.request.url,
|
|
|
- init: { headers: { Range: `bytes=${byteRanges}` } },
|
|
|
- })
|
|
|
- }
|
|
|
- const requests = chunks
|
|
|
- .map(chunk => {
|
|
|
- const path = `${perUserPrefix}/content/${file.contentId}/${chunk.hash}`
|
|
|
- const url = new URL(path, event.request.url)
|
|
|
- if (clsiServerId) {
|
|
|
- url.searchParams.set('clsiserverid', clsiServerId)
|
|
|
- }
|
|
|
- if (compileGroup) {
|
|
|
- url.searchParams.set('compileGroup', compileGroup)
|
|
|
- }
|
|
|
- return { chunk, url: url.toString() }
|
|
|
- })
|
|
|
- .concat(coalescedDynamicChunks)
|
|
|
- let cachedCount = 0
|
|
|
- let cachedBytes = 0
|
|
|
- let fetchedCount = 0
|
|
|
- let fetchedBytes = 0
|
|
|
- const reAssembledBlob = new Uint8Array(size)
|
|
|
- event.respondWith(
|
|
|
- Promise.all(
|
|
|
- requests.map(({ chunk, url, init }) =>
|
|
|
- fetch(url, init)
|
|
|
- .then(response => {
|
|
|
- if (!(response.status === 206 || response.status === 200)) {
|
|
|
- throw new OError(
|
|
|
- 'non successful response status: ' + response.status
|
|
|
- )
|
|
|
- }
|
|
|
- const boundary = getMultipartBoundary(response)
|
|
|
- if (Array.isArray(chunk) && !boundary) {
|
|
|
- throw new OError('missing boundary on multipart request', {
|
|
|
- headers: Object.fromEntries(response.headers.entries()),
|
|
|
- chunk,
|
|
|
- })
|
|
|
- }
|
|
|
- const blobFetchDate = getServerTime(response)
|
|
|
- const blobSize = getResponseSize(response)
|
|
|
- if (blobFetchDate && blobSize) {
|
|
|
- const chunkSize =
|
|
|
- Math.min(end, chunk.end) - Math.max(start, chunk.start)
|
|
|
- // Example: 2MB PDF, 1MB image, 128KB PDF.js chunk.
|
|
|
- // | pdf.js chunk |
|
|
|
- // | A BIG IMAGE BLOB |
|
|
|
- // | THE FULL PDF |
|
|
|
- if (blobFetchDate < pdfCreatedAt) {
|
|
|
- cachedCount++
|
|
|
- cachedBytes += chunkSize
|
|
|
- // Roll the position of the hash in the Map.
|
|
|
- cached.delete(chunk.hash)
|
|
|
- cached.add(chunk.hash)
|
|
|
- } else {
|
|
|
- // Blobs are fetched in bulk.
|
|
|
- fetchedCount++
|
|
|
- fetchedBytes += blobSize
|
|
|
- }
|
|
|
- }
|
|
|
- return response
|
|
|
- .blob()
|
|
|
- .then(blob => blob.arrayBuffer())
|
|
|
- .then(arraybuffer => {
|
|
|
- return {
|
|
|
- boundary,
|
|
|
- chunk,
|
|
|
- data: backFillObjectContext(chunk, arraybuffer),
|
|
|
- }
|
|
|
- })
|
|
|
- })
|
|
|
- .catch(error => {
|
|
|
- throw OError.tag(error, 'cannot fetch chunk', { url })
|
|
|
- })
|
|
|
- )
|
|
|
- )
|
|
|
- .then(rawResponses => {
|
|
|
- const responses = []
|
|
|
- for (const response of rawResponses) {
|
|
|
- if (response.boundary) {
|
|
|
- responses.push(
|
|
|
- ...getMultiPartResponses(response, file, metrics, verifyChunks)
|
|
|
- )
|
|
|
- } else {
|
|
|
- responses.push(response)
|
|
|
- }
|
|
|
- }
|
|
|
- responses.forEach(({ chunk, data }) => {
|
|
|
- // overlap:
|
|
|
- // | REQUESTED_RANGE |
|
|
|
- // | CHUNK |
|
|
|
- const offsetStart = Math.max(start - chunk.start, 0)
|
|
|
- // overlap:
|
|
|
- // | REQUESTED_RANGE |
|
|
|
- // | CHUNK |
|
|
|
- const offsetEnd = Math.max(chunk.end - end, 0)
|
|
|
- if (offsetStart > 0 || offsetEnd > 0) {
|
|
|
- // compute index positions for slice to handle case where offsetEnd=0
|
|
|
- const chunkSize = chunk.end - chunk.start
|
|
|
- data = data.subarray(offsetStart, chunkSize - offsetEnd)
|
|
|
- }
|
|
|
- const insertPosition = Math.max(chunk.start - start, 0)
|
|
|
- reAssembledBlob.set(data, insertPosition)
|
|
|
- })
|
|
|
-
|
|
|
- let verifyProcess = Promise.resolve(reAssembledBlob)
|
|
|
- if (verifyChunks) {
|
|
|
- verifyProcess = fetch(event.request)
|
|
|
- .then(response => response.arrayBuffer())
|
|
|
- .then(arrayBuffer => {
|
|
|
- const fullBlob = new Uint8Array(arrayBuffer)
|
|
|
- const stats = {}
|
|
|
- if (reAssembledBlob.byteLength !== fullBlob.byteLength) {
|
|
|
- stats.sizeDiffers = true
|
|
|
- } else if (
|
|
|
- !reAssembledBlob.every((v, idx) => v === fullBlob[idx])
|
|
|
- ) {
|
|
|
- stats.mismatch = true
|
|
|
- } else {
|
|
|
- stats.success = true
|
|
|
- }
|
|
|
- trackChunkVerify(metrics, stats)
|
|
|
- if (stats.success === true) {
|
|
|
- return reAssembledBlob
|
|
|
- } else {
|
|
|
- return fullBlob
|
|
|
- }
|
|
|
- })
|
|
|
- }
|
|
|
-
|
|
|
- return verifyProcess.then(blob => {
|
|
|
- trackDownloadStats(metrics, {
|
|
|
- size,
|
|
|
- cachedCount,
|
|
|
- cachedBytes,
|
|
|
- fetchedCount,
|
|
|
- fetchedBytes,
|
|
|
- })
|
|
|
- return new Response(blob, {
|
|
|
- status: 206,
|
|
|
- headers: {
|
|
|
- 'Accept-Ranges': 'bytes',
|
|
|
- 'Content-Length': size,
|
|
|
- 'Content-Range': `bytes ${start}-${last}/${file.size}`,
|
|
|
- 'Content-Type': 'application/pdf',
|
|
|
- },
|
|
|
- })
|
|
|
+ .then(blob => {
|
|
|
+ return new Response(blob, {
|
|
|
+ status: 206,
|
|
|
+ headers: {
|
|
|
+ 'Accept-Ranges': 'bytes',
|
|
|
+ 'Content-Length': end - start,
|
|
|
+ 'Content-Range': `bytes ${start}-${last}/${file.size}`,
|
|
|
+ 'Content-Type': 'application/pdf',
|
|
|
+ },
|
|
|
})
|
|
|
})
|
|
|
.catch(error => {
|
|
|
- fetchedBytes += size
|
|
|
metrics.failedCount++
|
|
|
- trackDownloadStats(metrics, {
|
|
|
- size,
|
|
|
- cachedCount: 0,
|
|
|
- cachedBytes: 0,
|
|
|
- fetchedCount,
|
|
|
- fetchedBytes,
|
|
|
- })
|
|
|
reportError(event, OError.tag(error, 'failed to compose pdf response'))
|
|
|
return fetch(event.request)
|
|
|
})
|
|
|
)
|
|
|
}
|
|
|
|
|
|
-/**
|
|
|
- *
|
|
|
- * @param {Response} response
|
|
|
- */
|
|
|
-function getServerTime(response) {
|
|
|
- const raw = response.headers.get('Date')
|
|
|
- if (!raw) return new Date()
|
|
|
- return new Date(raw)
|
|
|
-}
|
|
|
-
|
|
|
-/**
|
|
|
- *
|
|
|
- * @param {Response} response
|
|
|
- */
|
|
|
-function getResponseSize(response) {
|
|
|
- const raw = response.headers.get('Content-Length')
|
|
|
- if (!raw) return 0
|
|
|
- return parseInt(raw, 10)
|
|
|
-}
|
|
|
-
|
|
|
-/**
|
|
|
- *
|
|
|
- * @param {Response} response
|
|
|
- */
|
|
|
-function getMultipartBoundary(response) {
|
|
|
- const raw = response.headers.get('Content-Type')
|
|
|
- if (!raw.includes('multipart/byteranges')) return ''
|
|
|
- const idx = raw.indexOf('boundary=')
|
|
|
- if (idx === -1) return ''
|
|
|
- return raw.slice(idx + 'boundary='.length)
|
|
|
-}
|
|
|
-
|
|
|
-/**
|
|
|
- * @param {Object} response
|
|
|
- * @param {Object} file
|
|
|
- * @param {Object} metrics
|
|
|
- * @param {boolean} verifyChunks
|
|
|
- */
|
|
|
-function getMultiPartResponses(response, file, metrics, verifyChunks) {
|
|
|
- const { chunk: chunks, data, boundary } = response
|
|
|
- const responses = []
|
|
|
- let offsetStart = 0
|
|
|
- for (const chunk of chunks) {
|
|
|
- const header = `\r\n--${boundary}\r\nContent-Type: application/pdf\r\nContent-Range: bytes ${
|
|
|
- chunk.start
|
|
|
- }-${chunk.end - 1}/${file.size}\r\n\r\n`
|
|
|
- const headerSize = header.length
|
|
|
-
|
|
|
- // Verify header content. A proxy might have tampered with it.
|
|
|
- const headerRaw = ENCODER.encode(header)
|
|
|
- if (
|
|
|
- !data
|
|
|
- .subarray(offsetStart, offsetStart + headerSize)
|
|
|
- .every((v, idx) => v === headerRaw[idx])
|
|
|
- ) {
|
|
|
- metrics.headerVerifyFailure |= 0
|
|
|
- metrics.headerVerifyFailure++
|
|
|
- throw new OError('multipart response header does not match', {
|
|
|
- actual: new TextDecoder().decode(
|
|
|
- data.subarray(offsetStart, offsetStart + headerSize)
|
|
|
- ),
|
|
|
- expected: header,
|
|
|
- })
|
|
|
- }
|
|
|
-
|
|
|
- offsetStart += headerSize
|
|
|
- const chunkSize = chunk.end - chunk.start
|
|
|
- responses.push({
|
|
|
- chunk,
|
|
|
- data: data.subarray(offsetStart, offsetStart + chunkSize),
|
|
|
- })
|
|
|
- offsetStart += chunkSize
|
|
|
- }
|
|
|
- return responses
|
|
|
-}
|
|
|
-
|
|
|
/**
|
|
|
* @param {FetchEvent} event
|
|
|
* @param {Response} response
|
|
|
@@ -584,15 +258,11 @@ function getMultiPartResponses(response, file, metrics, verifyChunks) {
|
|
|
function handleCompileResponse(event, response, body) {
|
|
|
if (!body || body.status !== 'success') return
|
|
|
|
|
|
- const pdfCreatedAt = getServerTime(response)
|
|
|
-
|
|
|
for (const file of body.outputFiles) {
|
|
|
if (file.path !== 'output.pdf') continue // not the pdf used for rendering
|
|
|
- if (file.ranges) {
|
|
|
- file.ranges.forEach(backFillEdgeBounds)
|
|
|
+ if (file.ranges?.length) {
|
|
|
const { clsiServerId, compileGroup } = body
|
|
|
registerPdfContext(event.clientId, file.url, {
|
|
|
- pdfCreatedAt,
|
|
|
file,
|
|
|
clsiServerId,
|
|
|
compileGroup,
|
|
|
@@ -602,117 +272,6 @@ function handleCompileResponse(event, response, body) {
|
|
|
}
|
|
|
}
|
|
|
|
|
|
-const ENCODER = new TextEncoder()
|
|
|
-function backFillEdgeBounds(chunk) {
|
|
|
- if (chunk.objectId) {
|
|
|
- chunk.objectId = ENCODER.encode(chunk.objectId)
|
|
|
- chunk.start -= chunk.objectId.byteLength
|
|
|
- }
|
|
|
- return chunk
|
|
|
-}
|
|
|
-
|
|
|
-/**
|
|
|
- * @param chunk
|
|
|
- * @param {ArrayBuffer} arrayBuffer
|
|
|
- * @return {Uint8Array}
|
|
|
- */
|
|
|
-function backFillObjectContext(chunk, arrayBuffer) {
|
|
|
- if (!chunk.objectId) {
|
|
|
- // This is a dynamic chunk
|
|
|
- return new Uint8Array(arrayBuffer)
|
|
|
- }
|
|
|
- const { start, end, objectId } = chunk
|
|
|
- const header = Uint8Array.from(objectId)
|
|
|
- const fullBuffer = new Uint8Array(end - start)
|
|
|
- fullBuffer.set(header, 0)
|
|
|
- fullBuffer.set(new Uint8Array(arrayBuffer), objectId.length)
|
|
|
- return fullBuffer
|
|
|
-}
|
|
|
-
|
|
|
-/**
|
|
|
- * @param {Array} chunks
|
|
|
- * @param {number} start
|
|
|
- * @param {number} end
|
|
|
- * @returns {Array}
|
|
|
- */
|
|
|
-function getMatchingChunks(chunks, start, end) {
|
|
|
- const matchingChunks = []
|
|
|
- for (const chunk of chunks) {
|
|
|
- if (chunk.end <= start) {
|
|
|
- // no overlap:
|
|
|
- // | REQUESTED_RANGE |
|
|
|
- // | CHUNK |
|
|
|
- continue
|
|
|
- }
|
|
|
- if (chunk.start >= end) {
|
|
|
- // no overlap:
|
|
|
- // | REQUESTED_RANGE |
|
|
|
- // | CHUNK |
|
|
|
- break
|
|
|
- }
|
|
|
- matchingChunks.push(chunk)
|
|
|
- }
|
|
|
- return matchingChunks
|
|
|
-}
|
|
|
-
|
|
|
-/**
|
|
|
- * @param {Array} potentialChunks
|
|
|
- * @param {Set} cached
|
|
|
- * @param {Object} metrics
|
|
|
- */
|
|
|
-function cutRequestAmplification(potentialChunks, cached, metrics) {
|
|
|
- const chunks = []
|
|
|
- const newChunks = []
|
|
|
- let tooManyRequests = false
|
|
|
- for (const chunk of potentialChunks) {
|
|
|
- if (cached.has(chunk.hash)) {
|
|
|
- chunks.push(chunk)
|
|
|
- continue
|
|
|
- }
|
|
|
- if (newChunks.length < MAX_SUBREQUEST_COUNT) {
|
|
|
- chunks.push(chunk)
|
|
|
- newChunks.push(chunk)
|
|
|
- } else {
|
|
|
- tooManyRequests = true
|
|
|
- }
|
|
|
- }
|
|
|
- if (tooManyRequests) {
|
|
|
- metrics.tooManyRequestsCount++
|
|
|
- }
|
|
|
- if (cached.size > INCREMENTAL_CACHE_SIZE) {
|
|
|
- for (const key of cached) {
|
|
|
- if (cached.size < INCREMENTAL_CACHE_SIZE) {
|
|
|
- break
|
|
|
- }
|
|
|
- // Map keys are stored in insertion order.
|
|
|
- // We re-insert keys on cache hit, 'cached' is a cheap LRU.
|
|
|
- cached.delete(key)
|
|
|
- }
|
|
|
- }
|
|
|
- return { chunks, newChunks }
|
|
|
-}
|
|
|
-
|
|
|
-/**
|
|
|
- * @param {Array} chunks
|
|
|
- * @param {number} start
|
|
|
- * @param {number} end
|
|
|
- * @returns {Array}
|
|
|
- */
|
|
|
-function getInterleavingDynamicChunks(chunks, start, end) {
|
|
|
- const dynamicChunks = []
|
|
|
- for (const chunk of chunks) {
|
|
|
- if (start < chunk.start) {
|
|
|
- dynamicChunks.push({ start, end: chunk.start })
|
|
|
- }
|
|
|
- start = chunk.end
|
|
|
- }
|
|
|
-
|
|
|
- if (start < end) {
|
|
|
- dynamicChunks.push({ start, end })
|
|
|
- }
|
|
|
- return dynamicChunks
|
|
|
-}
|
|
|
-
|
|
|
/**
|
|
|
* @param {FetchEvent} event
|
|
|
*/
|