OurBigBook logoOurBigBook Docs OurBigBook logoOurBigBook.comSite Source code
web/bin/benchmark
#!/usr/bin/env node

const { execFileSync } = require('child_process')
const crypto = require('crypto')
const fs = require('fs')
const os = require('os')
const path = require('path')
const { performance } = require('perf_hooks')
const { databaseDialectHeader, explainDirectory, explainHeader } = require('../back/js')

const repoRoot = path.resolve(__dirname, '../..')
const defaultOutput = path.resolve(repoRoot, '../ourbigbook-media/benchmark.json')

async function detectDatabaseDialect(baseUrl, timeout=60000) {
  const response = await fetch(new URL('/api/', baseUrl), {
    redirect: 'manual', signal: AbortSignal.timeout(timeout),
  })
  await response.arrayBuffer()
  if (response.status !== 200) throw new Error(`Database detection failed: HTTP ${response.status} from /api/`)
  const dialect = response.headers.get(databaseDialectHeader)
  if (!['postgres', 'sqlite'].includes(dialect)) {
    throw new Error('Server did not report a supported database dialect. Restart it with the updated code.')
  }
  return dialect
}

function osVersion() {
  for (const filename of ['/etc/os-release', '/usr/lib/os-release']) {
    try {
      const match = fs.readFileSync(filename, 'utf8').match(/^PRETTY_NAME=(?:"([^"\r\n]*)"|'([^'\r\n]*)'|([^\r\n]*))$/m)
      if (match) return match[1] ?? match[2] ?? match[3]
    } catch (error) {
      // Distribution metadata is optional, including on non-Linux hosts.
    }
  }
  return null
}

function requestPath(pathname, params={}) {
  const query = new URLSearchParams(params)
  query.sort()
  return pathname + (query.size ? `?${query}` : '')
}

function pageOffsets(count, limit) {
  const last = Math.max(0, Math.ceil(count / limit) - 1)
  return [...new Set([0, Math.floor(last / 2), last])].map(page => page * limit)
}

function normalizePaths(paths) {
  const base = 'http://benchmark.invalid'
  const normalized = new Set()
  for (const request of paths) {
    let url
    try {
      url = new URL(request, base)
    } catch (error) {
      throw new Error(`Invalid benchmark path: ${request}`)
    }
    if (!request.startsWith('/') || request.startsWith('//') || url.origin !== base || url.hash) {
      throw new Error(`Benchmark paths must be same-origin paths without fragments: ${request}`)
    }
    normalized.add(request)
  }
  return [...normalized]
}

function buildPaths(db, { limit=20, pages=true }={}) {
  const paths = new Set(['/api/', '/api/min'])
  const add = (pathname, params) => paths.add(requestPath(pathname, params))
  const listings = [
    ['articles', db.articles, ['created', 'updated', 'score', 'id', 'follower-count', 'issues']],
    ['topics', db.nonempty_topics, ['created', 'updated', 'article-count', 'id']],
    ['users', db.users, ['created', 'score', 'discussions', 'comments', 'files', 'file-size']],
  ]
  for (const [object, count, sorts] of listings) {
    for (const offset of pageOffsets(count, limit)) {
      for (const sort of sorts) add(`/api/${object}`, { limit, offset, sort })
    }
  }
  const { author, article, issue } = db.samples
  if (author) {
    add('/api/users', { username: author.username })
    add('/api/users', { followedBy: author.username, limit })
    add('/api/users', { following: author.username, limit })
    for (const offset of pageOffsets(author.articles, limit)) {
      for (const sort of ['created', 'score', 'id']) {
        add('/api/articles', { author: author.username, limit, offset, sort })
      }
    }
    add('/api/articles', { followedBy: author.username, limit })
    add('/api/articles', { likedBy: author.username, limit })
    add('/api/articles', { parent: `@${author.username}`, limit })
    add('/api/articles/hash', { author: author.username, limit })
    add('/api/uploads/hash', { author: author.username, limit })
  }
  if (article) {
    add('/api/articles', { id: article.slug })
    add('/api/articles', { topicId: article.topicId, limit })
    if (article.topicId) {
      const search = article.topicId.slice(0, 8)
      add('/api/articles', { search, limit })
      add('/api/topics', { search, limit })
    }
    for (const sort of ['created', 'updated', 'score']) {
      add('/api/issues', { id: article.slug, sort })
    }
  }
  if (issue) {
    add('/api/issues', { id: issue.slug, number: issue.number })
    add(`/api/issues/${issue.number}/comments`, { id: issue.slug })
  }
  if (pages) {
    add('/cirosantilli')
    // HTML listings apply filters/order directions that differ from the API.
    // In particular these reproduce the expensive global article-list queries.
    for (const offset of pageOffsets(db.listed_articles, 20)) {
      for (const sort of ['created', 'score', 'id', 'id-desc']) {
        add('/-/articles', { page: offset / 20 + 1, sort })
      }
    }
    for (const [pathname, sorts] of [
      ['/', ['article-count', 'id']],
      ['/-/users', ['created', 'score']],
      ['/-/discussions', ['created', 'score']],
      ['/-/comments', ['created']],
      ['/-/files', ['created', 'size']],
    ]) {
      for (const sort of sorts) add(pathname, { sort })
    }
  }
  return [...paths]
}

async function collectDatabase(sequelize, { author: username }={}) {
  const { Op } = require('sequelize')
  const db = { dialect: sequelize.getDialect() }
  for (const [key, model] of Object.entries({
    articles: 'Article', users: 'User', topics: 'Topic', issues: 'Issue',
    comments: 'Comment', uploads: 'Upload', files: 'File', ids: 'Id', refs: 'Ref',
  })) {
    db[key] = await sequelize.models[model].count()
  }
  const { Article, User, Issue, Topic } = sequelize.models
  db.listed_articles = await Article.count({ where: { list: true } })
  db.nonempty_topics = await Topic.count({ where: { articleCount: { [Op.gt]: 0 } } })
  if (db.dialect === 'postgres') {
    const [[details]] = await sequelize.query(`SELECT current_database() AS name,
      current_setting('server_version') AS version,
      pg_database_size(current_database())::text AS size_bytes,
      current_setting('lc_collate') AS collation,
      current_setting('work_mem') AS work_mem,
      current_setting('shared_buffers') AS shared_buffers,
      current_setting('max_parallel_workers_per_gather') AS max_parallel_workers_per_gather,
      current_setting('jit') AS jit`)
    Object.assign(db, details, { size_bytes: Number(details.size_bytes) })
  } else {
    const [[details]] = await sequelize.query('SELECT sqlite_version() AS version')
    db.version = details.version
  }

  let author
  if (username) {
    author = await User.findOne({ where: { username }, attributes: ['id', 'username'], raw: true })
    if (!author) throw new Error(`Unknown author: ${username}`)
  } else {
    // Choose the largest author so per-author requests exercise real lists.
    const [largest] = await Article.findAll({
      attributes: ['authorId', [sequelize.fn('COUNT', sequelize.col('id')), 'n']],
      group: ['authorId'], order: [[sequelize.literal('n'), 'DESC'], ['authorId', 'ASC']],
      limit: 1, raw: true,
    })
    if (largest) author = await User.findByPk(largest.authorId, { attributes: ['id', 'username'], raw: true })
  }
  if (author) author.articles = await Article.count({ where: { authorId: author.id } })
  const article = await Article.findOne({
    // Home articles have an empty topicId: they would skip search benchmarks
    // and turn the topicId lookup into another unfiltered article listing.
    attributes: ['slug', 'topicId'], where: { list: true, topicId: { [Op.ne]: '' } },
    order: [['issueCount', 'DESC'], ['id', 'ASC']], raw: true,
  })
  const issue = await Issue.findOne({
    attributes: ['number'], where: { list: true }, order: [['id', 'ASC']],
    include: [{ model: Article, as: 'article', attributes: ['slug'], required: true }],
  })
  db.samples = {
    author: author || null,
    article,
    issue: issue ? { number: issue.number, slug: issue.article.slug } : null,
  }
  return db
}

function saveHistory(output, history) {
  fs.mkdirSync(path.dirname(output), { recursive: true })
  const temporary = `${output}.${process.pid}.tmp`
  try {
    fs.writeFileSync(temporary, JSON.stringify(history, null, 2) + '\n')
    fs.renameSync(temporary, output)
  } finally {
    if (fs.existsSync(temporary)) fs.unlinkSync(temporary)
  }
}

async function benchmark({ paths, about, output, explain, baseUrl='http://localhost:3000', runs=3, warmup=1, timeout=60000, log=console.log }) {
  if (explain === undefined) {
    explain = process.env.OURBIGBOOK_EXPLAIN && process.env.OURBIGBOOK_EXPLAIN !== '0'
      ? process.env.OURBIGBOOK_EXPLAIN : true
  }
  const history = fs.existsSync(output) ? JSON.parse(fs.readFileSync(output, 'utf8')) : []
  if (!Array.isArray(history)) throw new Error(`${output}: expected a JSON array; existing data was not changed`)
  const entry = {
    about: {
      ...about,
      timestamp: new Date().toISOString(),
      run_id: crypto.randomUUID(),
      base_url: baseUrl,
      timing_unit: 'ms',
      statistic: 'median',
      runs, warmup, timeout_ms: timeout,
      concurrency: 1,
      authentication: 'anonymous',
      planned_paths: paths.length,
      planned_requests: paths.length * (warmup + runs),
      completed: false,
    },
    results: {},
    samples: {},
    errors: {},
  }
  let explainOutput
  history.push(entry)
  saveHistory(output, history)
  for (const request of paths) {
    const samples = []
    try {
      for (let i = 0; i < warmup + runs; i++) {
        const requestId = explain ? crypto.randomUUID() : undefined
        if (requestId && entry.explain) {
          // Keep the ID even on HTTP timeout: the server may finish later and
          // leave a recoverable trace with this name in the spool directory.
          ;(entry.explain.requests[request] ||= []).push(requestId)
          saveHistory(output, history)
        }
        const start = performance.now()
        const response = await fetch(new URL(request, baseUrl), {
          redirect: 'manual', signal: AbortSignal.timeout(timeout),
          headers: requestId ? { [explainHeader]: requestId } : {},
        })
        // Include the complete response transfer, not just time to headers.
        const body = await response.arrayBuffer()
        const elapsed = performance.now() - start
        if (response.headers.has(explainHeader)) {
          entry.about.db_instrumentation = { provider: 'auto_explain', analyze: true, buffers: true, timing: false }
        }
        if (explain && response.headers.has(explainHeader)) {
          if (response.headers.get(explainHeader) !== requestId) {
            throw new Error('Server returned a mismatched query-plan request ID')
          }
          // Detect support using the actual benchmark response, including the
          // first warmup. Servers without tracing need no extra requests/files.
          if (!entry.explain) {
            const explainFilename = `${entry.about.run_id}.explain.jsonl`
            const explainRelative = path.join('benchmark', explainFilename)
            explainOutput = path.join(path.dirname(output), explainRelative)
            fs.mkdirSync(path.dirname(explainOutput), { recursive: true, mode: 0o700 })
            entry.explain = { file: explainRelative, format: 'jsonl', requests: { [request]: [requestId] } }
            fs.writeFileSync(explainOutput, '', { flag: 'wx', mode: 0o600 })
            saveHistory(output, history)
          }
          const filename = path.join(explainDirectory(explain), `${requestId}.json`)
          const deadline = performance.now() + timeout
          // The response can finish before an outstanding request-tracking query.
          while (!fs.existsSync(filename)) {
            if (performance.now() >= deadline) throw new Error(`Timed out waiting for query plans: ${filename}`)
            await new Promise(resolve => setTimeout(resolve, 25))
          }
          const trace = JSON.parse(fs.readFileSync(filename, 'utf8'))
          fs.appendFileSync(explainOutput, JSON.stringify({
            ...trace, run_id: entry.about.run_id,
            phase: i < warmup ? 'warmup' : 'sample',
            iteration: i < warmup ? i : i - warmup,
          }) + '\n', { mode: 0o600 })
          fs.unlinkSync(filename)
          saveHistory(output, history)
        }
        if (response.status !== 200) {
          const detail = Buffer.from(body).toString('utf8').slice(0, 2000)
          const hint = response.status === 404 ? ' Check that DATABASE_URL and database data match the running server.' : ''
          throw new Error(`HTTP ${response.status}: ${detail}${hint}`)
        }
        if (request.startsWith('/api/') && !request.startsWith('/api/uploads?')) {
          if (!response.headers.get('content-type')?.includes('application/json')) {
            throw new Error('Expected an API JSON response')
          }
          JSON.parse(Buffer.from(body).toString('utf8'))
        }
        if (i >= warmup) samples.push(elapsed)
      }
      const sorted = [...samples].sort((a, b) => a - b)
      const middle = Math.floor(sorted.length / 2)
      const median = sorted.length % 2 ? sorted[middle] : (sorted[middle - 1] + sorted[middle]) / 2
      entry.results[request] = Number(median.toFixed(3))
      entry.samples[request] = samples.map(value => Number(value.toFixed(3)))
      log(`${entry.results[request]} ms ${request}`)
    } catch (error) {
      entry.results[request] = null
      entry.samples[request] = samples
      entry.errors[request] = `${error.name}: ${error.message}`
      log(`FAILED ${request}: ${entry.errors[request]}`)
      // A timed-out server request may still be executing. Avoid piling on.
      break
    } finally {
      saveHistory(output, history)
    }
  }
  entry.about.completed = Object.keys(entry.results).length === paths.length && !Object.keys(entry.errors).length
  entry.about.finished_at = new Date().toISOString()
  saveHistory(output, history)
  return entry
}

async function main() {
  const { Command, InvalidArgumentError } = require('commander')
  const integer = minimum => value => {
    const number = Number(value)
    if (!Number.isSafeInteger(number) || number < minimum) throw new InvalidArgumentError(`Expected an integer >= ${minimum}`)
    return number
  }
  const program = new Command()
    .description('Benchmark local read APIs and listing pages; append median milliseconds and metadata to a JSON history.')
    .option('--url <url>', 'server origin', 'http://localhost:3000')
    .option('-o, --output <path>', 'benchmark history JSON', defaultOutput)
    .option('--runs <n>', 'measured requests per path', integer(1), 3)
    .option('--warmup <n>', 'unmeasured requests per path', integer(0), 1)
    .option('--timeout <ms>', 'request timeout in milliseconds', integer(1), 60000)
    .option('--limit <n>', 'API list page size', integer(1), 20)
    .option('--author <username>', 'author to benchmark (default: largest author)')
    .option('--api-only', 'skip HTML listing pages')
    .option('--explain [directory]', 'automatically collect traces when enabled on server (default directory: tmp/benchmark-explain)')
    .option('--no-explain', 'disable automatic query-plan collection')
    .argument('[paths...]', 'paths to benchmark instead of the automatically generated set')
    .parse()
  const opts = program.opts()
  const requestedPaths = normalizePaths(program.args)
  const url = new URL(opts.url)
  if (!['http:', 'https:'].includes(url.protocol) || url.username || url.password || url.pathname !== '/' || url.search || url.hash) {
    throw new Error('--url must be an HTTP(S) origin without credentials, a path, or query parameters')
  }
  // Detect before loading models: their configuration is cached on import.
  const dialect = await detectDatabaseDialect(url.origin, opts.timeout)
  process.env.OURBIGBOOK_POSTGRES = dialect === 'postgres' ? '1' : '0'
  console.log(`Detected server database: ${dialect}`)
  const models = require('../models')
  const sequelize = models.getSequelize(path.resolve(__dirname, '..'))
  let db
  try {
    db = await collectDatabase(sequelize, opts)
  } finally {
    await sequelize.close()
  }
  const git = args => execFileSync('git', ['-C', repoRoot, ...args], { encoding: 'utf8' }).trim()
  const cpus = os.cpus()
  const entry = await benchmark({
    ...opts, baseUrl: url.origin, output: path.resolve(opts.output),
    paths: requestedPaths.length ? requestedPaths : buildPaths(db, { limit: opts.limit, pages: !opts.apiOnly }),
    about: {
      git_sha: git(['rev-parse', 'HEAD']),
      git_dirty: git(['status', '--porcelain']) !== '',
      api_only: !!opts.apiOnly,
      limit: opts.limit,
      db,
      system: {
        pg_version: db.dialect === 'postgres' ? db.version : null,
        hostname: os.hostname(), platform: os.platform(), release: os.release(),
        os_version: osVersion(),
        arch: os.arch(), node_version: process.version,
        cpu_model: cpus[0]?.model, logical_cpus: cpus.length,
        memory_bytes: os.totalmem(),
      },
    },
  })
  console.log(`Saved ${Object.keys(entry.results).length} paths to ${path.resolve(opts.output)}`)
  if (!entry.about.completed) process.exitCode = 1
}

module.exports = { benchmark, buildPaths, collectDatabase, detectDatabaseDialect, normalizePaths }
if (require.main === module) main().catch(error => {
  console.error(error.message)
  process.exitCode = 1
})