Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
verify.js259 linesDownload Raw Back to lib
1'use strict'
2
3const {
4  mkdir,
5  readFile,
6  rm,
7  stat,
8  truncate,
9  writeFile,
10} = require('fs/promises')
11const contentPath = require('./content/path')
12const fsm = require('fs-minipass')
13const glob = require('./util/glob.js')
14const index = require('./entry-index')
15const path = require('path')
16const ssri = require('ssri')
17
18const hasOwnProperty = (obj, key) =>
19  Object.prototype.hasOwnProperty.call(obj, key)
20
21const verifyOpts = (opts) => ({
22  concurrency: 20,
23  log: { silly () {} },
24  ...opts,
25})
26
27module.exports = verify
28
29async function verify (cache, opts) {
30  opts = verifyOpts(opts)
31  opts.log.silly('verify', 'verifying cache at', cache)
32
33  const steps = [
34    markStartTime,
35    fixPerms,
36    garbageCollect,
37    rebuildIndex,
38    cleanTmp,
39    writeVerifile,
40    markEndTime,
41  ]
42
43  const stats = {}
44  for (const step of steps) {
45    const label = step.name
46    const start = new Date()
47    const s = await step(cache, opts)
48    if (s) {
49      Object.keys(s).forEach((k) => {
50        stats[k] = s[k]
51      })
52    }
53    const end = new Date()
54    if (!stats.runTime) {
55      stats.runTime = {}
56    }
57    stats.runTime[label] = end - start
58  }
59  stats.runTime.total = stats.endTime - stats.startTime
60  opts.log.silly(
61    'verify',
62    'verification finished for',
63    cache,
64    'in',
65    `${stats.runTime.total}ms`
66  )
67  return stats
68}
69
70async function markStartTime () {
71  return { startTime: new Date() }
72}
73
74async function markEndTime () {
75  return { endTime: new Date() }
76}
77
78async function fixPerms (cache, opts) {
79  opts.log.silly('verify', 'fixing cache permissions')
80  await mkdir(cache, { recursive: true })
81  return null
82}
83
84// Implements a naive mark-and-sweep tracing garbage collector.
85//
86// The algorithm is basically as follows:
87// 1. Read (and filter) all index entries ("pointers")
88// 2. Mark each integrity value as "live"
89// 3. Read entire filesystem tree in `content-vX/` dir
90// 4. If content is live, verify its checksum and delete it if it fails
91// 5. If content is not marked as live, rm it.
92//
93async function garbageCollect (cache, opts) {
94  opts.log.silly('verify', 'garbage collecting content')
95  const { default: pMap } = await import('p-map')
96  const indexStream = index.lsStream(cache)
97  const liveContent = new Set()
98  indexStream.on('data', (entry) => {
99    if (opts.filter && !opts.filter(entry)) {
100      return
101    }
102
103    // integrity is stringified, re-parse it so we can get each hash
104    const integrity = ssri.parse(entry.integrity)
105    for (const algo in integrity) {
106      liveContent.add(integrity[algo].toString())
107    }
108  })
109  await new Promise((resolve, reject) => {
110    indexStream.on('end', resolve).on('error', reject)
111  })
112  const contentDir = contentPath.contentDir(cache)
113  const files = await glob(path.join(contentDir, '**'), {
114    follow: false,
115    nodir: true,
116    nosort: true,
117  })
118  const stats = {
119    verifiedContent: 0,
120    reclaimedCount: 0,
121    reclaimedSize: 0,
122    badContentCount: 0,
123    keptSize: 0,
124  }
125  await pMap(
126    files,
127    async (f) => {
128      const split = f.split(/[/\\]/)
129      const digest = split.slice(split.length - 3).join('')
130      const algo = split[split.length - 4]
131      const integrity = ssri.fromHex(digest, algo)
132      if (liveContent.has(integrity.toString())) {
133        const info = await verifyContent(f, integrity)
134        if (!info.valid) {
135          stats.reclaimedCount++
136          stats.badContentCount++
137          stats.reclaimedSize += info.size
138        } else {
139          stats.verifiedContent++
140          stats.keptSize += info.size
141        }
142      } else {
143        // No entries refer to this content. We can delete.
144        stats.reclaimedCount++
145        const s = await stat(f)
146        await rm(f, { recursive: true, force: true })
147        stats.reclaimedSize += s.size
148      }
149      return stats
150    },
151    { concurrency: opts.concurrency }
152  )
153  return stats
154}
155
156async function verifyContent (filepath, sri) {
157  const contentInfo = {}
158  try {
159    const { size } = await stat(filepath)
160    contentInfo.size = size
161    contentInfo.valid = true
162    await ssri.checkStream(new fsm.ReadStream(filepath), sri)
163  } catch (err) {
164    if (err.code === 'ENOENT') {
165      return { size: 0, valid: false }
166    }
167    if (err.code !== 'EINTEGRITY') {
168      throw err
169    }
170
171    await rm(filepath, { recursive: true, force: true })
172    contentInfo.valid = false
173  }
174  return contentInfo
175}
176
177async function rebuildIndex (cache, opts) {
178  opts.log.silly('verify', 'rebuilding index')
179  const { default: pMap } = await import('p-map')
180  const entries = await index.ls(cache)
181  const stats = {
182    missingContent: 0,
183    rejectedEntries: 0,
184    totalEntries: 0,
185  }
186  const buckets = {}
187  for (const k in entries) {
188    /* istanbul ignore else */
189    if (hasOwnProperty(entries, k)) {
190      const hashed = index.hashKey(k)
191      const entry = entries[k]
192      const excluded = opts.filter && !opts.filter(entry)
193      excluded && stats.rejectedEntries++
194      if (buckets[hashed] && !excluded) {
195        buckets[hashed].push(entry)
196      } else if (buckets[hashed] && excluded) {
197        // skip
198      } else if (excluded) {
199        buckets[hashed] = []
200        buckets[hashed]._path = index.bucketPath(cache, k)
201      } else {
202        buckets[hashed] = [entry]
203        buckets[hashed]._path = index.bucketPath(cache, k)
204      }
205    }
206  }
207  await pMap(
208    Object.keys(buckets),
209    (key) => {
210      return rebuildBucket(cache, buckets[key], stats, opts)
211    },
212    { concurrency: opts.concurrency }
213  )
214  return stats
215}
216
217async function rebuildBucket (cache, bucket, stats) {
218  await truncate(bucket._path)
219  // This needs to be serialized because cacache explicitly
220  // lets very racy bucket conflicts clobber each other.
221  for (const entry of bucket) {
222    const content = contentPath(cache, entry.integrity)
223    try {
224      await stat(content)
225      await index.insert(cache, entry.key, entry.integrity, {
226        metadata: entry.metadata,
227        size: entry.size,
228        time: entry.time,
229      })
230      stats.totalEntries++
231    } catch (err) {
232      if (err.code === 'ENOENT') {
233        stats.rejectedEntries++
234        stats.missingContent++
235      } else {
236        throw err
237      }
238    }
239  }
240}
241
242function cleanTmp (cache, opts) {
243  opts.log.silly('verify', 'cleaning tmp directory')
244  return rm(path.join(cache, 'tmp'), { recursive: true, force: true })
245}
246
247async function writeVerifile (cache, opts) {
248  const verifile = path.join(cache, '_lastverified')
249  opts.log.silly('verify', 'writing verifile to ' + verifile)
250  return writeFile(verifile, `${Date.now()}`)
251}
252
253module.exports.lastRun = lastRun
254
255async function lastRun (cache) {
256  const data = await readFile(path.join(cache, '_lastverified'), { encoding: 'utf8' })
257  return new Date(+data)
258}
259 
codekingpro/portable-devtools · Team Ai