Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Binary file modifiedsrc/compile_cache_zstd.dict
Binary file not shown.
168 changes: 168 additions & 0 deletions tools/train_compile_cache_dict.mjs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,168 @@
#!/usr/bin/env node
// =============================================================================
// train_compile_cache_dict.mjs
//
// Maintainer tool that regenerates src/compile_cache_zstd.dict, the zstd
// dictionary embedded in the node binary and used to shrink compile-cache
// entries on disk (see src/compile_cache.cc).
//
// -----------------------------------------------------------------------------
// What it does
// -----------------------------------------------------------------------------
// The node compile cache stores V8 code caches (the bytecode/metadata blob V8
// produces when it compiles a module) so later runs can skip recompilation.
// These blobs share a lot of structure across files, so we zstd-compress them
// with a trained dictionary to cut the cache's on-disk footprint.
//
// This script builds that dictionary end to end:
//
// 1. Walks a fixed in-tree corpus (CORPUS below) in sorted order.
// 2. For each .js file, harvests a V8 code cache via vm.compileFunction with
// produceCachedData — the same shape the CommonJS loader produces at
// runtime — and writes each blob to a temp directory.
// 3. Feeds all the harvested blobs to `zstd --train`, capping the result at
// MAXDICT (16 KiB) bytes.
// 4. Overwrites src/compile_cache_zstd.dict with the trained dictionary.
//
// The shipped src/compile_cache_zstd.dict is the source of truth; this script
// exists to document and reproduce exactly how that file was made, so a future
// maintainer can regenerate it (e.g. after a V8/corpus change) and review the
// resulting diff.
//
// -----------------------------------------------------------------------------
// Usage
// -----------------------------------------------------------------------------
// node tools/train_compile_cache_dict.mjs
//
// Run from anywhere (paths are resolved relative to this file). The output is
// written in place to src/compile_cache_zstd.dict; inspect the diff afterwards
// and commit it if it looks right. Progress is printed to stderr.
//
// Prerequisites:
// * the `zstd` CLI on PATH, version REQUIRED_ZSTD (matching deps/zstd).
// * a built node on PATH (this is the node you invoke it with).
//
// Note: --predictable is added automatically (see below) — you do not need to
// pass it yourself.
//
// -----------------------------------------------------------------------------
// Reproducibility
// -----------------------------------------------------------------------------
// The output is byte-for-byte stable only if all of the following are pinned:
//
// * node is run with --predictable. V8 otherwise randomizes the string hash
// seed per process, and that seed leaks into vm.compileFunction cachedData,
// so every harvest (and therefore every trained dictionary) would differ
// run-to-run even on the same machine. This script re-executes itself with
// --predictable if needed.
// * the corpus and its order are fixed (CORPUS below, walked sorted).
// * the node build is fixed: cachedData embeds the V8 version and build
// flags, so a different node produces a different (still valid) dictionary.
// * the zstd CLI matches deps/zstd (currently 1.5.7). ZDICT training defaults
// change between zstd releases; this script refuses to run on a mismatch.
//
// Training on --predictable caches is fine even though the runtime consumes
// non-predictable caches: the dictionary only supplies shared substrings, and
// Persist() keeps min(plain, dict) per entry, so a less-than-ideal dictionary
// can never make any entry larger.
// =============================================================================

import { spawnSync } from 'node:child_process';
import { readFileSync, writeFileSync, mkdtempSync, readdirSync,
rmSync } from 'node:fs';
import { join, relative, dirname } from 'node:path';
import { tmpdir } from 'node:os';
import { fileURLToPath } from 'node:url';

const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..');
const OUT = join(ROOT, 'src', 'compile_cache_zstd.dict');
const MAXDICT = 16384;
const REQUIRED_ZSTD = '1.5.7'; // keep in sync with deps/zstd/lib/zstd.h

// Fixed corpus, relative to the repo root. Chosen to be diverse, always
// present in a checkout, and disjoint from the held-out measurement corpora
// (e.g. test/parallel) used to report size numbers.
const CORPUS = ['lib', 'tools', 'deps/npm/node_modules'];

// V8 randomizes the hash seed per process; re-exec under --predictable so the
// harvested caches — and the trained dictionary — are deterministic.
if (!process.execArgv.includes('--predictable')) {
const r = spawnSync(process.execPath,
['--predictable', fileURLToPath(import.meta.url), ...process.argv.slice(2)],
{ stdio: 'inherit' });
process.exit(r.status ?? 1);
}

function checkZstd() {
const r = spawnSync('zstd', ['--version'], { encoding: 'utf8' });
if (r.status !== 0) {
console.error('error: zstd CLI not found on PATH'); process.exit(1);
}
const m = r.stdout.match(/v(\d+\.\d+\.\d+)/);
if (!m || m[1] !== REQUIRED_ZSTD) {
console.error(`error: zstd ${REQUIRED_ZSTD} required (matching deps/zstd), ` +
`found ${m ? m[1] : 'unknown'}`);
process.exit(1);
}
}

const PARAMS = ['exports', 'require', 'module', '__filename', '__dirname'];

function* walk(dir) {
let ents;
try { ents = readdirSync(dir, { withFileTypes: true }); } catch { return; }
for (const e of ents.sort((a, b) => (a.name < b.name ? -1 : 1))) {
const p = join(dir, e.name);
if (e.isDirectory()) yield* walk(p);
else if (e.isFile() && p.endsWith('.js')) yield p;
}
}

async function main() {
checkZstd();
const { default: vm } = await import('node:vm');

const files = [];
for (const root of CORPUS) for (const f of walk(join(ROOT, root))) files.push(f);
files.sort();

const samples = mkdtempSync(join(tmpdir(), 'cc-dict-'));
const sampleFiles = [];
let ok = 0;
let r;
try {
for (const f of files) {
let code;
try { code = readFileSync(f, 'utf8'); } catch { continue; }
try {
const fn = vm.compileFunction(code, PARAMS,
{ filename: f, produceCachedData: true });
const cd = fn.cachedData;
if (cd && cd.length > 0) {
const name = relative(ROOT, f).replace(/[\\/]/g, '_') + '.cache';
const out = join(samples, name);
writeFileSync(out, cd);
sampleFiles.push(out);
ok++;
}
} catch { /* skip modules V8 can't compile standalone */ }
}
sampleFiles.sort();
console.error(`harvested ${ok}/${files.length} code caches from ` +
`${CORPUS.join(', ')}`);

// One sample path per file is spread on argv; the fixed corpus (~1.4k
// files, a few hundred KB of paths) stays well under ARG_MAX.
r = spawnSync('zstd',
['--train', ...sampleFiles, `--maxdict=${MAXDICT}`, '-f', '-o', OUT],
{ stdio: ['ignore', 'ignore', 'inherit'] });
} finally {
rmSync(samples, { recursive: true, force: true });
}
if (r.status !== 0) { console.error('error: zstd --train failed'); process.exit(1); }

const size = readFileSync(OUT).length;
console.error(`wrote ${relative(ROOT, OUT)} (${size} bytes)`);
}

main().catch((e) => { console.error(e); process.exit(1); });
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Binary file modifiedsrc/compile_cache_zstd.dict
Binary file not shown.
168 changes: 168 additions & 0 deletions tools/train_compile_cache_dict.mjs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,168 @@
#!/usr/bin/env node
// =============================================================================
// train_compile_cache_dict.mjs
//
// Maintainer tool that regenerates src/compile_cache_zstd.dict, the zstd
// dictionary embedded in the node binary and used to shrink compile-cache
// entries on disk (see src/compile_cache.cc).
//
// -----------------------------------------------------------------------------
// What it does
// -----------------------------------------------------------------------------
// The node compile cache stores V8 code caches (the bytecode/metadata blob V8
// produces when it compiles a module) so later runs can skip recompilation.
// These blobs share a lot of structure across files, so we zstd-compress them
// with a trained dictionary to cut the cache's on-disk footprint.
//
// This script builds that dictionary end to end:
//
// 1. Walks a fixed in-tree corpus (CORPUS below) in sorted order.
// 2. For each .js file, harvests a V8 code cache via vm.compileFunction with
// produceCachedData — the same shape the CommonJS loader produces at
// runtime — and writes each blob to a temp directory.
// 3. Feeds all the harvested blobs to `zstd --train`, capping the result at
// MAXDICT (16 KiB) bytes.
// 4. Overwrites src/compile_cache_zstd.dict with the trained dictionary.
//
// The shipped src/compile_cache_zstd.dict is the source of truth; this script
// exists to document and reproduce exactly how that file was made, so a future
// maintainer can regenerate it (e.g. after a V8/corpus change) and review the
// resulting diff.
//
// -----------------------------------------------------------------------------
// Usage
// -----------------------------------------------------------------------------
// node tools/train_compile_cache_dict.mjs
//
// Run from anywhere (paths are resolved relative to this file). The output is
// written in place to src/compile_cache_zstd.dict; inspect the diff afterwards
// and commit it if it looks right. Progress is printed to stderr.
//
// Prerequisites:
// * the `zstd` CLI on PATH, version REQUIRED_ZSTD (matching deps/zstd).
// * a built node on PATH (this is the node you invoke it with).
//
// Note: --predictable is added automatically (see below) — you do not need to
// pass it yourself.
//
// -----------------------------------------------------------------------------
// Reproducibility
// -----------------------------------------------------------------------------
// The output is byte-for-byte stable only if all of the following are pinned:
//
// * node is run with --predictable. V8 otherwise randomizes the string hash
// seed per process, and that seed leaks into vm.compileFunction cachedData,
// so every harvest (and therefore every trained dictionary) would differ
// run-to-run even on the same machine. This script re-executes itself with
// --predictable if needed.
// * the corpus and its order are fixed (CORPUS below, walked sorted).
// * the node build is fixed: cachedData embeds the V8 version and build
// flags, so a different node produces a different (still valid) dictionary.
// * the zstd CLI matches deps/zstd (currently 1.5.7). ZDICT training defaults
// change between zstd releases; this script refuses to run on a mismatch.
//
// Training on --predictable caches is fine even though the runtime consumes
// non-predictable caches: the dictionary only supplies shared substrings, and
// Persist() keeps min(plain, dict) per entry, so a less-than-ideal dictionary
// can never make any entry larger.
// =============================================================================

import { spawnSync } from 'node:child_process';
import { readFileSync, writeFileSync, mkdtempSync, readdirSync,
rmSync } from 'node:fs';
import { join, relative, dirname } from 'node:path';
import { tmpdir } from 'node:os';
import { fileURLToPath } from 'node:url';

const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..');
const OUT = join(ROOT, 'src', 'compile_cache_zstd.dict');
const MAXDICT = 16384;
const REQUIRED_ZSTD = '1.5.7'; // keep in sync with deps/zstd/lib/zstd.h

// Fixed corpus, relative to the repo root. Chosen to be diverse, always
// present in a checkout, and disjoint from the held-out measurement corpora
// (e.g. test/parallel) used to report size numbers.
const CORPUS = ['lib', 'tools', 'deps/npm/node_modules'];

// V8 randomizes the hash seed per process; re-exec under --predictable so the
// harvested caches — and the trained dictionary — are deterministic.
if (!process.execArgv.includes('--predictable')) {
const r = spawnSync(process.execPath,
['--predictable', fileURLToPath(import.meta.url), ...process.argv.slice(2)],
{ stdio: 'inherit' });
process.exit(r.status ?? 1);
}

function checkZstd() {
const r = spawnSync('zstd', ['--version'], { encoding: 'utf8' });
if (r.status !== 0) {
console.error('error: zstd CLI not found on PATH'); process.exit(1);
}
const m = r.stdout.match(/v(\d+\.\d+\.\d+)/);
if (!m || m[1] !== REQUIRED_ZSTD) {
console.error(`error: zstd ${REQUIRED_ZSTD} required (matching deps/zstd), ` +
`found ${m ? m[1] : 'unknown'}`);
process.exit(1);
}
}

const PARAMS = ['exports', 'require', 'module', '__filename', '__dirname'];

function* walk(dir) {
let ents;
try { ents = readdirSync(dir, { withFileTypes: true }); } catch { return; }
for (const e of ents.sort((a, b) => (a.name < b.name ? -1 : 1))) {
const p = join(dir, e.name);
if (e.isDirectory()) yield* walk(p);
else if (e.isFile() && p.endsWith('.js')) yield p;
}
}

async function main() {
checkZstd();
const { default: vm } = await import('node:vm');

const files = [];
for (const root of CORPUS) for (const f of walk(join(ROOT, root))) files.push(f);
files.sort();

const samples = mkdtempSync(join(tmpdir(), 'cc-dict-'));
const sampleFiles = [];
let ok = 0;
let r;
try {
for (const f of files) {
let code;
try { code = readFileSync(f, 'utf8'); } catch { continue; }
try {
const fn = vm.compileFunction(code, PARAMS,
{ filename: f, produceCachedData: true });
const cd = fn.cachedData;
if (cd && cd.length > 0) {
const name = relative(ROOT, f).replace(/[\\/]/g, '_') + '.cache';
const out = join(samples, name);
writeFileSync(out, cd);
sampleFiles.push(out);
ok++;
}
} catch { /* skip modules V8 can't compile standalone */ }
}
sampleFiles.sort();
console.error(`harvested ${ok}/${files.length} code caches from ` +
`${CORPUS.join(', ')}`);

// One sample path per file is spread on argv; the fixed corpus (~1.4k
// files, a few hundred KB of paths) stays well under ARG_MAX.
r = spawnSync('zstd',
['--train', ...sampleFiles, `--maxdict=${MAXDICT}`, '-f', '-o', OUT],
{ stdio: ['ignore', 'ignore', 'inherit'] });
} finally {
rmSync(samples, { recursive: true, force: true });
}
if (r.status !== 0) { console.error('error: zstd --train failed'); process.exit(1); }

const size = readFileSync(OUT).length;
console.error(`wrote ${relative(ROOT, OUT)} (${size} bytes)`);
}

main().catch((e) => { console.error(e); process.exit(1); });
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Binary file modifiedsrc/compile_cache_zstd.dict
Binary file not shown.
168 changes: 168 additions & 0 deletions tools/train_compile_cache_dict.mjs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,168 @@
#!/usr/bin/env node
// =============================================================================
// train_compile_cache_dict.mjs
//
// Maintainer tool that regenerates src/compile_cache_zstd.dict, the zstd
// dictionary embedded in the node binary and used to shrink compile-cache
// entries on disk (see src/compile_cache.cc).
//
// -----------------------------------------------------------------------------
// What it does
// -----------------------------------------------------------------------------
// The node compile cache stores V8 code caches (the bytecode/metadata blob V8
// produces when it compiles a module) so later runs can skip recompilation.
// These blobs share a lot of structure across files, so we zstd-compress them
// with a trained dictionary to cut the cache's on-disk footprint.
//
// This script builds that dictionary end to end:
//
// 1. Walks a fixed in-tree corpus (CORPUS below) in sorted order.
// 2. For each .js file, harvests a V8 code cache via vm.compileFunction with
// produceCachedData — the same shape the CommonJS loader produces at
// runtime — and writes each blob to a temp directory.
// 3. Feeds all the harvested blobs to `zstd --train`, capping the result at
// MAXDICT (16 KiB) bytes.
// 4. Overwrites src/compile_cache_zstd.dict with the trained dictionary.
//
// The shipped src/compile_cache_zstd.dict is the source of truth; this script
// exists to document and reproduce exactly how that file was made, so a future
// maintainer can regenerate it (e.g. after a V8/corpus change) and review the
// resulting diff.
//
// -----------------------------------------------------------------------------
// Usage
// -----------------------------------------------------------------------------
// node tools/train_compile_cache_dict.mjs
//
// Run from anywhere (paths are resolved relative to this file). The output is
// written in place to src/compile_cache_zstd.dict; inspect the diff afterwards
// and commit it if it looks right. Progress is printed to stderr.
//
// Prerequisites:
// * the `zstd` CLI on PATH, version REQUIRED_ZSTD (matching deps/zstd).
// * a built node on PATH (this is the node you invoke it with).
//
// Note: --predictable is added automatically (see below) — you do not need to
// pass it yourself.
//
// -----------------------------------------------------------------------------
// Reproducibility
// -----------------------------------------------------------------------------
// The output is byte-for-byte stable only if all of the following are pinned:
//
// * node is run with --predictable. V8 otherwise randomizes the string hash
// seed per process, and that seed leaks into vm.compileFunction cachedData,
// so every harvest (and therefore every trained dictionary) would differ
// run-to-run even on the same machine. This script re-executes itself with
// --predictable if needed.
// * the corpus and its order are fixed (CORPUS below, walked sorted).
// * the node build is fixed: cachedData embeds the V8 version and build
// flags, so a different node produces a different (still valid) dictionary.
// * the zstd CLI matches deps/zstd (currently 1.5.7). ZDICT training defaults
// change between zstd releases; this script refuses to run on a mismatch.
//
// Training on --predictable caches is fine even though the runtime consumes
// non-predictable caches: the dictionary only supplies shared substrings, and
// Persist() keeps min(plain, dict) per entry, so a less-than-ideal dictionary
// can never make any entry larger.
// =============================================================================

import { spawnSync } from 'node:child_process';
import { readFileSync, writeFileSync, mkdtempSync, readdirSync,
rmSync } from 'node:fs';
import { join, relative, dirname } from 'node:path';
import { tmpdir } from 'node:os';
import { fileURLToPath } from 'node:url';

const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..');
const OUT = join(ROOT, 'src', 'compile_cache_zstd.dict');
const MAXDICT = 16384;
const REQUIRED_ZSTD = '1.5.7'; // keep in sync with deps/zstd/lib/zstd.h

// Fixed corpus, relative to the repo root. Chosen to be diverse, always
// present in a checkout, and disjoint from the held-out measurement corpora
// (e.g. test/parallel) used to report size numbers.
const CORPUS = ['lib', 'tools', 'deps/npm/node_modules'];

// V8 randomizes the hash seed per process; re-exec under --predictable so the
// harvested caches — and the trained dictionary — are deterministic.
if (!process.execArgv.includes('--predictable')) {
const r = spawnSync(process.execPath,
['--predictable', fileURLToPath(import.meta.url), ...process.argv.slice(2)],
{ stdio: 'inherit' });
process.exit(r.status ?? 1);
}

function checkZstd() {
const r = spawnSync('zstd', ['--version'], { encoding: 'utf8' });
if (r.status !== 0) {
console.error('error: zstd CLI not found on PATH'); process.exit(1);
}
const m = r.stdout.match(/v(\d+\.\d+\.\d+)/);
if (!m || m[1] !== REQUIRED_ZSTD) {
console.error(`error: zstd ${REQUIRED_ZSTD} required (matching deps/zstd), ` +
`found ${m ? m[1] : 'unknown'}`);
process.exit(1);
}
}

const PARAMS = ['exports', 'require', 'module', '__filename', '__dirname'];

function* walk(dir) {
let ents;
try { ents = readdirSync(dir, { withFileTypes: true }); } catch { return; }
for (const e of ents.sort((a, b) => (a.name < b.name ? -1 : 1))) {
const p = join(dir, e.name);
if (e.isDirectory()) yield* walk(p);
else if (e.isFile() && p.endsWith('.js')) yield p;
}
}

async function main() {
checkZstd();
const { default: vm } = await import('node:vm');

const files = [];
for (const root of CORPUS) for (const f of walk(join(ROOT, root))) files.push(f);
files.sort();

const samples = mkdtempSync(join(tmpdir(), 'cc-dict-'));
const sampleFiles = [];
let ok = 0;
let r;
try {
for (const f of files) {
let code;
try { code = readFileSync(f, 'utf8'); } catch { continue; }
try {
const fn = vm.compileFunction(code, PARAMS,
{ filename: f, produceCachedData: true });
const cd = fn.cachedData;
if (cd && cd.length > 0) {
const name = relative(ROOT, f).replace(/[\\/]/g, '_') + '.cache';
const out = join(samples, name);
writeFileSync(out, cd);
sampleFiles.push(out);
ok++;
}
} catch { /* skip modules V8 can't compile standalone */ }
}
sampleFiles.sort();
console.error(`harvested ${ok}/${files.length} code caches from ` +
`${CORPUS.join(', ')}`);

// One sample path per file is spread on argv; the fixed corpus (~1.4k
// files, a few hundred KB of paths) stays well under ARG_MAX.
r = spawnSync('zstd',
['--train', ...sampleFiles, `--maxdict=${MAXDICT}`, '-f', '-o', OUT],
{ stdio: ['ignore', 'ignore', 'inherit'] });
} finally {
rmSync(samples, { recursive: true, force: true });
}
if (r.status !== 0) { console.error('error: zstd --train failed'); process.exit(1); }

const size = readFileSync(OUT).length;
console.error(`wrote ${relative(ROOT, OUT)} (${size} bytes)`);
}

main().catch((e) => { console.error(e); process.exit(1); });
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Binary file modifiedsrc/compile_cache_zstd.dict
Binary file not shown.
168 changes: 168 additions & 0 deletions tools/train_compile_cache_dict.mjs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,168 @@
#!/usr/bin/env node
// =============================================================================
// train_compile_cache_dict.mjs
//
// Maintainer tool that regenerates src/compile_cache_zstd.dict, the zstd
// dictionary embedded in the node binary and used to shrink compile-cache
// entries on disk (see src/compile_cache.cc).
//
// -----------------------------------------------------------------------------
// What it does
// -----------------------------------------------------------------------------
// The node compile cache stores V8 code caches (the bytecode/metadata blob V8
// produces when it compiles a module) so later runs can skip recompilation.
// These blobs share a lot of structure across files, so we zstd-compress them
// with a trained dictionary to cut the cache's on-disk footprint.
//
// This script builds that dictionary end to end:
//
// 1. Walks a fixed in-tree corpus (CORPUS below) in sorted order.
// 2. For each .js file, harvests a V8 code cache via vm.compileFunction with
// produceCachedData — the same shape the CommonJS loader produces at
// runtime — and writes each blob to a temp directory.
// 3. Feeds all the harvested blobs to `zstd --train`, capping the result at
// MAXDICT (16 KiB) bytes.
// 4. Overwrites src/compile_cache_zstd.dict with the trained dictionary.
//
// The shipped src/compile_cache_zstd.dict is the source of truth; this script
// exists to document and reproduce exactly how that file was made, so a future
// maintainer can regenerate it (e.g. after a V8/corpus change) and review the
// resulting diff.
//
// -----------------------------------------------------------------------------
// Usage
// -----------------------------------------------------------------------------
// node tools/train_compile_cache_dict.mjs
//
// Run from anywhere (paths are resolved relative to this file). The output is
// written in place to src/compile_cache_zstd.dict; inspect the diff afterwards
// and commit it if it looks right. Progress is printed to stderr.
//
// Prerequisites:
// * the `zstd` CLI on PATH, version REQUIRED_ZSTD (matching deps/zstd).
// * a built node on PATH (this is the node you invoke it with).
//
// Note: --predictable is added automatically (see below) — you do not need to
// pass it yourself.
//
// -----------------------------------------------------------------------------
// Reproducibility
// -----------------------------------------------------------------------------
// The output is byte-for-byte stable only if all of the following are pinned:
//
// * node is run with --predictable. V8 otherwise randomizes the string hash
// seed per process, and that seed leaks into vm.compileFunction cachedData,
// so every harvest (and therefore every trained dictionary) would differ
// run-to-run even on the same machine. This script re-executes itself with
// --predictable if needed.
// * the corpus and its order are fixed (CORPUS below, walked sorted).
// * the node build is fixed: cachedData embeds the V8 version and build
// flags, so a different node produces a different (still valid) dictionary.
// * the zstd CLI matches deps/zstd (currently 1.5.7). ZDICT training defaults
// change between zstd releases; this script refuses to run on a mismatch.
//
// Training on --predictable caches is fine even though the runtime consumes
// non-predictable caches: the dictionary only supplies shared substrings, and
// Persist() keeps min(plain, dict) per entry, so a less-than-ideal dictionary
// can never make any entry larger.
// =============================================================================

import { spawnSync } from 'node:child_process';
import { readFileSync, writeFileSync, mkdtempSync, readdirSync,
rmSync } from 'node:fs';
import { join, relative, dirname } from 'node:path';
import { tmpdir } from 'node:os';
import { fileURLToPath } from 'node:url';

const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..');
const OUT = join(ROOT, 'src', 'compile_cache_zstd.dict');
const MAXDICT = 16384;
const REQUIRED_ZSTD = '1.5.7'; // keep in sync with deps/zstd/lib/zstd.h

// Fixed corpus, relative to the repo root. Chosen to be diverse, always
// present in a checkout, and disjoint from the held-out measurement corpora
// (e.g. test/parallel) used to report size numbers.
const CORPUS = ['lib', 'tools', 'deps/npm/node_modules'];

// V8 randomizes the hash seed per process; re-exec under --predictable so the
// harvested caches — and the trained dictionary — are deterministic.
if (!process.execArgv.includes('--predictable')) {
const r = spawnSync(process.execPath,
['--predictable', fileURLToPath(import.meta.url), ...process.argv.slice(2)],
{ stdio: 'inherit' });
process.exit(r.status ?? 1);
}

function checkZstd() {
const r = spawnSync('zstd', ['--version'], { encoding: 'utf8' });
if (r.status !== 0) {
console.error('error: zstd CLI not found on PATH'); process.exit(1);
}
const m = r.stdout.match(/v(\d+\.\d+\.\d+)/);
if (!m || m[1] !== REQUIRED_ZSTD) {
console.error(`error: zstd ${REQUIRED_ZSTD} required (matching deps/zstd), ` +
`found ${m ? m[1] : 'unknown'}`);
process.exit(1);
}
}

const PARAMS = ['exports', 'require', 'module', '__filename', '__dirname'];

function* walk(dir) {
let ents;
try { ents = readdirSync(dir, { withFileTypes: true }); } catch { return; }
for (const e of ents.sort((a, b) => (a.name < b.name ? -1 : 1))) {
const p = join(dir, e.name);
if (e.isDirectory()) yield* walk(p);
else if (e.isFile() && p.endsWith('.js')) yield p;
}
}

async function main() {
checkZstd();
const { default: vm } = await import('node:vm');

const files = [];
for (const root of CORPUS) for (const f of walk(join(ROOT, root))) files.push(f);
files.sort();

const samples = mkdtempSync(join(tmpdir(), 'cc-dict-'));
const sampleFiles = [];
let ok = 0;
let r;
try {
for (const f of files) {
let code;
try { code = readFileSync(f, 'utf8'); } catch { continue; }
try {
const fn = vm.compileFunction(code, PARAMS,
{ filename: f, produceCachedData: true });
const cd = fn.cachedData;
if (cd && cd.length > 0) {
const name = relative(ROOT, f).replace(/[\\/]/g, '_') + '.cache';
const out = join(samples, name);
writeFileSync(out, cd);
sampleFiles.push(out);
ok++;
}
} catch { /* skip modules V8 can't compile standalone */ }
}
sampleFiles.sort();
console.error(`harvested ${ok}/${files.length} code caches from ` +
`${CORPUS.join(', ')}`);

// One sample path per file is spread on argv; the fixed corpus (~1.4k
// files, a few hundred KB of paths) stays well under ARG_MAX.
r = spawnSync('zstd',
['--train', ...sampleFiles, `--maxdict=${MAXDICT}`, '-f', '-o', OUT],
{ stdio: ['ignore', 'ignore', 'inherit'] });
} finally {
rmSync(samples, { recursive: true, force: true });
}
if (r.status !== 0) { console.error('error: zstd --train failed'); process.exit(1); }

const size = readFileSync(OUT).length;
console.error(`wrote ${relative(ROOT, OUT)} (${size} bytes)`);
}

main().catch((e) => { console.error(e); process.exit(1); });
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Binary file modifiedsrc/compile_cache_zstd.dict
Binary file not shown.
168 changes: 168 additions & 0 deletions tools/train_compile_cache_dict.mjs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,168 @@
#!/usr/bin/env node
// =============================================================================
// train_compile_cache_dict.mjs
//
// Maintainer tool that regenerates src/compile_cache_zstd.dict, the zstd
// dictionary embedded in the node binary and used to shrink compile-cache
// entries on disk (see src/compile_cache.cc).
//
// -----------------------------------------------------------------------------
// What it does
// -----------------------------------------------------------------------------
// The node compile cache stores V8 code caches (the bytecode/metadata blob V8
// produces when it compiles a module) so later runs can skip recompilation.
// These blobs share a lot of structure across files, so we zstd-compress them
// with a trained dictionary to cut the cache's on-disk footprint.
//
// This script builds that dictionary end to end:
//
// 1. Walks a fixed in-tree corpus (CORPUS below) in sorted order.
// 2. For each .js file, harvests a V8 code cache via vm.compileFunction with
// produceCachedData — the same shape the CommonJS loader produces at
// runtime — and writes each blob to a temp directory.
// 3. Feeds all the harvested blobs to `zstd --train`, capping the result at
// MAXDICT (16 KiB) bytes.
// 4. Overwrites src/compile_cache_zstd.dict with the trained dictionary.
//
// The shipped src/compile_cache_zstd.dict is the source of truth; this script
// exists to document and reproduce exactly how that file was made, so a future
// maintainer can regenerate it (e.g. after a V8/corpus change) and review the
// resulting diff.
//
// -----------------------------------------------------------------------------
// Usage
// -----------------------------------------------------------------------------
// node tools/train_compile_cache_dict.mjs
//
// Run from anywhere (paths are resolved relative to this file). The output is
// written in place to src/compile_cache_zstd.dict; inspect the diff afterwards
// and commit it if it looks right. Progress is printed to stderr.
//
// Prerequisites:
// * the `zstd` CLI on PATH, version REQUIRED_ZSTD (matching deps/zstd).
// * a built node on PATH (this is the node you invoke it with).
//
// Note: --predictable is added automatically (see below) — you do not need to
// pass it yourself.
//
// -----------------------------------------------------------------------------
// Reproducibility
// -----------------------------------------------------------------------------
// The output is byte-for-byte stable only if all of the following are pinned:
//
// * node is run with --predictable. V8 otherwise randomizes the string hash
// seed per process, and that seed leaks into vm.compileFunction cachedData,
// so every harvest (and therefore every trained dictionary) would differ
// run-to-run even on the same machine. This script re-executes itself with
// --predictable if needed.
// * the corpus and its order are fixed (CORPUS below, walked sorted).
// * the node build is fixed: cachedData embeds the V8 version and build
// flags, so a different node produces a different (still valid) dictionary.
// * the zstd CLI matches deps/zstd (currently 1.5.7). ZDICT training defaults
// change between zstd releases; this script refuses to run on a mismatch.
//
// Training on --predictable caches is fine even though the runtime consumes
// non-predictable caches: the dictionary only supplies shared substrings, and
// Persist() keeps min(plain, dict) per entry, so a less-than-ideal dictionary
// can never make any entry larger.
// =============================================================================

import { spawnSync } from 'node:child_process';
import { readFileSync, writeFileSync, mkdtempSync, readdirSync,
rmSync } from 'node:fs';
import { join, relative, dirname } from 'node:path';
import { tmpdir } from 'node:os';
import { fileURLToPath } from 'node:url';

const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..');
const OUT = join(ROOT, 'src', 'compile_cache_zstd.dict');
const MAXDICT = 16384;
const REQUIRED_ZSTD = '1.5.7'; // keep in sync with deps/zstd/lib/zstd.h

// Fixed corpus, relative to the repo root. Chosen to be diverse, always
// present in a checkout, and disjoint from the held-out measurement corpora
// (e.g. test/parallel) used to report size numbers.
const CORPUS = ['lib', 'tools', 'deps/npm/node_modules'];

// V8 randomizes the hash seed per process; re-exec under --predictable so the
// harvested caches — and the trained dictionary — are deterministic.
if (!process.execArgv.includes('--predictable')) {
const r = spawnSync(process.execPath,
['--predictable', fileURLToPath(import.meta.url), ...process.argv.slice(2)],
{ stdio: 'inherit' });
process.exit(r.status ?? 1);
}

function checkZstd() {
const r = spawnSync('zstd', ['--version'], { encoding: 'utf8' });
if (r.status !== 0) {
console.error('error: zstd CLI not found on PATH'); process.exit(1);
}
const m = r.stdout.match(/v(\d+\.\d+\.\d+)/);
if (!m || m[1] !== REQUIRED_ZSTD) {
console.error(`error: zstd ${REQUIRED_ZSTD} required (matching deps/zstd), ` +
`found ${m ? m[1] : 'unknown'}`);
process.exit(1);
}
}

const PARAMS = ['exports', 'require', 'module', '__filename', '__dirname'];

function* walk(dir) {
let ents;
try { ents = readdirSync(dir, { withFileTypes: true }); } catch { return; }
for (const e of ents.sort((a, b) => (a.name < b.name ? -1 : 1))) {
const p = join(dir, e.name);
if (e.isDirectory()) yield* walk(p);
else if (e.isFile() && p.endsWith('.js')) yield p;
}
}

async function main() {
checkZstd();
const { default: vm } = await import('node:vm');

const files = [];
for (const root of CORPUS) for (const f of walk(join(ROOT, root))) files.push(f);
files.sort();

const samples = mkdtempSync(join(tmpdir(), 'cc-dict-'));
const sampleFiles = [];
let ok = 0;
let r;
try {
for (const f of files) {
let code;
try { code = readFileSync(f, 'utf8'); } catch { continue; }
try {
const fn = vm.compileFunction(code, PARAMS,
{ filename: f, produceCachedData: true });
const cd = fn.cachedData;
if (cd && cd.length > 0) {
const name = relative(ROOT, f).replace(/[\\/]/g, '_') + '.cache';
const out = join(samples, name);
writeFileSync(out, cd);
sampleFiles.push(out);
ok++;
}
} catch { /* skip modules V8 can't compile standalone */ }
}
sampleFiles.sort();
console.error(`harvested ${ok}/${files.length} code caches from ` +
`${CORPUS.join(', ')}`);

// One sample path per file is spread on argv; the fixed corpus (~1.4k
// files, a few hundred KB of paths) stays well under ARG_MAX.
r = spawnSync('zstd',
['--train', ...sampleFiles, `--maxdict=${MAXDICT}`, '-f', '-o', OUT],
{ stdio: ['ignore', 'ignore', 'inherit'] });
} finally {
rmSync(samples, { recursive: true, force: true });
}
if (r.status !== 0) { console.error('error: zstd --train failed'); process.exit(1); }

const size = readFileSync(OUT).length;
console.error(`wrote ${relative(ROOT, OUT)} (${size} bytes)`);
}

main().catch((e) => { console.error(e); process.exit(1); });
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Binary file modifiedsrc/compile_cache_zstd.dict
Binary file not shown.
168 changes: 168 additions & 0 deletions tools/train_compile_cache_dict.mjs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,168 @@
#!/usr/bin/env node
// =============================================================================
// train_compile_cache_dict.mjs
//
// Maintainer tool that regenerates src/compile_cache_zstd.dict, the zstd
// dictionary embedded in the node binary and used to shrink compile-cache
// entries on disk (see src/compile_cache.cc).
//
// -----------------------------------------------------------------------------
// What it does
// -----------------------------------------------------------------------------
// The node compile cache stores V8 code caches (the bytecode/metadata blob V8
// produces when it compiles a module) so later runs can skip recompilation.
// These blobs share a lot of structure across files, so we zstd-compress them
// with a trained dictionary to cut the cache's on-disk footprint.
//
// This script builds that dictionary end to end:
//
// 1. Walks a fixed in-tree corpus (CORPUS below) in sorted order.
// 2. For each .js file, harvests a V8 code cache via vm.compileFunction with
// produceCachedData — the same shape the CommonJS loader produces at
// runtime — and writes each blob to a temp directory.
// 3. Feeds all the harvested blobs to `zstd --train`, capping the result at
// MAXDICT (16 KiB) bytes.
// 4. Overwrites src/compile_cache_zstd.dict with the trained dictionary.
//
// The shipped src/compile_cache_zstd.dict is the source of truth; this script
// exists to document and reproduce exactly how that file was made, so a future
// maintainer can regenerate it (e.g. after a V8/corpus change) and review the
// resulting diff.
//
// -----------------------------------------------------------------------------
// Usage
// -----------------------------------------------------------------------------
// node tools/train_compile_cache_dict.mjs
//
// Run from anywhere (paths are resolved relative to this file). The output is
// written in place to src/compile_cache_zstd.dict; inspect the diff afterwards
// and commit it if it looks right. Progress is printed to stderr.
//
// Prerequisites:
// * the `zstd` CLI on PATH, version REQUIRED_ZSTD (matching deps/zstd).
// * a built node on PATH (this is the node you invoke it with).
//
// Note: --predictable is added automatically (see below) — you do not need to
// pass it yourself.
//
// -----------------------------------------------------------------------------
// Reproducibility
// -----------------------------------------------------------------------------
// The output is byte-for-byte stable only if all of the following are pinned:
//
// * node is run with --predictable. V8 otherwise randomizes the string hash
// seed per process, and that seed leaks into vm.compileFunction cachedData,
// so every harvest (and therefore every trained dictionary) would differ
// run-to-run even on the same machine. This script re-executes itself with
// --predictable if needed.
// * the corpus and its order are fixed (CORPUS below, walked sorted).
// * the node build is fixed: cachedData embeds the V8 version and build
// flags, so a different node produces a different (still valid) dictionary.
// * the zstd CLI matches deps/zstd (currently 1.5.7). ZDICT training defaults
// change between zstd releases; this script refuses to run on a mismatch.
//
// Training on --predictable caches is fine even though the runtime consumes
// non-predictable caches: the dictionary only supplies shared substrings, and
// Persist() keeps min(plain, dict) per entry, so a less-than-ideal dictionary
// can never make any entry larger.
// =============================================================================

import { spawnSync } from 'node:child_process';
import { readFileSync, writeFileSync, mkdtempSync, readdirSync,
rmSync } from 'node:fs';
import { join, relative, dirname } from 'node:path';
import { tmpdir } from 'node:os';
import { fileURLToPath } from 'node:url';

const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..');
const OUT = join(ROOT, 'src', 'compile_cache_zstd.dict');
const MAXDICT = 16384;
const REQUIRED_ZSTD = '1.5.7'; // keep in sync with deps/zstd/lib/zstd.h

// Fixed corpus, relative to the repo root. Chosen to be diverse, always
// present in a checkout, and disjoint from the held-out measurement corpora
// (e.g. test/parallel) used to report size numbers.
const CORPUS = ['lib', 'tools', 'deps/npm/node_modules'];

// V8 randomizes the hash seed per process; re-exec under --predictable so the
// harvested caches — and the trained dictionary — are deterministic.
if (!process.execArgv.includes('--predictable')) {
const r = spawnSync(process.execPath,
['--predictable', fileURLToPath(import.meta.url), ...process.argv.slice(2)],
{ stdio: 'inherit' });
process.exit(r.status ?? 1);
}

function checkZstd() {
const r = spawnSync('zstd', ['--version'], { encoding: 'utf8' });
if (r.status !== 0) {
console.error('error: zstd CLI not found on PATH'); process.exit(1);
}
const m = r.stdout.match(/v(\d+\.\d+\.\d+)/);
if (!m || m[1] !== REQUIRED_ZSTD) {
console.error(`error: zstd ${REQUIRED_ZSTD} required (matching deps/zstd), ` +
`found ${m ? m[1] : 'unknown'}`);
process.exit(1);
}
}

const PARAMS = ['exports', 'require', 'module', '__filename', '__dirname'];

function* walk(dir) {
let ents;
try { ents = readdirSync(dir, { withFileTypes: true }); } catch { return; }
for (const e of ents.sort((a, b) => (a.name < b.name ? -1 : 1))) {
const p = join(dir, e.name);
if (e.isDirectory()) yield* walk(p);
else if (e.isFile() && p.endsWith('.js')) yield p;
}
}

async function main() {
checkZstd();
const { default: vm } = await import('node:vm');

const files = [];
for (const root of CORPUS) for (const f of walk(join(ROOT, root))) files.push(f);
files.sort();

const samples = mkdtempSync(join(tmpdir(), 'cc-dict-'));
const sampleFiles = [];
let ok = 0;
let r;
try {
for (const f of files) {
let code;
try { code = readFileSync(f, 'utf8'); } catch { continue; }
try {
const fn = vm.compileFunction(code, PARAMS,
{ filename: f, produceCachedData: true });
const cd = fn.cachedData;
if (cd && cd.length > 0) {
const name = relative(ROOT, f).replace(/[\\/]/g, '_') + '.cache';
const out = join(samples, name);
writeFileSync(out, cd);
sampleFiles.push(out);
ok++;
}
} catch { /* skip modules V8 can't compile standalone */ }
}
sampleFiles.sort();
console.error(`harvested ${ok}/${files.length} code caches from ` +
`${CORPUS.join(', ')}`);

// One sample path per file is spread on argv; the fixed corpus (~1.4k
// files, a few hundred KB of paths) stays well under ARG_MAX.
r = spawnSync('zstd',
['--train', ...sampleFiles, `--maxdict=${MAXDICT}`, '-f', '-o', OUT],
{ stdio: ['ignore', 'ignore', 'inherit'] });
} finally {
rmSync(samples, { recursive: true, force: true });
}
if (r.status !== 0) { console.error('error: zstd --train failed'); process.exit(1); }

const size = readFileSync(OUT).length;
console.error(`wrote ${relative(ROOT, OUT)} (${size} bytes)`);
}

main().catch((e) => { console.error(e); process.exit(1); });
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Binary file modifiedsrc/compile_cache_zstd.dict
Binary file not shown.
168 changes: 168 additions & 0 deletions tools/train_compile_cache_dict.mjs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,168 @@
#!/usr/bin/env node
// =============================================================================
// train_compile_cache_dict.mjs
//
// Maintainer tool that regenerates src/compile_cache_zstd.dict, the zstd
// dictionary embedded in the node binary and used to shrink compile-cache
// entries on disk (see src/compile_cache.cc).
//
// -----------------------------------------------------------------------------
// What it does
// -----------------------------------------------------------------------------
// The node compile cache stores V8 code caches (the bytecode/metadata blob V8
// produces when it compiles a module) so later runs can skip recompilation.
// These blobs share a lot of structure across files, so we zstd-compress them
// with a trained dictionary to cut the cache's on-disk footprint.
//
// This script builds that dictionary end to end:
//
// 1. Walks a fixed in-tree corpus (CORPUS below) in sorted order.
// 2. For each .js file, harvests a V8 code cache via vm.compileFunction with
// produceCachedData — the same shape the CommonJS loader produces at
// runtime — and writes each blob to a temp directory.
// 3. Feeds all the harvested blobs to `zstd --train`, capping the result at
// MAXDICT (16 KiB) bytes.
// 4. Overwrites src/compile_cache_zstd.dict with the trained dictionary.
//
// The shipped src/compile_cache_zstd.dict is the source of truth; this script
// exists to document and reproduce exactly how that file was made, so a future
// maintainer can regenerate it (e.g. after a V8/corpus change) and review the
// resulting diff.
//
// -----------------------------------------------------------------------------
// Usage
// -----------------------------------------------------------------------------
// node tools/train_compile_cache_dict.mjs
//
// Run from anywhere (paths are resolved relative to this file). The output is
// written in place to src/compile_cache_zstd.dict; inspect the diff afterwards
// and commit it if it looks right. Progress is printed to stderr.
//
// Prerequisites:
// * the `zstd` CLI on PATH, version REQUIRED_ZSTD (matching deps/zstd).
// * a built node on PATH (this is the node you invoke it with).
//
// Note: --predictable is added automatically (see below) — you do not need to
// pass it yourself.
//
// -----------------------------------------------------------------------------
// Reproducibility
// -----------------------------------------------------------------------------
// The output is byte-for-byte stable only if all of the following are pinned:
//
// * node is run with --predictable. V8 otherwise randomizes the string hash
// seed per process, and that seed leaks into vm.compileFunction cachedData,
// so every harvest (and therefore every trained dictionary) would differ
// run-to-run even on the same machine. This script re-executes itself with
// --predictable if needed.
// * the corpus and its order are fixed (CORPUS below, walked sorted).
// * the node build is fixed: cachedData embeds the V8 version and build
// flags, so a different node produces a different (still valid) dictionary.
// * the zstd CLI matches deps/zstd (currently 1.5.7). ZDICT training defaults
// change between zstd releases; this script refuses to run on a mismatch.
//
// Training on --predictable caches is fine even though the runtime consumes
// non-predictable caches: the dictionary only supplies shared substrings, and
// Persist() keeps min(plain, dict) per entry, so a less-than-ideal dictionary
// can never make any entry larger.
// =============================================================================

import { spawnSync } from 'node:child_process';
import { readFileSync, writeFileSync, mkdtempSync, readdirSync,
rmSync } from 'node:fs';
import { join, relative, dirname } from 'node:path';
import { tmpdir } from 'node:os';
import { fileURLToPath } from 'node:url';

const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..');
const OUT = join(ROOT, 'src', 'compile_cache_zstd.dict');
const MAXDICT = 16384;
const REQUIRED_ZSTD = '1.5.7'; // keep in sync with deps/zstd/lib/zstd.h

// Fixed corpus, relative to the repo root. Chosen to be diverse, always
// present in a checkout, and disjoint from the held-out measurement corpora
// (e.g. test/parallel) used to report size numbers.
const CORPUS = ['lib', 'tools', 'deps/npm/node_modules'];

// V8 randomizes the hash seed per process; re-exec under --predictable so the
// harvested caches — and the trained dictionary — are deterministic.
if (!process.execArgv.includes('--predictable')) {
const r = spawnSync(process.execPath,
['--predictable', fileURLToPath(import.meta.url), ...process.argv.slice(2)],
{ stdio: 'inherit' });
process.exit(r.status ?? 1);
}

function checkZstd() {
const r = spawnSync('zstd', ['--version'], { encoding: 'utf8' });
if (r.status !== 0) {
console.error('error: zstd CLI not found on PATH'); process.exit(1);
}
const m = r.stdout.match(/v(\d+\.\d+\.\d+)/);
if (!m || m[1] !== REQUIRED_ZSTD) {
console.error(`error: zstd ${REQUIRED_ZSTD} required (matching deps/zstd), ` +
`found ${m ? m[1] : 'unknown'}`);
process.exit(1);
}
}

const PARAMS = ['exports', 'require', 'module', '__filename', '__dirname'];

function* walk(dir) {
let ents;
try { ents = readdirSync(dir, { withFileTypes: true }); } catch { return; }
for (const e of ents.sort((a, b) => (a.name < b.name ? -1 : 1))) {
const p = join(dir, e.name);
if (e.isDirectory()) yield* walk(p);
else if (e.isFile() && p.endsWith('.js')) yield p;
}
}

async function main() {
checkZstd();
const { default: vm } = await import('node:vm');

const files = [];
for (const root of CORPUS) for (const f of walk(join(ROOT, root))) files.push(f);
files.sort();

const samples = mkdtempSync(join(tmpdir(), 'cc-dict-'));
const sampleFiles = [];
let ok = 0;
let r;
try {
for (const f of files) {
let code;
try { code = readFileSync(f, 'utf8'); } catch { continue; }
try {
const fn = vm.compileFunction(code, PARAMS,
{ filename: f, produceCachedData: true });
const cd = fn.cachedData;
if (cd && cd.length > 0) {
const name = relative(ROOT, f).replace(/[\\/]/g, '_') + '.cache';
const out = join(samples, name);
writeFileSync(out, cd);
sampleFiles.push(out);
ok++;
}
} catch { /* skip modules V8 can't compile standalone */ }
}
sampleFiles.sort();
console.error(`harvested ${ok}/${files.length} code caches from ` +
`${CORPUS.join(', ')}`);

// One sample path per file is spread on argv; the fixed corpus (~1.4k
// files, a few hundred KB of paths) stays well under ARG_MAX.
r = spawnSync('zstd',
['--train', ...sampleFiles, `--maxdict=${MAXDICT}`, '-f', '-o', OUT],
{ stdio: ['ignore', 'ignore', 'inherit'] });
} finally {
rmSync(samples, { recursive: true, force: true });
}
if (r.status !== 0) { console.error('error: zstd --train failed'); process.exit(1); }

const size = readFileSync(OUT).length;
console.error(`wrote ${relative(ROOT, OUT)} (${size} bytes)`);
}

main().catch((e) => { console.error(e); process.exit(1); });
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
Binary file modifiedsrc/compile_cache_zstd.dict
Binary file not shown.
168 changes: 168 additions & 0 deletions tools/train_compile_cache_dict.mjs
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,168 @@
#!/usr/bin/env node
// =============================================================================
// train_compile_cache_dict.mjs
//
// Maintainer tool that regenerates src/compile_cache_zstd.dict, the zstd
// dictionary embedded in the node binary and used to shrink compile-cache
// entries on disk (see src/compile_cache.cc).
//
// -----------------------------------------------------------------------------
// What it does
// -----------------------------------------------------------------------------
// The node compile cache stores V8 code caches (the bytecode/metadata blob V8
// produces when it compiles a module) so later runs can skip recompilation.
// These blobs share a lot of structure across files, so we zstd-compress them
// with a trained dictionary to cut the cache's on-disk footprint.
//
// This script builds that dictionary end to end:
//
// 1. Walks a fixed in-tree corpus (CORPUS below) in sorted order.
// 2. For each .js file, harvests a V8 code cache via vm.compileFunction with
// produceCachedData — the same shape the CommonJS loader produces at
// runtime — and writes each blob to a temp directory.
// 3. Feeds all the harvested blobs to `zstd --train`, capping the result at
// MAXDICT (16 KiB) bytes.
// 4. Overwrites src/compile_cache_zstd.dict with the trained dictionary.
//
// The shipped src/compile_cache_zstd.dict is the source of truth; this script
// exists to document and reproduce exactly how that file was made, so a future
// maintainer can regenerate it (e.g. after a V8/corpus change) and review the
// resulting diff.
//
// -----------------------------------------------------------------------------
// Usage
// -----------------------------------------------------------------------------
// node tools/train_compile_cache_dict.mjs
//
// Run from anywhere (paths are resolved relative to this file). The output is
// written in place to src/compile_cache_zstd.dict; inspect the diff afterwards
// and commit it if it looks right. Progress is printed to stderr.
//
// Prerequisites:
// * the `zstd` CLI on PATH, version REQUIRED_ZSTD (matching deps/zstd).
// * a built node on PATH (this is the node you invoke it with).
//
// Note: --predictable is added automatically (see below) — you do not need to
// pass it yourself.
//
// -----------------------------------------------------------------------------
// Reproducibility
// -----------------------------------------------------------------------------
// The output is byte-for-byte stable only if all of the following are pinned:
//
// * node is run with --predictable. V8 otherwise randomizes the string hash
// seed per process, and that seed leaks into vm.compileFunction cachedData,
// so every harvest (and therefore every trained dictionary) would differ
// run-to-run even on the same machine. This script re-executes itself with
// --predictable if needed.
// * the corpus and its order are fixed (CORPUS below, walked sorted).
// * the node build is fixed: cachedData embeds the V8 version and build
// flags, so a different node produces a different (still valid) dictionary.
// * the zstd CLI matches deps/zstd (currently 1.5.7). ZDICT training defaults
// change between zstd releases; this script refuses to run on a mismatch.
//
// Training on --predictable caches is fine even though the runtime consumes
// non-predictable caches: the dictionary only supplies shared substrings, and
// Persist() keeps min(plain, dict) per entry, so a less-than-ideal dictionary
// can never make any entry larger.
// =============================================================================

import { spawnSync } from 'node:child_process';
import { readFileSync, writeFileSync, mkdtempSync, readdirSync,
rmSync } from 'node:fs';
import { join, relative, dirname } from 'node:path';
import { tmpdir } from 'node:os';
import { fileURLToPath } from 'node:url';

const ROOT = join(dirname(fileURLToPath(import.meta.url)), '..');
const OUT = join(ROOT, 'src', 'compile_cache_zstd.dict');
const MAXDICT = 16384;
const REQUIRED_ZSTD = '1.5.7'; // keep in sync with deps/zstd/lib/zstd.h

// Fixed corpus, relative to the repo root. Chosen to be diverse, always
// present in a checkout, and disjoint from the held-out measurement corpora
// (e.g. test/parallel) used to report size numbers.
const CORPUS = ['lib', 'tools', 'deps/npm/node_modules'];

// V8 randomizes the hash seed per process; re-exec under --predictable so the
// harvested caches — and the trained dictionary — are deterministic.
if (!process.execArgv.includes('--predictable')) {
const r = spawnSync(process.execPath,
['--predictable', fileURLToPath(import.meta.url), ...process.argv.slice(2)],
{ stdio: 'inherit' });
process.exit(r.status ?? 1);
}

function checkZstd() {
const r = spawnSync('zstd', ['--version'], { encoding: 'utf8' });
if (r.status !== 0) {
console.error('error: zstd CLI not found on PATH'); process.exit(1);
}
const m = r.stdout.match(/v(\d+\.\d+\.\d+)/);
if (!m || m[1] !== REQUIRED_ZSTD) {
console.error(`error: zstd ${REQUIRED_ZSTD} required (matching deps/zstd), ` +
`found ${m ? m[1] : 'unknown'}`);
process.exit(1);
}
}

const PARAMS = ['exports', 'require', 'module', '__filename', '__dirname'];

function* walk(dir) {
let ents;
try { ents = readdirSync(dir, { withFileTypes: true }); } catch { return; }
for (const e of ents.sort((a, b) => (a.name < b.name ? -1 : 1))) {
const p = join(dir, e.name);
if (e.isDirectory()) yield* walk(p);
else if (e.isFile() && p.endsWith('.js')) yield p;
}
}

async function main() {
checkZstd();
const { default: vm } = await import('node:vm');

const files = [];
for (const root of CORPUS) for (const f of walk(join(ROOT, root))) files.push(f);
files.sort();

const samples = mkdtempSync(join(tmpdir(), 'cc-dict-'));
const sampleFiles = [];
let ok = 0;
let r;
try {
for (const f of files) {
let code;
try { code = readFileSync(f, 'utf8'); } catch { continue; }
try {
const fn = vm.compileFunction(code, PARAMS,
{ filename: f, produceCachedData: true });
const cd = fn.cachedData;
if (cd && cd.length > 0) {
const name = relative(ROOT, f).replace(/[\\/]/g, '_') + '.cache';
const out = join(samples, name);
writeFileSync(out, cd);
sampleFiles.push(out);
ok++;
}
} catch { /* skip modules V8 can't compile standalone */ }
}
sampleFiles.sort();
console.error(`harvested ${ok}/${files.length} code caches from ` +
`${CORPUS.join(', ')}`);

// One sample path per file is spread on argv; the fixed corpus (~1.4k
// files, a few hundred KB of paths) stays well under ARG_MAX.
r = spawnSync('zstd',
['--train', ...sampleFiles, `--maxdict=${MAXDICT}`, '-f', '-o', OUT],
{ stdio: ['ignore', 'ignore', 'inherit'] });
} finally {
rmSync(samples, { recursive: true, force: true });
}
if (r.status !== 0) { console.error('error: zstd --train failed'); process.exit(1); }

const size = readFileSync(OUT).length;
console.error(`wrote ${relative(ROOT, OUT)} (${size} bytes)`);
}

main().catch((e) => { console.error(e); process.exit(1); });