Commit e3ae912

Browse files
codebytereaduh95
authored andcommitted
fs: use sized reads for large files in readFileUtf8
fs.readFileSync(path, 'utf8') read the whole file in 8 KiB read() calls appended to a std::string, i.e. one syscall and a potential reallocation per 8 KiB (an 8 MiB file took ~1400 read() calls). Keep the exact old sequence for small files (one read into the 8 KiB stack buffer, one read reporting EOF). Once a read fills the stack buffer, read the rest directly into one heap buffer sized from fstat() (plus one byte so that the EOF read does not force growth), growing geometrically only when the size is unavailable or wrong. The size is only an allocation hint: reading continues until read() reports EOF, so procfs/sysfs files, FIFOs, files that change while being read and file descriptors positioned mid-file behave as before, and the bytes handed to StringBytes::Encode() are exactly the ones read. Signed-off-by: Shelley Vohr <shelley.vohr@gmail.com> PR-URL: #65328 Reviewed-By: Matteo Collina <matteo.collina@gmail.com> Reviewed-By: Yagiz Nizipli <yagiz@nizipli.com>
1 parent bc1e19b commit e3ae912

2 files changed

Lines changed: 160 additions & 2 deletions

File tree

β€Žsrc/node_file.ccβ€Ž

Lines changed: 62 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -2931,10 +2931,20 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29312931
uv_fs_req_cleanup(&req);
29322932
});
29332933

2934+
// Past the first 8 KiB, read into one heap buffer sized from fstat(); the
2935+
// size is only a hint, reading continues until read() reports EOF.
29342936
std::string result{};
29352937
char buffer[8192];
29362938
uv_buf_t buf = uv_buf_init(buffer, sizeof(buffer));
29372939

2940+
char* big = nullptr;
2941+
size_t big_len = 0;
2942+
size_t big_cap = 0;
2943+
bool sized = false;
2944+
auto free_big = OnScopeLeave([&big]() { free(big); });
2945+
constexprsize_tkMinChunk = 64 * 1024;
2946+
constexprsize_tkMaxChunk = 8 * 1024 * 1024;
2947+
29382948
FS_SYNC_TRACE_BEGIN(read);
29392949
while (true) {
29402950
auto r = uv_fs_read(nullptr, &req, file, &buf, 1, -1, nullptr);
@@ -2947,12 +2957,62 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29472957
if (r <= 0) {
29482958
break;
29492959
}
2950-
result.append(buf.base, r);
2960+
if (big == nullptr) {
2961+
result.append(buf.base, r);
2962+
if (static_cast<size_t>(r) < sizeof(buffer)) {
2963+
continue;
2964+
}
2965+
// Switch to the heap buffer.
2966+
uv_fs_req_cleanup(&req);
2967+
big_cap = kMinChunk;
2968+
big = UncheckedMalloc<char>(big_cap);
2969+
if (big == nullptr) {
2970+
FS_SYNC_TRACE_END(read);
2971+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
2972+
}
2973+
memcpy(big, result.data(), result.size());
2974+
big_len = result.size();
2975+
result = std::string();
2976+
} else {
2977+
big_len += static_cast<size_t>(r);
2978+
}
2979+
if (big_len == big_cap) {
2980+
// +1 leaves room for the read() that reports EOF.
2981+
size_t new_cap =
2982+
big_cap + std::min(kMaxChunk, std::max(kMinChunk, big_cap));
2983+
if (!sized) {
2984+
sized = true;
2985+
uv_fs_req_cleanup(&req);
2986+
uv_fs_t stat_req;
2987+
if (uv_fs_fstat(nullptr, &stat_req, file, nullptr) == 0) {
2988+
constuv_stat_t* const st =
2989+
static_cast<constuv_stat_t*>(stat_req.ptr);
2990+
if ((st->st_mode & S_IFMT) == S_IFREG &&
2991+
static_cast<uint64_t>(st->st_size) > big_len &&
2992+
static_cast<uint64_t>(st->st_size) <
2993+
static_cast<uint64_t>(v8::String::kMaxLength)) {
2994+
new_cap = static_cast<size_t>(st->st_size) + 1;
2995+
}
2996+
}
2997+
uv_fs_req_cleanup(&stat_req);
2998+
}
2999+
char* const grown = UncheckedRealloc<char>(big, new_cap);
3000+
if (grown == nullptr) {
3001+
FS_SYNC_TRACE_END(read);
3002+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
3003+
}
3004+
big = grown;
3005+
big_cap = new_cap;
3006+
}
3007+
buf = uv_buf_init(big + big_len, std::min(kMaxChunk, big_cap - big_len));
29513008
}
29523009
FS_SYNC_TRACE_END(read);
29533010

29543011
Local<Value> val;
2955-
if (!ToV8Value(env->context(), result, isolate).ToLocal(&val)) {
3012+
const std::string_view content = big != nullptr
3013+
? std::string_view(big, big_len)
3014+
: std::string_view(result);
3015+
if (!ToV8Value(env->context(), content, isolate).ToLocal(&val)) {
29563016
return;
29573017
}
29583018

Lines changed: 98 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,98 @@
1+
'use strict';
2+
// fs.readFileSync(path, 'utf8') takes a dedicated native path. Its result must
3+
// equal fs.readFileSync(path).toString('utf8') for every file size (in
4+
// particular around its internal 8 KiB stack buffer and for multi-megabyte
5+
// files), for file descriptors positioned mid-file, and for files whose
6+
// reported size is wrong (procfs reports 0, sysfs reports a page).
7+
constcommon=require('../common');
8+
consttmpdir=require('../common/tmpdir');
9+
constassert=require('assert');
10+
constfs=require('fs');
11+
12+
tmpdir.refresh();
13+
14+
functioncontent(size){
15+
// Multi-byte characters straddling every possible chunk boundary.
16+
constunit='abcdé€\u{1F600}\n';
17+
lets=unit.repeat(Math.ceil(size/unit.length));
18+
s=s.slice(0,size);
19+
// Avoid ending on a lone surrogate produced by slice().
20+
if(/[\ud800-\udbff]$/.test(s))s=s.slice(0,-1)+'x';
21+
returns;
22+
}
23+
24+
constsizes=[0,1,8190,8191,8192,8193,8194,16383,16384,16385,
25+
65535,65536,65537,100000,(1<<20)-1,1<<20,(1<<20)+1,
26+
(8<<20)+5];
27+
for(constsizeofsizes){
28+
constfile=tmpdir.resolve(`f-${size}.txt`);
29+
conststr=content(size);
30+
fs.writeFileSync(file,str);
31+
constexpected=fs.readFileSync(file).toString('utf8');
32+
assert.strictEqual(fs.readFileSync(file,'utf8'),expected,`size ${size} by path`);
33+
assert.strictEqual(fs.readFileSync(file,{encoding: 'utf-8'}),expected,`size ${size} utf-8 alias`);
34+
// By fd: from the start (leaves the fd at EOF), then at EOF, then from a
35+
// mid-file position on a fresh fd.
36+
letfd=fs.openSync(file,'r');
37+
try{
38+
assert.strictEqual(fs.readFileSync(fd,'utf8'),expected,`size ${size} by fd`);
39+
assert.strictEqual(fs.readFileSync(fd,'utf8'),'',`size ${size} by fd at EOF`);
40+
}finally{
41+
fs.closeSync(fd);
42+
}
43+
if(size>10){
44+
fd=fs.openSync(file,'r');
45+
try{
46+
// Advance the fd 3 bytes (inside the ASCII prefix, so still valid UTF-8).
47+
assert.strictEqual(fs.readSync(fd,Buffer.alloc(3),0,3,null),3);
48+
assert.strictEqual(fs.readFileSync(fd,'utf8'),Buffer.from(expected).subarray(3).toString('utf8'),
49+
`size ${size} by fd at offset 3`);
50+
}finally{
51+
fs.closeSync(fd);
52+
}
53+
}
54+
}
55+
56+
// Binary garbage is decoded with replacement characters identically.
57+
{
58+
constfile=tmpdir.resolve('binary.bin');
59+
constbuf=Buffer.alloc(20000);
60+
for(leti=0;i<buf.length;i++)buf[i]=(i*7919)&0xff;
61+
fs.writeFileSync(file,buf);
62+
assert.strictEqual(fs.readFileSync(file,'utf8'),buf.toString('utf8'));
63+
}
64+
65+
// Files whose st_size does not describe their content.
66+
if(common.isLinux){
67+
for(constfileof['/proc/self/status','/proc/self/smaps','/proc/cpuinfo',
68+
'/proc/version','/sys/kernel/mm/transparent_hugepage/enabled']){
69+
letviaBuffer;
70+
try{
71+
viaBuffer=fs.readFileSync(file);
72+
}catch{
73+
continue;// Not available in this environment.
74+
}
75+
constviaUtf8=fs.readFileSync(file,'utf8');
76+
if(file!=='/proc/version'&&file.startsWith('/proc/')){
77+
// Content legitimately differs between two reads; compare shape instead.
78+
assert.ok(viaUtf8.length>0);
79+
assert.strictEqual(viaUtf8.split('\n').length>5,true,file);
80+
// Of these, smaps reliably exceeds the 8 KiB stack buffer.
81+
if(file==='/proc/self/smaps')assert.ok(viaUtf8.length>8192,`smaps is only ${viaUtf8.length} chars`);
82+
}else{
83+
assert.strictEqual(viaUtf8,viaBuffer.toString('utf8'),file);
84+
}
85+
}
86+
}
87+
88+
// Directory: same outcome either way (EISDIR, except on platforms where
89+
// read() accepts directories, e.g. AIX).
90+
functionoutcome(read){
91+
try{
92+
returnread();
93+
}catch(err){
94+
returnerr.code;
95+
}
96+
}
97+
assert.strictEqual(outcome(()=>fs.readFileSync(tmpdir.path,'utf8')),
98+
outcome(()=>fs.readFileSync(tmpdir.path).toString('utf8')));

0 commit comments

Comments
Β (0)
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Add copy buttons to all
 blocks\n(function() {\n function addCopyButtons() {\n document.querySelectorAll('pre code').forEach(function(codeBlock) {\n if (codeBlock.parentElement.hasAttribute('data-copy-added')) return;\n codeBlock.parentElement.setAttribute('data-copy-added', 'true');\n \n var btn = document.createElement('button');\n btn.textContent = 'Copy';\n btn.style.cssText = 'position:absolute;top:4px;right:4px;padding:2px 8px;font-size:11px;background:#4ecdc4;border:none;border-radius:4px;color:#1a1a2e;cursor:pointer;opacity:0.7;transition:opacity 0.2s;';\n btn.onmouseover = function() { this.style.opacity = '1'; };\n btn.onmouseout = function() { this.style.opacity = '0.7'; };\n btn.onclick = function() {\n navigator.clipboard.writeText(codeBlock.textContent).then(function() {\n btn.textContent = 'Copied!';\n setTimeout(function() { btn.textContent = 'Copy'; }, 1500);\n });\n };\n codeBlock.parentElement.style.position = 'relative';\n codeBlock.parentElement.appendChild(btn);\n });\n }\n \n addCopyButtons();\n \n // Re-run on dynamic content\n var observer = new MutationObserver(addCopyButtons);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Add Copy Buttons to Code Blocks");
}
} catch(__e) { console.warn('[Userscript:Add Copy Buttons to Code Blocks]', __e); }
})();
(function(){
try {
var __m = "github.com";
var __re = new RegExp('^' + "github\\.com" + '
Skip to content

Commit e3ae912

Browse files
codebytereaduh95
authored andcommitted
fs: use sized reads for large files in readFileUtf8
fs.readFileSync(path, 'utf8') read the whole file in 8 KiB read() calls appended to a std::string, i.e. one syscall and a potential reallocation per 8 KiB (an 8 MiB file took ~1400 read() calls). Keep the exact old sequence for small files (one read into the 8 KiB stack buffer, one read reporting EOF). Once a read fills the stack buffer, read the rest directly into one heap buffer sized from fstat() (plus one byte so that the EOF read does not force growth), growing geometrically only when the size is unavailable or wrong. The size is only an allocation hint: reading continues until read() reports EOF, so procfs/sysfs files, FIFOs, files that change while being read and file descriptors positioned mid-file behave as before, and the bytes handed to StringBytes::Encode() are exactly the ones read. Signed-off-by: Shelley Vohr <shelley.vohr@gmail.com> PR-URL: #65328 Reviewed-By: Matteo Collina <matteo.collina@gmail.com> Reviewed-By: Yagiz Nizipli <yagiz@nizipli.com>
1 parent bc1e19b commit e3ae912

2 files changed

Lines changed: 160 additions & 2 deletions

File tree

β€Žsrc/node_file.ccβ€Ž

Lines changed: 62 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -2931,10 +2931,20 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29312931
uv_fs_req_cleanup(&req);
29322932
});
29332933

2934+
// Past the first 8 KiB, read into one heap buffer sized from fstat(); the
2935+
// size is only a hint, reading continues until read() reports EOF.
29342936
std::string result{};
29352937
char buffer[8192];
29362938
uv_buf_t buf = uv_buf_init(buffer, sizeof(buffer));
29372939

2940+
char* big = nullptr;
2941+
size_t big_len = 0;
2942+
size_t big_cap = 0;
2943+
bool sized = false;
2944+
auto free_big = OnScopeLeave([&big]() { free(big); });
2945+
constexprsize_tkMinChunk = 64 * 1024;
2946+
constexprsize_tkMaxChunk = 8 * 1024 * 1024;
2947+
29382948
FS_SYNC_TRACE_BEGIN(read);
29392949
while (true) {
29402950
auto r = uv_fs_read(nullptr, &req, file, &buf, 1, -1, nullptr);
@@ -2947,12 +2957,62 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29472957
if (r <= 0) {
29482958
break;
29492959
}
2950-
result.append(buf.base, r);
2960+
if (big == nullptr) {
2961+
result.append(buf.base, r);
2962+
if (static_cast<size_t>(r) < sizeof(buffer)) {
2963+
continue;
2964+
}
2965+
// Switch to the heap buffer.
2966+
uv_fs_req_cleanup(&req);
2967+
big_cap = kMinChunk;
2968+
big = UncheckedMalloc<char>(big_cap);
2969+
if (big == nullptr) {
2970+
FS_SYNC_TRACE_END(read);
2971+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
2972+
}
2973+
memcpy(big, result.data(), result.size());
2974+
big_len = result.size();
2975+
result = std::string();
2976+
} else {
2977+
big_len += static_cast<size_t>(r);
2978+
}
2979+
if (big_len == big_cap) {
2980+
// +1 leaves room for the read() that reports EOF.
2981+
size_t new_cap =
2982+
big_cap + std::min(kMaxChunk, std::max(kMinChunk, big_cap));
2983+
if (!sized) {
2984+
sized = true;
2985+
uv_fs_req_cleanup(&req);
2986+
uv_fs_t stat_req;
2987+
if (uv_fs_fstat(nullptr, &stat_req, file, nullptr) == 0) {
2988+
constuv_stat_t* const st =
2989+
static_cast<constuv_stat_t*>(stat_req.ptr);
2990+
if ((st->st_mode & S_IFMT) == S_IFREG &&
2991+
static_cast<uint64_t>(st->st_size) > big_len &&
2992+
static_cast<uint64_t>(st->st_size) <
2993+
static_cast<uint64_t>(v8::String::kMaxLength)) {
2994+
new_cap = static_cast<size_t>(st->st_size) + 1;
2995+
}
2996+
}
2997+
uv_fs_req_cleanup(&stat_req);
2998+
}
2999+
char* const grown = UncheckedRealloc<char>(big, new_cap);
3000+
if (grown == nullptr) {
3001+
FS_SYNC_TRACE_END(read);
3002+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
3003+
}
3004+
big = grown;
3005+
big_cap = new_cap;
3006+
}
3007+
buf = uv_buf_init(big + big_len, std::min(kMaxChunk, big_cap - big_len));
29513008
}
29523009
FS_SYNC_TRACE_END(read);
29533010

29543011
Local<Value> val;
2955-
if (!ToV8Value(env->context(), result, isolate).ToLocal(&val)) {
3012+
const std::string_view content = big != nullptr
3013+
? std::string_view(big, big_len)
3014+
: std::string_view(result);
3015+
if (!ToV8Value(env->context(), content, isolate).ToLocal(&val)) {
29563016
return;
29573017
}
29583018

Lines changed: 98 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,98 @@
1+
'use strict';
2+
// fs.readFileSync(path, 'utf8') takes a dedicated native path. Its result must
3+
// equal fs.readFileSync(path).toString('utf8') for every file size (in
4+
// particular around its internal 8 KiB stack buffer and for multi-megabyte
5+
// files), for file descriptors positioned mid-file, and for files whose
6+
// reported size is wrong (procfs reports 0, sysfs reports a page).
7+
constcommon=require('../common');
8+
consttmpdir=require('../common/tmpdir');
9+
constassert=require('assert');
10+
constfs=require('fs');
11+
12+
tmpdir.refresh();
13+
14+
functioncontent(size){
15+
// Multi-byte characters straddling every possible chunk boundary.
16+
constunit='abcdé€\u{1F600}\n';
17+
lets=unit.repeat(Math.ceil(size/unit.length));
18+
s=s.slice(0,size);
19+
// Avoid ending on a lone surrogate produced by slice().
20+
if(/[\ud800-\udbff]$/.test(s))s=s.slice(0,-1)+'x';
21+
returns;
22+
}
23+
24+
constsizes=[0,1,8190,8191,8192,8193,8194,16383,16384,16385,
25+
65535,65536,65537,100000,(1<<20)-1,1<<20,(1<<20)+1,
26+
(8<<20)+5];
27+
for(constsizeofsizes){
28+
constfile=tmpdir.resolve(`f-${size}.txt`);
29+
conststr=content(size);
30+
fs.writeFileSync(file,str);
31+
constexpected=fs.readFileSync(file).toString('utf8');
32+
assert.strictEqual(fs.readFileSync(file,'utf8'),expected,`size ${size} by path`);
33+
assert.strictEqual(fs.readFileSync(file,{encoding: 'utf-8'}),expected,`size ${size} utf-8 alias`);
34+
// By fd: from the start (leaves the fd at EOF), then at EOF, then from a
35+
// mid-file position on a fresh fd.
36+
letfd=fs.openSync(file,'r');
37+
try{
38+
assert.strictEqual(fs.readFileSync(fd,'utf8'),expected,`size ${size} by fd`);
39+
assert.strictEqual(fs.readFileSync(fd,'utf8'),'',`size ${size} by fd at EOF`);
40+
}finally{
41+
fs.closeSync(fd);
42+
}
43+
if(size>10){
44+
fd=fs.openSync(file,'r');
45+
try{
46+
// Advance the fd 3 bytes (inside the ASCII prefix, so still valid UTF-8).
47+
assert.strictEqual(fs.readSync(fd,Buffer.alloc(3),0,3,null),3);
48+
assert.strictEqual(fs.readFileSync(fd,'utf8'),Buffer.from(expected).subarray(3).toString('utf8'),
49+
`size ${size} by fd at offset 3`);
50+
}finally{
51+
fs.closeSync(fd);
52+
}
53+
}
54+
}
55+
56+
// Binary garbage is decoded with replacement characters identically.
57+
{
58+
constfile=tmpdir.resolve('binary.bin');
59+
constbuf=Buffer.alloc(20000);
60+
for(leti=0;i<buf.length;i++)buf[i]=(i*7919)&0xff;
61+
fs.writeFileSync(file,buf);
62+
assert.strictEqual(fs.readFileSync(file,'utf8'),buf.toString('utf8'));
63+
}
64+
65+
// Files whose st_size does not describe their content.
66+
if(common.isLinux){
67+
for(constfileof['/proc/self/status','/proc/self/smaps','/proc/cpuinfo',
68+
'/proc/version','/sys/kernel/mm/transparent_hugepage/enabled']){
69+
letviaBuffer;
70+
try{
71+
viaBuffer=fs.readFileSync(file);
72+
}catch{
73+
continue;// Not available in this environment.
74+
}
75+
constviaUtf8=fs.readFileSync(file,'utf8');
76+
if(file!=='/proc/version'&&file.startsWith('/proc/')){
77+
// Content legitimately differs between two reads; compare shape instead.
78+
assert.ok(viaUtf8.length>0);
79+
assert.strictEqual(viaUtf8.split('\n').length>5,true,file);
80+
// Of these, smaps reliably exceeds the 8 KiB stack buffer.
81+
if(file==='/proc/self/smaps')assert.ok(viaUtf8.length>8192,`smaps is only ${viaUtf8.length} chars`);
82+
}else{
83+
assert.strictEqual(viaUtf8,viaBuffer.toString('utf8'),file);
84+
}
85+
}
86+
}
87+
88+
// Directory: same outcome either way (EISDIR, except on platforms where
89+
// read() accepts directories, e.g. AIX).
90+
functionoutcome(read){
91+
try{
92+
returnread();
93+
}catch(err){
94+
returnerr.code;
95+
}
96+
}
97+
assert.strictEqual(outcome(()=>fs.readFileSync(tmpdir.path,'utf8')),
98+
outcome(()=>fs.readFileSync(tmpdir.path).toString('utf8')));

0 commit comments

Comments
Β (0)
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Force GitHub README to respect dark mode\n(function() {\n var style = document.createElement('style');\n style.textContent = '\n .markdown-body {\n color-scheme: dark light;\n }\n .markdown-body pre { background: #161b22 !important; }\n .markdown-body code { background: rgba(110, 118, 129, 0.4) !important; }\n .markdown-body table th, .markdown-body table td { border-color: #30363d !important; }\n .markdown-body img { background: #0d1117; }\n .markdown-body blockquote { border-left-color: #8b949e; }\n .markdown-body hr { border-color: #30363d; }\n ';\n document.head.appendChild(style);\n})();", "GitHub Dark Mode README Fix"); } } catch(__e) { console.warn('[Userscript:GitHub Dark Mode README Fix]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content

Commit e3ae912

Browse files
codebytereaduh95
authored andcommitted
fs: use sized reads for large files in readFileUtf8
fs.readFileSync(path, 'utf8') read the whole file in 8 KiB read() calls appended to a std::string, i.e. one syscall and a potential reallocation per 8 KiB (an 8 MiB file took ~1400 read() calls). Keep the exact old sequence for small files (one read into the 8 KiB stack buffer, one read reporting EOF). Once a read fills the stack buffer, read the rest directly into one heap buffer sized from fstat() (plus one byte so that the EOF read does not force growth), growing geometrically only when the size is unavailable or wrong. The size is only an allocation hint: reading continues until read() reports EOF, so procfs/sysfs files, FIFOs, files that change while being read and file descriptors positioned mid-file behave as before, and the bytes handed to StringBytes::Encode() are exactly the ones read. Signed-off-by: Shelley Vohr <shelley.vohr@gmail.com> PR-URL: #65328 Reviewed-By: Matteo Collina <matteo.collina@gmail.com> Reviewed-By: Yagiz Nizipli <yagiz@nizipli.com>
1 parent bc1e19b commit e3ae912

2 files changed

Lines changed: 160 additions & 2 deletions

File tree

β€Žsrc/node_file.ccβ€Ž

Lines changed: 62 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -2931,10 +2931,20 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29312931
uv_fs_req_cleanup(&req);
29322932
});
29332933

2934+
// Past the first 8 KiB, read into one heap buffer sized from fstat(); the
2935+
// size is only a hint, reading continues until read() reports EOF.
29342936
std::string result{};
29352937
char buffer[8192];
29362938
uv_buf_t buf = uv_buf_init(buffer, sizeof(buffer));
29372939

2940+
char* big = nullptr;
2941+
size_t big_len = 0;
2942+
size_t big_cap = 0;
2943+
bool sized = false;
2944+
auto free_big = OnScopeLeave([&big]() { free(big); });
2945+
constexprsize_tkMinChunk = 64 * 1024;
2946+
constexprsize_tkMaxChunk = 8 * 1024 * 1024;
2947+
29382948
FS_SYNC_TRACE_BEGIN(read);
29392949
while (true) {
29402950
auto r = uv_fs_read(nullptr, &req, file, &buf, 1, -1, nullptr);
@@ -2947,12 +2957,62 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29472957
if (r <= 0) {
29482958
break;
29492959
}
2950-
result.append(buf.base, r);
2960+
if (big == nullptr) {
2961+
result.append(buf.base, r);
2962+
if (static_cast<size_t>(r) < sizeof(buffer)) {
2963+
continue;
2964+
}
2965+
// Switch to the heap buffer.
2966+
uv_fs_req_cleanup(&req);
2967+
big_cap = kMinChunk;
2968+
big = UncheckedMalloc<char>(big_cap);
2969+
if (big == nullptr) {
2970+
FS_SYNC_TRACE_END(read);
2971+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
2972+
}
2973+
memcpy(big, result.data(), result.size());
2974+
big_len = result.size();
2975+
result = std::string();
2976+
} else {
2977+
big_len += static_cast<size_t>(r);
2978+
}
2979+
if (big_len == big_cap) {
2980+
// +1 leaves room for the read() that reports EOF.
2981+
size_t new_cap =
2982+
big_cap + std::min(kMaxChunk, std::max(kMinChunk, big_cap));
2983+
if (!sized) {
2984+
sized = true;
2985+
uv_fs_req_cleanup(&req);
2986+
uv_fs_t stat_req;
2987+
if (uv_fs_fstat(nullptr, &stat_req, file, nullptr) == 0) {
2988+
constuv_stat_t* const st =
2989+
static_cast<constuv_stat_t*>(stat_req.ptr);
2990+
if ((st->st_mode & S_IFMT) == S_IFREG &&
2991+
static_cast<uint64_t>(st->st_size) > big_len &&
2992+
static_cast<uint64_t>(st->st_size) <
2993+
static_cast<uint64_t>(v8::String::kMaxLength)) {
2994+
new_cap = static_cast<size_t>(st->st_size) + 1;
2995+
}
2996+
}
2997+
uv_fs_req_cleanup(&stat_req);
2998+
}
2999+
char* const grown = UncheckedRealloc<char>(big, new_cap);
3000+
if (grown == nullptr) {
3001+
FS_SYNC_TRACE_END(read);
3002+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
3003+
}
3004+
big = grown;
3005+
big_cap = new_cap;
3006+
}
3007+
buf = uv_buf_init(big + big_len, std::min(kMaxChunk, big_cap - big_len));
29513008
}
29523009
FS_SYNC_TRACE_END(read);
29533010

29543011
Local<Value> val;
2955-
if (!ToV8Value(env->context(), result, isolate).ToLocal(&val)) {
3012+
const std::string_view content = big != nullptr
3013+
? std::string_view(big, big_len)
3014+
: std::string_view(result);
3015+
if (!ToV8Value(env->context(), content, isolate).ToLocal(&val)) {
29563016
return;
29573017
}
29583018

Lines changed: 98 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,98 @@
1+
'use strict';
2+
// fs.readFileSync(path, 'utf8') takes a dedicated native path. Its result must
3+
// equal fs.readFileSync(path).toString('utf8') for every file size (in
4+
// particular around its internal 8 KiB stack buffer and for multi-megabyte
5+
// files), for file descriptors positioned mid-file, and for files whose
6+
// reported size is wrong (procfs reports 0, sysfs reports a page).
7+
constcommon=require('../common');
8+
consttmpdir=require('../common/tmpdir');
9+
constassert=require('assert');
10+
constfs=require('fs');
11+
12+
tmpdir.refresh();
13+
14+
functioncontent(size){
15+
// Multi-byte characters straddling every possible chunk boundary.
16+
constunit='abcdé€\u{1F600}\n';
17+
lets=unit.repeat(Math.ceil(size/unit.length));
18+
s=s.slice(0,size);
19+
// Avoid ending on a lone surrogate produced by slice().
20+
if(/[\ud800-\udbff]$/.test(s))s=s.slice(0,-1)+'x';
21+
returns;
22+
}
23+
24+
constsizes=[0,1,8190,8191,8192,8193,8194,16383,16384,16385,
25+
65535,65536,65537,100000,(1<<20)-1,1<<20,(1<<20)+1,
26+
(8<<20)+5];
27+
for(constsizeofsizes){
28+
constfile=tmpdir.resolve(`f-${size}.txt`);
29+
conststr=content(size);
30+
fs.writeFileSync(file,str);
31+
constexpected=fs.readFileSync(file).toString('utf8');
32+
assert.strictEqual(fs.readFileSync(file,'utf8'),expected,`size ${size} by path`);
33+
assert.strictEqual(fs.readFileSync(file,{encoding: 'utf-8'}),expected,`size ${size} utf-8 alias`);
34+
// By fd: from the start (leaves the fd at EOF), then at EOF, then from a
35+
// mid-file position on a fresh fd.
36+
letfd=fs.openSync(file,'r');
37+
try{
38+
assert.strictEqual(fs.readFileSync(fd,'utf8'),expected,`size ${size} by fd`);
39+
assert.strictEqual(fs.readFileSync(fd,'utf8'),'',`size ${size} by fd at EOF`);
40+
}finally{
41+
fs.closeSync(fd);
42+
}
43+
if(size>10){
44+
fd=fs.openSync(file,'r');
45+
try{
46+
// Advance the fd 3 bytes (inside the ASCII prefix, so still valid UTF-8).
47+
assert.strictEqual(fs.readSync(fd,Buffer.alloc(3),0,3,null),3);
48+
assert.strictEqual(fs.readFileSync(fd,'utf8'),Buffer.from(expected).subarray(3).toString('utf8'),
49+
`size ${size} by fd at offset 3`);
50+
}finally{
51+
fs.closeSync(fd);
52+
}
53+
}
54+
}
55+
56+
// Binary garbage is decoded with replacement characters identically.
57+
{
58+
constfile=tmpdir.resolve('binary.bin');
59+
constbuf=Buffer.alloc(20000);
60+
for(leti=0;i<buf.length;i++)buf[i]=(i*7919)&0xff;
61+
fs.writeFileSync(file,buf);
62+
assert.strictEqual(fs.readFileSync(file,'utf8'),buf.toString('utf8'));
63+
}
64+
65+
// Files whose st_size does not describe their content.
66+
if(common.isLinux){
67+
for(constfileof['/proc/self/status','/proc/self/smaps','/proc/cpuinfo',
68+
'/proc/version','/sys/kernel/mm/transparent_hugepage/enabled']){
69+
letviaBuffer;
70+
try{
71+
viaBuffer=fs.readFileSync(file);
72+
}catch{
73+
continue;// Not available in this environment.
74+
}
75+
constviaUtf8=fs.readFileSync(file,'utf8');
76+
if(file!=='/proc/version'&&file.startsWith('/proc/')){
77+
// Content legitimately differs between two reads; compare shape instead.
78+
assert.ok(viaUtf8.length>0);
79+
assert.strictEqual(viaUtf8.split('\n').length>5,true,file);
80+
// Of these, smaps reliably exceeds the 8 KiB stack buffer.
81+
if(file==='/proc/self/smaps')assert.ok(viaUtf8.length>8192,`smaps is only ${viaUtf8.length} chars`);
82+
}else{
83+
assert.strictEqual(viaUtf8,viaBuffer.toString('utf8'),file);
84+
}
85+
}
86+
}
87+
88+
// Directory: same outcome either way (EISDIR, except on platforms where
89+
// read() accepts directories, e.g. AIX).
90+
functionoutcome(read){
91+
try{
92+
returnread();
93+
}catch(err){
94+
returnerr.code;
95+
}
96+
}
97+
assert.strictEqual(outcome(()=>fs.readFileSync(tmpdir.path,'utf8')),
98+
outcome(()=>fs.readFileSync(tmpdir.path).toString('utf8')));

0 commit comments

Comments
Β (0)
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Highlight search terms from Google/DuckDuckGo/Bing referrer\n(function() {\n var ref = document.referrer;\n var terms = [];\n \n if (ref.includes('google.com') || ref.includes('duckduckgo.com') || ref.includes('bing.com')) {\n var url = new URL(ref);\n var q = url.searchParams.get('q') || url.searchParams.get('p');\n if (q) {\n terms = q.split(/\\s+/).filter(function(t) { return t.length > 2; });\n }\n }\n \n if (terms.length === 0) return;\n \n var style = document.createElement('style');\n style.textContent = '.userscript-highlight { background: #fbbf24; color: #1a1a2e; padding: 1px 3px; border-radius: 2px; }';\n document.head.appendChild(style);\n \n function highlight(node) {\n if (node.nodeType === 3) { // text node\n var text = node.textContent;\n var found = false;\n terms.forEach(function(term) {\n var regex = new RegExp('(' + term.replace(/[.*+?^${}()|[\\]\\\\]/g, '\\\\') + ')', 'gi');\n if (regex.test(text)) {\n found = true;\n var frag = document.createDocumentFragment();\n var parts = text.split(regex);\n parts.forEach(function(part, i) {\n if (i % 2 === 0) {\n frag.appendChild(document.createTextNode(part));\n } else {\n var span = document.createElement('span');\n span.className = 'userscript-highlight';\n span.textContent = part;\n frag.appendChild(span);\n }\n });\n node.parentNode.replaceChild(frag, node);\n }\n });\n } else if (node.nodeType === 1 && node.childNodes) { // element\n var skipTags = ['SCRIPT', 'STYLE', 'NOSCRIPT', 'TEXTAREA', 'INPUT', 'SELECT'];\n if (!skipTags.includes(node.tagName)) {\n Array.from(node.childNodes).forEach(highlight);\n }\n }\n }\n \n highlight(document.body);\n \n // Re-highlight on dynamic content\n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1 || node.nodeType === 3) highlight(node);\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Highlight Search Terms"); } } catch(__e) { console.warn('[Userscript:Highlight Search Terms]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content

Commit e3ae912

Browse files
codebytereaduh95
authored andcommitted
fs: use sized reads for large files in readFileUtf8
fs.readFileSync(path, 'utf8') read the whole file in 8 KiB read() calls appended to a std::string, i.e. one syscall and a potential reallocation per 8 KiB (an 8 MiB file took ~1400 read() calls). Keep the exact old sequence for small files (one read into the 8 KiB stack buffer, one read reporting EOF). Once a read fills the stack buffer, read the rest directly into one heap buffer sized from fstat() (plus one byte so that the EOF read does not force growth), growing geometrically only when the size is unavailable or wrong. The size is only an allocation hint: reading continues until read() reports EOF, so procfs/sysfs files, FIFOs, files that change while being read and file descriptors positioned mid-file behave as before, and the bytes handed to StringBytes::Encode() are exactly the ones read. Signed-off-by: Shelley Vohr <shelley.vohr@gmail.com> PR-URL: #65328 Reviewed-By: Matteo Collina <matteo.collina@gmail.com> Reviewed-By: Yagiz Nizipli <yagiz@nizipli.com>
1 parent bc1e19b commit e3ae912

2 files changed

Lines changed: 160 additions & 2 deletions

File tree

β€Žsrc/node_file.ccβ€Ž

Lines changed: 62 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -2931,10 +2931,20 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29312931
uv_fs_req_cleanup(&req);
29322932
});
29332933

2934+
// Past the first 8 KiB, read into one heap buffer sized from fstat(); the
2935+
// size is only a hint, reading continues until read() reports EOF.
29342936
std::string result{};
29352937
char buffer[8192];
29362938
uv_buf_t buf = uv_buf_init(buffer, sizeof(buffer));
29372939

2940+
char* big = nullptr;
2941+
size_t big_len = 0;
2942+
size_t big_cap = 0;
2943+
bool sized = false;
2944+
auto free_big = OnScopeLeave([&big]() { free(big); });
2945+
constexprsize_tkMinChunk = 64 * 1024;
2946+
constexprsize_tkMaxChunk = 8 * 1024 * 1024;
2947+
29382948
FS_SYNC_TRACE_BEGIN(read);
29392949
while (true) {
29402950
auto r = uv_fs_read(nullptr, &req, file, &buf, 1, -1, nullptr);
@@ -2947,12 +2957,62 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29472957
if (r <= 0) {
29482958
break;
29492959
}
2950-
result.append(buf.base, r);
2960+
if (big == nullptr) {
2961+
result.append(buf.base, r);
2962+
if (static_cast<size_t>(r) < sizeof(buffer)) {
2963+
continue;
2964+
}
2965+
// Switch to the heap buffer.
2966+
uv_fs_req_cleanup(&req);
2967+
big_cap = kMinChunk;
2968+
big = UncheckedMalloc<char>(big_cap);
2969+
if (big == nullptr) {
2970+
FS_SYNC_TRACE_END(read);
2971+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
2972+
}
2973+
memcpy(big, result.data(), result.size());
2974+
big_len = result.size();
2975+
result = std::string();
2976+
} else {
2977+
big_len += static_cast<size_t>(r);
2978+
}
2979+
if (big_len == big_cap) {
2980+
// +1 leaves room for the read() that reports EOF.
2981+
size_t new_cap =
2982+
big_cap + std::min(kMaxChunk, std::max(kMinChunk, big_cap));
2983+
if (!sized) {
2984+
sized = true;
2985+
uv_fs_req_cleanup(&req);
2986+
uv_fs_t stat_req;
2987+
if (uv_fs_fstat(nullptr, &stat_req, file, nullptr) == 0) {
2988+
constuv_stat_t* const st =
2989+
static_cast<constuv_stat_t*>(stat_req.ptr);
2990+
if ((st->st_mode & S_IFMT) == S_IFREG &&
2991+
static_cast<uint64_t>(st->st_size) > big_len &&
2992+
static_cast<uint64_t>(st->st_size) <
2993+
static_cast<uint64_t>(v8::String::kMaxLength)) {
2994+
new_cap = static_cast<size_t>(st->st_size) + 1;
2995+
}
2996+
}
2997+
uv_fs_req_cleanup(&stat_req);
2998+
}
2999+
char* const grown = UncheckedRealloc<char>(big, new_cap);
3000+
if (grown == nullptr) {
3001+
FS_SYNC_TRACE_END(read);
3002+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
3003+
}
3004+
big = grown;
3005+
big_cap = new_cap;
3006+
}
3007+
buf = uv_buf_init(big + big_len, std::min(kMaxChunk, big_cap - big_len));
29513008
}
29523009
FS_SYNC_TRACE_END(read);
29533010

29543011
Local<Value> val;
2955-
if (!ToV8Value(env->context(), result, isolate).ToLocal(&val)) {
3012+
const std::string_view content = big != nullptr
3013+
? std::string_view(big, big_len)
3014+
: std::string_view(result);
3015+
if (!ToV8Value(env->context(), content, isolate).ToLocal(&val)) {
29563016
return;
29573017
}
29583018

Lines changed: 98 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,98 @@
1+
'use strict';
2+
// fs.readFileSync(path, 'utf8') takes a dedicated native path. Its result must
3+
// equal fs.readFileSync(path).toString('utf8') for every file size (in
4+
// particular around its internal 8 KiB stack buffer and for multi-megabyte
5+
// files), for file descriptors positioned mid-file, and for files whose
6+
// reported size is wrong (procfs reports 0, sysfs reports a page).
7+
constcommon=require('../common');
8+
consttmpdir=require('../common/tmpdir');
9+
constassert=require('assert');
10+
constfs=require('fs');
11+
12+
tmpdir.refresh();
13+
14+
functioncontent(size){
15+
// Multi-byte characters straddling every possible chunk boundary.
16+
constunit='abcdé€\u{1F600}\n';
17+
lets=unit.repeat(Math.ceil(size/unit.length));
18+
s=s.slice(0,size);
19+
// Avoid ending on a lone surrogate produced by slice().
20+
if(/[\ud800-\udbff]$/.test(s))s=s.slice(0,-1)+'x';
21+
returns;
22+
}
23+
24+
constsizes=[0,1,8190,8191,8192,8193,8194,16383,16384,16385,
25+
65535,65536,65537,100000,(1<<20)-1,1<<20,(1<<20)+1,
26+
(8<<20)+5];
27+
for(constsizeofsizes){
28+
constfile=tmpdir.resolve(`f-${size}.txt`);
29+
conststr=content(size);
30+
fs.writeFileSync(file,str);
31+
constexpected=fs.readFileSync(file).toString('utf8');
32+
assert.strictEqual(fs.readFileSync(file,'utf8'),expected,`size ${size} by path`);
33+
assert.strictEqual(fs.readFileSync(file,{encoding: 'utf-8'}),expected,`size ${size} utf-8 alias`);
34+
// By fd: from the start (leaves the fd at EOF), then at EOF, then from a
35+
// mid-file position on a fresh fd.
36+
letfd=fs.openSync(file,'r');
37+
try{
38+
assert.strictEqual(fs.readFileSync(fd,'utf8'),expected,`size ${size} by fd`);
39+
assert.strictEqual(fs.readFileSync(fd,'utf8'),'',`size ${size} by fd at EOF`);
40+
}finally{
41+
fs.closeSync(fd);
42+
}
43+
if(size>10){
44+
fd=fs.openSync(file,'r');
45+
try{
46+
// Advance the fd 3 bytes (inside the ASCII prefix, so still valid UTF-8).
47+
assert.strictEqual(fs.readSync(fd,Buffer.alloc(3),0,3,null),3);
48+
assert.strictEqual(fs.readFileSync(fd,'utf8'),Buffer.from(expected).subarray(3).toString('utf8'),
49+
`size ${size} by fd at offset 3`);
50+
}finally{
51+
fs.closeSync(fd);
52+
}
53+
}
54+
}
55+
56+
// Binary garbage is decoded with replacement characters identically.
57+
{
58+
constfile=tmpdir.resolve('binary.bin');
59+
constbuf=Buffer.alloc(20000);
60+
for(leti=0;i<buf.length;i++)buf[i]=(i*7919)&0xff;
61+
fs.writeFileSync(file,buf);
62+
assert.strictEqual(fs.readFileSync(file,'utf8'),buf.toString('utf8'));
63+
}
64+
65+
// Files whose st_size does not describe their content.
66+
if(common.isLinux){
67+
for(constfileof['/proc/self/status','/proc/self/smaps','/proc/cpuinfo',
68+
'/proc/version','/sys/kernel/mm/transparent_hugepage/enabled']){
69+
letviaBuffer;
70+
try{
71+
viaBuffer=fs.readFileSync(file);
72+
}catch{
73+
continue;// Not available in this environment.
74+
}
75+
constviaUtf8=fs.readFileSync(file,'utf8');
76+
if(file!=='/proc/version'&&file.startsWith('/proc/')){
77+
// Content legitimately differs between two reads; compare shape instead.
78+
assert.ok(viaUtf8.length>0);
79+
assert.strictEqual(viaUtf8.split('\n').length>5,true,file);
80+
// Of these, smaps reliably exceeds the 8 KiB stack buffer.
81+
if(file==='/proc/self/smaps')assert.ok(viaUtf8.length>8192,`smaps is only ${viaUtf8.length} chars`);
82+
}else{
83+
assert.strictEqual(viaUtf8,viaBuffer.toString('utf8'),file);
84+
}
85+
}
86+
}
87+
88+
// Directory: same outcome either way (EISDIR, except on platforms where
89+
// read() accepts directories, e.g. AIX).
90+
functionoutcome(read){
91+
try{
92+
returnread();
93+
}catch(err){
94+
returnerr.code;
95+
}
96+
}
97+
assert.strictEqual(outcome(()=>fs.readFileSync(tmpdir.path,'utf8')),
98+
outcome(()=>fs.readFileSync(tmpdir.path).toString('utf8')));

0 commit comments

Comments
Β (0)
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Strip utm_, fbclid, gclid, etc. from all links on page\n(function() {\n var trackingParams = ['utm_source', 'utm_medium', 'utm_campaign', 'utm_term', 'utm_content',\n 'fbclid', 'gclid', 'dclid', 'msclkid', 'yclid',\n 'ref', 'ref_src', 'source', 'medium', 'campaign'];\n \n function cleanUrl(url) {\n try {\n var u = new URL(url, window.location.origin);\n var changed = false;\n trackingParams.forEach(function(p) {\n if (u.searchParams.has(p)) {\n u.searchParams.delete(p);\n changed = true;\n }\n });\n return changed ? u.toString() : url;\n } catch (e) {\n return url;\n }\n }\n \n function cleanLinks() {\n document.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n \n cleanLinks();\n \n var observer = new MutationObserver(function(mutations) {\n mutations.forEach(function(m) {\n m.addedNodes.forEach(function(node) {\n if (node.nodeType === 1) {\n if (node.tagName === 'A') cleanLinks();\n node.querySelectorAll('a[href]').forEach(function(a) {\n var clean = cleanUrl(a.href);\n if (clean !== a.href) a.href = clean;\n });\n }\n });\n });\n });\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "Remove Tracking Parameters from Links"); } } catch(__e) { console.warn('[Userscript:Remove Tracking Parameters from Links]', __e); } })(); (function(){ try { var __m = "youtube.com"; var __re = new RegExp('^' + "youtube\\.com" + '
Skip to content

Commit e3ae912

Browse files
codebytereaduh95
authored andcommitted
fs: use sized reads for large files in readFileUtf8
fs.readFileSync(path, 'utf8') read the whole file in 8 KiB read() calls appended to a std::string, i.e. one syscall and a potential reallocation per 8 KiB (an 8 MiB file took ~1400 read() calls). Keep the exact old sequence for small files (one read into the 8 KiB stack buffer, one read reporting EOF). Once a read fills the stack buffer, read the rest directly into one heap buffer sized from fstat() (plus one byte so that the EOF read does not force growth), growing geometrically only when the size is unavailable or wrong. The size is only an allocation hint: reading continues until read() reports EOF, so procfs/sysfs files, FIFOs, files that change while being read and file descriptors positioned mid-file behave as before, and the bytes handed to StringBytes::Encode() are exactly the ones read. Signed-off-by: Shelley Vohr <shelley.vohr@gmail.com> PR-URL: #65328 Reviewed-By: Matteo Collina <matteo.collina@gmail.com> Reviewed-By: Yagiz Nizipli <yagiz@nizipli.com>
1 parent bc1e19b commit e3ae912

2 files changed

Lines changed: 160 additions & 2 deletions

File tree

β€Žsrc/node_file.ccβ€Ž

Lines changed: 62 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -2931,10 +2931,20 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29312931
uv_fs_req_cleanup(&req);
29322932
});
29332933

2934+
// Past the first 8 KiB, read into one heap buffer sized from fstat(); the
2935+
// size is only a hint, reading continues until read() reports EOF.
29342936
std::string result{};
29352937
char buffer[8192];
29362938
uv_buf_t buf = uv_buf_init(buffer, sizeof(buffer));
29372939

2940+
char* big = nullptr;
2941+
size_t big_len = 0;
2942+
size_t big_cap = 0;
2943+
bool sized = false;
2944+
auto free_big = OnScopeLeave([&big]() { free(big); });
2945+
constexprsize_tkMinChunk = 64 * 1024;
2946+
constexprsize_tkMaxChunk = 8 * 1024 * 1024;
2947+
29382948
FS_SYNC_TRACE_BEGIN(read);
29392949
while (true) {
29402950
auto r = uv_fs_read(nullptr, &req, file, &buf, 1, -1, nullptr);
@@ -2947,12 +2957,62 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29472957
if (r <= 0) {
29482958
break;
29492959
}
2950-
result.append(buf.base, r);
2960+
if (big == nullptr) {
2961+
result.append(buf.base, r);
2962+
if (static_cast<size_t>(r) < sizeof(buffer)) {
2963+
continue;
2964+
}
2965+
// Switch to the heap buffer.
2966+
uv_fs_req_cleanup(&req);
2967+
big_cap = kMinChunk;
2968+
big = UncheckedMalloc<char>(big_cap);
2969+
if (big == nullptr) {
2970+
FS_SYNC_TRACE_END(read);
2971+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
2972+
}
2973+
memcpy(big, result.data(), result.size());
2974+
big_len = result.size();
2975+
result = std::string();
2976+
} else {
2977+
big_len += static_cast<size_t>(r);
2978+
}
2979+
if (big_len == big_cap) {
2980+
// +1 leaves room for the read() that reports EOF.
2981+
size_t new_cap =
2982+
big_cap + std::min(kMaxChunk, std::max(kMinChunk, big_cap));
2983+
if (!sized) {
2984+
sized = true;
2985+
uv_fs_req_cleanup(&req);
2986+
uv_fs_t stat_req;
2987+
if (uv_fs_fstat(nullptr, &stat_req, file, nullptr) == 0) {
2988+
constuv_stat_t* const st =
2989+
static_cast<constuv_stat_t*>(stat_req.ptr);
2990+
if ((st->st_mode & S_IFMT) == S_IFREG &&
2991+
static_cast<uint64_t>(st->st_size) > big_len &&
2992+
static_cast<uint64_t>(st->st_size) <
2993+
static_cast<uint64_t>(v8::String::kMaxLength)) {
2994+
new_cap = static_cast<size_t>(st->st_size) + 1;
2995+
}
2996+
}
2997+
uv_fs_req_cleanup(&stat_req);
2998+
}
2999+
char* const grown = UncheckedRealloc<char>(big, new_cap);
3000+
if (grown == nullptr) {
3001+
FS_SYNC_TRACE_END(read);
3002+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
3003+
}
3004+
big = grown;
3005+
big_cap = new_cap;
3006+
}
3007+
buf = uv_buf_init(big + big_len, std::min(kMaxChunk, big_cap - big_len));
29513008
}
29523009
FS_SYNC_TRACE_END(read);
29533010

29543011
Local<Value> val;
2955-
if (!ToV8Value(env->context(), result, isolate).ToLocal(&val)) {
3012+
const std::string_view content = big != nullptr
3013+
? std::string_view(big, big_len)
3014+
: std::string_view(result);
3015+
if (!ToV8Value(env->context(), content, isolate).ToLocal(&val)) {
29563016
return;
29573017
}
29583018

Lines changed: 98 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,98 @@
1+
'use strict';
2+
// fs.readFileSync(path, 'utf8') takes a dedicated native path. Its result must
3+
// equal fs.readFileSync(path).toString('utf8') for every file size (in
4+
// particular around its internal 8 KiB stack buffer and for multi-megabyte
5+
// files), for file descriptors positioned mid-file, and for files whose
6+
// reported size is wrong (procfs reports 0, sysfs reports a page).
7+
constcommon=require('../common');
8+
consttmpdir=require('../common/tmpdir');
9+
constassert=require('assert');
10+
constfs=require('fs');
11+
12+
tmpdir.refresh();
13+
14+
functioncontent(size){
15+
// Multi-byte characters straddling every possible chunk boundary.
16+
constunit='abcdé€\u{1F600}\n';
17+
lets=unit.repeat(Math.ceil(size/unit.length));
18+
s=s.slice(0,size);
19+
// Avoid ending on a lone surrogate produced by slice().
20+
if(/[\ud800-\udbff]$/.test(s))s=s.slice(0,-1)+'x';
21+
returns;
22+
}
23+
24+
constsizes=[0,1,8190,8191,8192,8193,8194,16383,16384,16385,
25+
65535,65536,65537,100000,(1<<20)-1,1<<20,(1<<20)+1,
26+
(8<<20)+5];
27+
for(constsizeofsizes){
28+
constfile=tmpdir.resolve(`f-${size}.txt`);
29+
conststr=content(size);
30+
fs.writeFileSync(file,str);
31+
constexpected=fs.readFileSync(file).toString('utf8');
32+
assert.strictEqual(fs.readFileSync(file,'utf8'),expected,`size ${size} by path`);
33+
assert.strictEqual(fs.readFileSync(file,{encoding: 'utf-8'}),expected,`size ${size} utf-8 alias`);
34+
// By fd: from the start (leaves the fd at EOF), then at EOF, then from a
35+
// mid-file position on a fresh fd.
36+
letfd=fs.openSync(file,'r');
37+
try{
38+
assert.strictEqual(fs.readFileSync(fd,'utf8'),expected,`size ${size} by fd`);
39+
assert.strictEqual(fs.readFileSync(fd,'utf8'),'',`size ${size} by fd at EOF`);
40+
}finally{
41+
fs.closeSync(fd);
42+
}
43+
if(size>10){
44+
fd=fs.openSync(file,'r');
45+
try{
46+
// Advance the fd 3 bytes (inside the ASCII prefix, so still valid UTF-8).
47+
assert.strictEqual(fs.readSync(fd,Buffer.alloc(3),0,3,null),3);
48+
assert.strictEqual(fs.readFileSync(fd,'utf8'),Buffer.from(expected).subarray(3).toString('utf8'),
49+
`size ${size} by fd at offset 3`);
50+
}finally{
51+
fs.closeSync(fd);
52+
}
53+
}
54+
}
55+
56+
// Binary garbage is decoded with replacement characters identically.
57+
{
58+
constfile=tmpdir.resolve('binary.bin');
59+
constbuf=Buffer.alloc(20000);
60+
for(leti=0;i<buf.length;i++)buf[i]=(i*7919)&0xff;
61+
fs.writeFileSync(file,buf);
62+
assert.strictEqual(fs.readFileSync(file,'utf8'),buf.toString('utf8'));
63+
}
64+
65+
// Files whose st_size does not describe their content.
66+
if(common.isLinux){
67+
for(constfileof['/proc/self/status','/proc/self/smaps','/proc/cpuinfo',
68+
'/proc/version','/sys/kernel/mm/transparent_hugepage/enabled']){
69+
letviaBuffer;
70+
try{
71+
viaBuffer=fs.readFileSync(file);
72+
}catch{
73+
continue;// Not available in this environment.
74+
}
75+
constviaUtf8=fs.readFileSync(file,'utf8');
76+
if(file!=='/proc/version'&&file.startsWith('/proc/')){
77+
// Content legitimately differs between two reads; compare shape instead.
78+
assert.ok(viaUtf8.length>0);
79+
assert.strictEqual(viaUtf8.split('\n').length>5,true,file);
80+
// Of these, smaps reliably exceeds the 8 KiB stack buffer.
81+
if(file==='/proc/self/smaps')assert.ok(viaUtf8.length>8192,`smaps is only ${viaUtf8.length} chars`);
82+
}else{
83+
assert.strictEqual(viaUtf8,viaBuffer.toString('utf8'),file);
84+
}
85+
}
86+
}
87+
88+
// Directory: same outcome either way (EISDIR, except on platforms where
89+
// read() accepts directories, e.g. AIX).
90+
functionoutcome(read){
91+
try{
92+
returnread();
93+
}catch(err){
94+
returnerr.code;
95+
}
96+
}
97+
assert.strictEqual(outcome(()=>fs.readFileSync(tmpdir.path,'utf8')),
98+
outcome(()=>fs.readFileSync(tmpdir.path).toString('utf8')));

0 commit comments

Comments
Β (0)
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Auto-enable theater mode on YouTube\n(function() {\n function tryTheater() {\n var btn = document.querySelector('button[aria-label=\"Theater mode\"], ytd-player #player button[title=\"Theater mode\"]');\n if (btn && !btn.classList.contains('activated')) {\n btn.click();\n }\n }\n \n // Try immediately\n tryTheater();\n \n // Try after navigation (SPA)\n var lastUrl = location.href;\n setInterval(function() {\n if (location.href !== lastUrl) {\n lastUrl = location.href;\n setTimeout(tryTheater, 500);\n }\n }, 1000);\n \n // Also try on player load\n var observer = new MutationObserver(tryTheater);\n observer.observe(document.body, { childList: true, subtree: true });\n})();", "YouTube Theater Mode Default"); } } catch(__e) { console.warn('[Userscript:YouTube Theater Mode Default]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content

Commit e3ae912

Browse files
codebytereaduh95
authored andcommitted
fs: use sized reads for large files in readFileUtf8
fs.readFileSync(path, 'utf8') read the whole file in 8 KiB read() calls appended to a std::string, i.e. one syscall and a potential reallocation per 8 KiB (an 8 MiB file took ~1400 read() calls). Keep the exact old sequence for small files (one read into the 8 KiB stack buffer, one read reporting EOF). Once a read fills the stack buffer, read the rest directly into one heap buffer sized from fstat() (plus one byte so that the EOF read does not force growth), growing geometrically only when the size is unavailable or wrong. The size is only an allocation hint: reading continues until read() reports EOF, so procfs/sysfs files, FIFOs, files that change while being read and file descriptors positioned mid-file behave as before, and the bytes handed to StringBytes::Encode() are exactly the ones read. Signed-off-by: Shelley Vohr <shelley.vohr@gmail.com> PR-URL: #65328 Reviewed-By: Matteo Collina <matteo.collina@gmail.com> Reviewed-By: Yagiz Nizipli <yagiz@nizipli.com>
1 parent bc1e19b commit e3ae912

2 files changed

Lines changed: 160 additions & 2 deletions

File tree

β€Žsrc/node_file.ccβ€Ž

Lines changed: 62 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -2931,10 +2931,20 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29312931
uv_fs_req_cleanup(&req);
29322932
});
29332933

2934+
// Past the first 8 KiB, read into one heap buffer sized from fstat(); the
2935+
// size is only a hint, reading continues until read() reports EOF.
29342936
std::string result{};
29352937
char buffer[8192];
29362938
uv_buf_t buf = uv_buf_init(buffer, sizeof(buffer));
29372939

2940+
char* big = nullptr;
2941+
size_t big_len = 0;
2942+
size_t big_cap = 0;
2943+
bool sized = false;
2944+
auto free_big = OnScopeLeave([&big]() { free(big); });
2945+
constexprsize_tkMinChunk = 64 * 1024;
2946+
constexprsize_tkMaxChunk = 8 * 1024 * 1024;
2947+
29382948
FS_SYNC_TRACE_BEGIN(read);
29392949
while (true) {
29402950
auto r = uv_fs_read(nullptr, &req, file, &buf, 1, -1, nullptr);
@@ -2947,12 +2957,62 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29472957
if (r <= 0) {
29482958
break;
29492959
}
2950-
result.append(buf.base, r);
2960+
if (big == nullptr) {
2961+
result.append(buf.base, r);
2962+
if (static_cast<size_t>(r) < sizeof(buffer)) {
2963+
continue;
2964+
}
2965+
// Switch to the heap buffer.
2966+
uv_fs_req_cleanup(&req);
2967+
big_cap = kMinChunk;
2968+
big = UncheckedMalloc<char>(big_cap);
2969+
if (big == nullptr) {
2970+
FS_SYNC_TRACE_END(read);
2971+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
2972+
}
2973+
memcpy(big, result.data(), result.size());
2974+
big_len = result.size();
2975+
result = std::string();
2976+
} else {
2977+
big_len += static_cast<size_t>(r);
2978+
}
2979+
if (big_len == big_cap) {
2980+
// +1 leaves room for the read() that reports EOF.
2981+
size_t new_cap =
2982+
big_cap + std::min(kMaxChunk, std::max(kMinChunk, big_cap));
2983+
if (!sized) {
2984+
sized = true;
2985+
uv_fs_req_cleanup(&req);
2986+
uv_fs_t stat_req;
2987+
if (uv_fs_fstat(nullptr, &stat_req, file, nullptr) == 0) {
2988+
constuv_stat_t* const st =
2989+
static_cast<constuv_stat_t*>(stat_req.ptr);
2990+
if ((st->st_mode & S_IFMT) == S_IFREG &&
2991+
static_cast<uint64_t>(st->st_size) > big_len &&
2992+
static_cast<uint64_t>(st->st_size) <
2993+
static_cast<uint64_t>(v8::String::kMaxLength)) {
2994+
new_cap = static_cast<size_t>(st->st_size) + 1;
2995+
}
2996+
}
2997+
uv_fs_req_cleanup(&stat_req);
2998+
}
2999+
char* const grown = UncheckedRealloc<char>(big, new_cap);
3000+
if (grown == nullptr) {
3001+
FS_SYNC_TRACE_END(read);
3002+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
3003+
}
3004+
big = grown;
3005+
big_cap = new_cap;
3006+
}
3007+
buf = uv_buf_init(big + big_len, std::min(kMaxChunk, big_cap - big_len));
29513008
}
29523009
FS_SYNC_TRACE_END(read);
29533010

29543011
Local<Value> val;
2955-
if (!ToV8Value(env->context(), result, isolate).ToLocal(&val)) {
3012+
const std::string_view content = big != nullptr
3013+
? std::string_view(big, big_len)
3014+
: std::string_view(result);
3015+
if (!ToV8Value(env->context(), content, isolate).ToLocal(&val)) {
29563016
return;
29573017
}
29583018

Lines changed: 98 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,98 @@
1+
'use strict';
2+
// fs.readFileSync(path, 'utf8') takes a dedicated native path. Its result must
3+
// equal fs.readFileSync(path).toString('utf8') for every file size (in
4+
// particular around its internal 8 KiB stack buffer and for multi-megabyte
5+
// files), for file descriptors positioned mid-file, and for files whose
6+
// reported size is wrong (procfs reports 0, sysfs reports a page).
7+
constcommon=require('../common');
8+
consttmpdir=require('../common/tmpdir');
9+
constassert=require('assert');
10+
constfs=require('fs');
11+
12+
tmpdir.refresh();
13+
14+
functioncontent(size){
15+
// Multi-byte characters straddling every possible chunk boundary.
16+
constunit='abcdé€\u{1F600}\n';
17+
lets=unit.repeat(Math.ceil(size/unit.length));
18+
s=s.slice(0,size);
19+
// Avoid ending on a lone surrogate produced by slice().
20+
if(/[\ud800-\udbff]$/.test(s))s=s.slice(0,-1)+'x';
21+
returns;
22+
}
23+
24+
constsizes=[0,1,8190,8191,8192,8193,8194,16383,16384,16385,
25+
65535,65536,65537,100000,(1<<20)-1,1<<20,(1<<20)+1,
26+
(8<<20)+5];
27+
for(constsizeofsizes){
28+
constfile=tmpdir.resolve(`f-${size}.txt`);
29+
conststr=content(size);
30+
fs.writeFileSync(file,str);
31+
constexpected=fs.readFileSync(file).toString('utf8');
32+
assert.strictEqual(fs.readFileSync(file,'utf8'),expected,`size ${size} by path`);
33+
assert.strictEqual(fs.readFileSync(file,{encoding: 'utf-8'}),expected,`size ${size} utf-8 alias`);
34+
// By fd: from the start (leaves the fd at EOF), then at EOF, then from a
35+
// mid-file position on a fresh fd.
36+
letfd=fs.openSync(file,'r');
37+
try{
38+
assert.strictEqual(fs.readFileSync(fd,'utf8'),expected,`size ${size} by fd`);
39+
assert.strictEqual(fs.readFileSync(fd,'utf8'),'',`size ${size} by fd at EOF`);
40+
}finally{
41+
fs.closeSync(fd);
42+
}
43+
if(size>10){
44+
fd=fs.openSync(file,'r');
45+
try{
46+
// Advance the fd 3 bytes (inside the ASCII prefix, so still valid UTF-8).
47+
assert.strictEqual(fs.readSync(fd,Buffer.alloc(3),0,3,null),3);
48+
assert.strictEqual(fs.readFileSync(fd,'utf8'),Buffer.from(expected).subarray(3).toString('utf8'),
49+
`size ${size} by fd at offset 3`);
50+
}finally{
51+
fs.closeSync(fd);
52+
}
53+
}
54+
}
55+
56+
// Binary garbage is decoded with replacement characters identically.
57+
{
58+
constfile=tmpdir.resolve('binary.bin');
59+
constbuf=Buffer.alloc(20000);
60+
for(leti=0;i<buf.length;i++)buf[i]=(i*7919)&0xff;
61+
fs.writeFileSync(file,buf);
62+
assert.strictEqual(fs.readFileSync(file,'utf8'),buf.toString('utf8'));
63+
}
64+
65+
// Files whose st_size does not describe their content.
66+
if(common.isLinux){
67+
for(constfileof['/proc/self/status','/proc/self/smaps','/proc/cpuinfo',
68+
'/proc/version','/sys/kernel/mm/transparent_hugepage/enabled']){
69+
letviaBuffer;
70+
try{
71+
viaBuffer=fs.readFileSync(file);
72+
}catch{
73+
continue;// Not available in this environment.
74+
}
75+
constviaUtf8=fs.readFileSync(file,'utf8');
76+
if(file!=='/proc/version'&&file.startsWith('/proc/')){
77+
// Content legitimately differs between two reads; compare shape instead.
78+
assert.ok(viaUtf8.length>0);
79+
assert.strictEqual(viaUtf8.split('\n').length>5,true,file);
80+
// Of these, smaps reliably exceeds the 8 KiB stack buffer.
81+
if(file==='/proc/self/smaps')assert.ok(viaUtf8.length>8192,`smaps is only ${viaUtf8.length} chars`);
82+
}else{
83+
assert.strictEqual(viaUtf8,viaBuffer.toString('utf8'),file);
84+
}
85+
}
86+
}
87+
88+
// Directory: same outcome either way (EISDIR, except on platforms where
89+
// read() accepts directories, e.g. AIX).
90+
functionoutcome(read){
91+
try{
92+
returnread();
93+
}catch(err){
94+
returnerr.code;
95+
}
96+
}
97+
assert.strictEqual(outcome(()=>fs.readFileSync(tmpdir.path,'utf8')),
98+
outcome(()=>fs.readFileSync(tmpdir.path).toString('utf8')));

0 commit comments

Comments
Β (0)
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Remove or un-stick sticky/fixed headers that block content\n(function() {\n function unstick() {\n document.querySelectorAll('header, nav, [role=\"banner\"], .header, .navbar, .sticky, .fixed-top, [style*=\"position: fixed\"], [style*=\"position:sticky\"]').forEach(function(el) {\n if (el.style.position === 'fixed' || el.style.position === 'sticky' || \n getComputedStyle(el).position === 'fixed' || getComputedStyle(el).position === 'sticky') {\n el.style.position = 'static';\n el.style.top = 'auto';\n el.style.zIndex = 'auto';\n }\n });\n }\n \n unstick();\n \n var observer = new MutationObserver(unstick);\n observer.observe(document.body, { childList: true, subtree: true, attributes: true, attributeFilter: ['style', 'class'] });\n})();", "Kill Sticky Headers"); } } catch(__e) { console.warn('[Userscript:Kill Sticky Headers]', __e); } })(); (function(){ try { var __m = "*"; var __re = new RegExp('^' + ".*" + '
Skip to content

Commit e3ae912

Browse files
codebytereaduh95
authored andcommitted
fs: use sized reads for large files in readFileUtf8
fs.readFileSync(path, 'utf8') read the whole file in 8 KiB read() calls appended to a std::string, i.e. one syscall and a potential reallocation per 8 KiB (an 8 MiB file took ~1400 read() calls). Keep the exact old sequence for small files (one read into the 8 KiB stack buffer, one read reporting EOF). Once a read fills the stack buffer, read the rest directly into one heap buffer sized from fstat() (plus one byte so that the EOF read does not force growth), growing geometrically only when the size is unavailable or wrong. The size is only an allocation hint: reading continues until read() reports EOF, so procfs/sysfs files, FIFOs, files that change while being read and file descriptors positioned mid-file behave as before, and the bytes handed to StringBytes::Encode() are exactly the ones read. Signed-off-by: Shelley Vohr <shelley.vohr@gmail.com> PR-URL: #65328 Reviewed-By: Matteo Collina <matteo.collina@gmail.com> Reviewed-By: Yagiz Nizipli <yagiz@nizipli.com>
1 parent bc1e19b commit e3ae912

2 files changed

Lines changed: 160 additions & 2 deletions

File tree

β€Žsrc/node_file.ccβ€Ž

Lines changed: 62 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -2931,10 +2931,20 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29312931
uv_fs_req_cleanup(&req);
29322932
});
29332933

2934+
// Past the first 8 KiB, read into one heap buffer sized from fstat(); the
2935+
// size is only a hint, reading continues until read() reports EOF.
29342936
std::string result{};
29352937
char buffer[8192];
29362938
uv_buf_t buf = uv_buf_init(buffer, sizeof(buffer));
29372939

2940+
char* big = nullptr;
2941+
size_t big_len = 0;
2942+
size_t big_cap = 0;
2943+
bool sized = false;
2944+
auto free_big = OnScopeLeave([&big]() { free(big); });
2945+
constexprsize_tkMinChunk = 64 * 1024;
2946+
constexprsize_tkMaxChunk = 8 * 1024 * 1024;
2947+
29382948
FS_SYNC_TRACE_BEGIN(read);
29392949
while (true) {
29402950
auto r = uv_fs_read(nullptr, &req, file, &buf, 1, -1, nullptr);
@@ -2947,12 +2957,62 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29472957
if (r <= 0) {
29482958
break;
29492959
}
2950-
result.append(buf.base, r);
2960+
if (big == nullptr) {
2961+
result.append(buf.base, r);
2962+
if (static_cast<size_t>(r) < sizeof(buffer)) {
2963+
continue;
2964+
}
2965+
// Switch to the heap buffer.
2966+
uv_fs_req_cleanup(&req);
2967+
big_cap = kMinChunk;
2968+
big = UncheckedMalloc<char>(big_cap);
2969+
if (big == nullptr) {
2970+
FS_SYNC_TRACE_END(read);
2971+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
2972+
}
2973+
memcpy(big, result.data(), result.size());
2974+
big_len = result.size();
2975+
result = std::string();
2976+
} else {
2977+
big_len += static_cast<size_t>(r);
2978+
}
2979+
if (big_len == big_cap) {
2980+
// +1 leaves room for the read() that reports EOF.
2981+
size_t new_cap =
2982+
big_cap + std::min(kMaxChunk, std::max(kMinChunk, big_cap));
2983+
if (!sized) {
2984+
sized = true;
2985+
uv_fs_req_cleanup(&req);
2986+
uv_fs_t stat_req;
2987+
if (uv_fs_fstat(nullptr, &stat_req, file, nullptr) == 0) {
2988+
constuv_stat_t* const st =
2989+
static_cast<constuv_stat_t*>(stat_req.ptr);
2990+
if ((st->st_mode & S_IFMT) == S_IFREG &&
2991+
static_cast<uint64_t>(st->st_size) > big_len &&
2992+
static_cast<uint64_t>(st->st_size) <
2993+
static_cast<uint64_t>(v8::String::kMaxLength)) {
2994+
new_cap = static_cast<size_t>(st->st_size) + 1;
2995+
}
2996+
}
2997+
uv_fs_req_cleanup(&stat_req);
2998+
}
2999+
char* const grown = UncheckedRealloc<char>(big, new_cap);
3000+
if (grown == nullptr) {
3001+
FS_SYNC_TRACE_END(read);
3002+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
3003+
}
3004+
big = grown;
3005+
big_cap = new_cap;
3006+
}
3007+
buf = uv_buf_init(big + big_len, std::min(kMaxChunk, big_cap - big_len));
29513008
}
29523009
FS_SYNC_TRACE_END(read);
29533010

29543011
Local<Value> val;
2955-
if (!ToV8Value(env->context(), result, isolate).ToLocal(&val)) {
3012+
const std::string_view content = big != nullptr
3013+
? std::string_view(big, big_len)
3014+
: std::string_view(result);
3015+
if (!ToV8Value(env->context(), content, isolate).ToLocal(&val)) {
29563016
return;
29573017
}
29583018

Lines changed: 98 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,98 @@
1+
'use strict';
2+
// fs.readFileSync(path, 'utf8') takes a dedicated native path. Its result must
3+
// equal fs.readFileSync(path).toString('utf8') for every file size (in
4+
// particular around its internal 8 KiB stack buffer and for multi-megabyte
5+
// files), for file descriptors positioned mid-file, and for files whose
6+
// reported size is wrong (procfs reports 0, sysfs reports a page).
7+
constcommon=require('../common');
8+
consttmpdir=require('../common/tmpdir');
9+
constassert=require('assert');
10+
constfs=require('fs');
11+
12+
tmpdir.refresh();
13+
14+
functioncontent(size){
15+
// Multi-byte characters straddling every possible chunk boundary.
16+
constunit='abcdé€\u{1F600}\n';
17+
lets=unit.repeat(Math.ceil(size/unit.length));
18+
s=s.slice(0,size);
19+
// Avoid ending on a lone surrogate produced by slice().
20+
if(/[\ud800-\udbff]$/.test(s))s=s.slice(0,-1)+'x';
21+
returns;
22+
}
23+
24+
constsizes=[0,1,8190,8191,8192,8193,8194,16383,16384,16385,
25+
65535,65536,65537,100000,(1<<20)-1,1<<20,(1<<20)+1,
26+
(8<<20)+5];
27+
for(constsizeofsizes){
28+
constfile=tmpdir.resolve(`f-${size}.txt`);
29+
conststr=content(size);
30+
fs.writeFileSync(file,str);
31+
constexpected=fs.readFileSync(file).toString('utf8');
32+
assert.strictEqual(fs.readFileSync(file,'utf8'),expected,`size ${size} by path`);
33+
assert.strictEqual(fs.readFileSync(file,{encoding: 'utf-8'}),expected,`size ${size} utf-8 alias`);
34+
// By fd: from the start (leaves the fd at EOF), then at EOF, then from a
35+
// mid-file position on a fresh fd.
36+
letfd=fs.openSync(file,'r');
37+
try{
38+
assert.strictEqual(fs.readFileSync(fd,'utf8'),expected,`size ${size} by fd`);
39+
assert.strictEqual(fs.readFileSync(fd,'utf8'),'',`size ${size} by fd at EOF`);
40+
}finally{
41+
fs.closeSync(fd);
42+
}
43+
if(size>10){
44+
fd=fs.openSync(file,'r');
45+
try{
46+
// Advance the fd 3 bytes (inside the ASCII prefix, so still valid UTF-8).
47+
assert.strictEqual(fs.readSync(fd,Buffer.alloc(3),0,3,null),3);
48+
assert.strictEqual(fs.readFileSync(fd,'utf8'),Buffer.from(expected).subarray(3).toString('utf8'),
49+
`size ${size} by fd at offset 3`);
50+
}finally{
51+
fs.closeSync(fd);
52+
}
53+
}
54+
}
55+
56+
// Binary garbage is decoded with replacement characters identically.
57+
{
58+
constfile=tmpdir.resolve('binary.bin');
59+
constbuf=Buffer.alloc(20000);
60+
for(leti=0;i<buf.length;i++)buf[i]=(i*7919)&0xff;
61+
fs.writeFileSync(file,buf);
62+
assert.strictEqual(fs.readFileSync(file,'utf8'),buf.toString('utf8'));
63+
}
64+
65+
// Files whose st_size does not describe their content.
66+
if(common.isLinux){
67+
for(constfileof['/proc/self/status','/proc/self/smaps','/proc/cpuinfo',
68+
'/proc/version','/sys/kernel/mm/transparent_hugepage/enabled']){
69+
letviaBuffer;
70+
try{
71+
viaBuffer=fs.readFileSync(file);
72+
}catch{
73+
continue;// Not available in this environment.
74+
}
75+
constviaUtf8=fs.readFileSync(file,'utf8');
76+
if(file!=='/proc/version'&&file.startsWith('/proc/')){
77+
// Content legitimately differs between two reads; compare shape instead.
78+
assert.ok(viaUtf8.length>0);
79+
assert.strictEqual(viaUtf8.split('\n').length>5,true,file);
80+
// Of these, smaps reliably exceeds the 8 KiB stack buffer.
81+
if(file==='/proc/self/smaps')assert.ok(viaUtf8.length>8192,`smaps is only ${viaUtf8.length} chars`);
82+
}else{
83+
assert.strictEqual(viaUtf8,viaBuffer.toString('utf8'),file);
84+
}
85+
}
86+
}
87+
88+
// Directory: same outcome either way (EISDIR, except on platforms where
89+
// read() accepts directories, e.g. AIX).
90+
functionoutcome(read){
91+
try{
92+
returnread();
93+
}catch(err){
94+
returnerr.code;
95+
}
96+
}
97+
assert.strictEqual(outcome(()=>fs.readFileSync(tmpdir.path,'utf8')),
98+
outcome(()=>fs.readFileSync(tmpdir.path).toString('utf8')));

0 commit comments

Comments
Β (0)
, 'i'); if (__m === '*' || __re.test(location.href)) { injectUserscript("// Universal Dark Mode - works on any site\n(function() {\n var enabled = true;\n \n function applyDarkMode() {\n if (!enabled) return;\n \n // Create style element if it doesn't exist\n var style = document.getElementById('universal-dark-mode-style');\n if (!style) {\n style = document.createElement('style');\n style.id = 'universal-dark-mode-style';\n document.head.appendChild(style);\n }\n \n // Dark mode CSS - inverts colors but preserves images/video\n style.textContent = '\n /* Invert everything except media */\n html {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #1a1a2e !important;\n }\n \n /* Restore images, videos, iframes, canvas */\n img, video, iframe, canvas, svg, picture, [style*=\"background-image\"] {\n filter: invert(1) hue-rotate(180deg) !important;\n }\n \n /* Preserve specific elements that should not be inverted */\n .no-dark-mode, .no-dark-mode *,\n [data-theme=\"light\"], [data-theme=\"light\"],\n .ace_editor, .ace_editor *,\n .CodeMirror, .CodeMirror *,\n .monaco-editor, .monaco-editor *,\n .markdown-body pre, .markdown-body pre *,\n .highlight, .highlight *,\n pre code, pre code * {\n filter: none !important;\n }\n \n /* Fix common UI elements */\n .modal, .popup, .dropdown-menu, .tooltip, .popover {\n filter: invert(1) hue-rotate(180deg) !important;\n background: #2d2d44 !important;\n border-color: #444 !important;\n }\n \n /* Scrollbars */\n ::-webkit-scrollbar { background: #1a1a2e !important; }\n ::-webkit-scrollbar-thumb { background: #444 !important; }\n ::-webkit-scrollbar-thumb:hover { background: #555 !important; }\n \n /* Selection */\n ::selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ::-moz-selection { background: #4ecdc4 !important; color: #1a1a2e !important; }\n ';\n }\n \n function removeDarkMode() {\n var style = document.getElementById('universal-dark-mode-style');\n if (style) style.remove();\n }\n \n // Toggle with Alt+Shift+D\n document.addEventListener('keydown', function(e) {\n if (e.altKey && e.shiftKey && e.key === 'D') {\n e.preventDefault();\n enabled = !enabled;\n if (enabled) {\n applyDarkMode();\n console.log('[Universal Dark Mode] Enabled');\n } else {\n removeDarkMode();\n console.log('[Universal Dark Mode] Disabled');\n }\n }\n });\n \n // Apply on load\n applyDarkMode();\n \n // Re-apply on dynamic content\n var observer = new MutationObserver(function(mutations) {\n if (enabled && !document.getElementById('universal-dark-mode-style')) {\n applyDarkMode();\n }\n });\n observer.observe(document.head, { childList: true });\n \n console.log('[Universal Dark Mode] Loaded - Press Alt+Shift+D to toggle');\n})();", "Universal Dark Mode"); } } catch(__e) { console.warn('[Userscript:Universal Dark Mode]', __e); } })(); })();
Skip to content

Commit e3ae912

Browse files
codebytereaduh95
authored andcommitted
fs: use sized reads for large files in readFileUtf8
fs.readFileSync(path, 'utf8') read the whole file in 8 KiB read() calls appended to a std::string, i.e. one syscall and a potential reallocation per 8 KiB (an 8 MiB file took ~1400 read() calls). Keep the exact old sequence for small files (one read into the 8 KiB stack buffer, one read reporting EOF). Once a read fills the stack buffer, read the rest directly into one heap buffer sized from fstat() (plus one byte so that the EOF read does not force growth), growing geometrically only when the size is unavailable or wrong. The size is only an allocation hint: reading continues until read() reports EOF, so procfs/sysfs files, FIFOs, files that change while being read and file descriptors positioned mid-file behave as before, and the bytes handed to StringBytes::Encode() are exactly the ones read. Signed-off-by: Shelley Vohr <shelley.vohr@gmail.com> PR-URL: #65328 Reviewed-By: Matteo Collina <matteo.collina@gmail.com> Reviewed-By: Yagiz Nizipli <yagiz@nizipli.com>
1 parent bc1e19b commit e3ae912

2 files changed

Lines changed: 160 additions & 2 deletions

File tree

β€Žsrc/node_file.ccβ€Ž

Lines changed: 62 additions & 2 deletions
Original file line numberDiff line numberDiff line change
@@ -2931,10 +2931,20 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29312931
uv_fs_req_cleanup(&req);
29322932
});
29332933

2934+
// Past the first 8 KiB, read into one heap buffer sized from fstat(); the
2935+
// size is only a hint, reading continues until read() reports EOF.
29342936
std::string result{};
29352937
char buffer[8192];
29362938
uv_buf_t buf = uv_buf_init(buffer, sizeof(buffer));
29372939

2940+
char* big = nullptr;
2941+
size_t big_len = 0;
2942+
size_t big_cap = 0;
2943+
bool sized = false;
2944+
auto free_big = OnScopeLeave([&big]() { free(big); });
2945+
constexprsize_tkMinChunk = 64 * 1024;
2946+
constexprsize_tkMaxChunk = 8 * 1024 * 1024;
2947+
29382948
FS_SYNC_TRACE_BEGIN(read);
29392949
while (true) {
29402950
auto r = uv_fs_read(nullptr, &req, file, &buf, 1, -1, nullptr);
@@ -2947,12 +2957,62 @@ static void ReadFileUtf8(const FunctionCallbackInfo<Value>& args) {
29472957
if (r <= 0) {
29482958
break;
29492959
}
2950-
result.append(buf.base, r);
2960+
if (big == nullptr) {
2961+
result.append(buf.base, r);
2962+
if (static_cast<size_t>(r) < sizeof(buffer)) {
2963+
continue;
2964+
}
2965+
// Switch to the heap buffer.
2966+
uv_fs_req_cleanup(&req);
2967+
big_cap = kMinChunk;
2968+
big = UncheckedMalloc<char>(big_cap);
2969+
if (big == nullptr) {
2970+
FS_SYNC_TRACE_END(read);
2971+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
2972+
}
2973+
memcpy(big, result.data(), result.size());
2974+
big_len = result.size();
2975+
result = std::string();
2976+
} else {
2977+
big_len += static_cast<size_t>(r);
2978+
}
2979+
if (big_len == big_cap) {
2980+
// +1 leaves room for the read() that reports EOF.
2981+
size_t new_cap =
2982+
big_cap + std::min(kMaxChunk, std::max(kMinChunk, big_cap));
2983+
if (!sized) {
2984+
sized = true;
2985+
uv_fs_req_cleanup(&req);
2986+
uv_fs_t stat_req;
2987+
if (uv_fs_fstat(nullptr, &stat_req, file, nullptr) == 0) {
2988+
constuv_stat_t* const st =
2989+
static_cast<constuv_stat_t*>(stat_req.ptr);
2990+
if ((st->st_mode & S_IFMT) == S_IFREG &&
2991+
static_cast<uint64_t>(st->st_size) > big_len &&
2992+
static_cast<uint64_t>(st->st_size) <
2993+
static_cast<uint64_t>(v8::String::kMaxLength)) {
2994+
new_cap = static_cast<size_t>(st->st_size) + 1;
2995+
}
2996+
}
2997+
uv_fs_req_cleanup(&stat_req);
2998+
}
2999+
char* const grown = UncheckedRealloc<char>(big, new_cap);
3000+
if (grown == nullptr) {
3001+
FS_SYNC_TRACE_END(read);
3002+
returnTHROW_ERR_MEMORY_ALLOCATION_FAILED(env);
3003+
}
3004+
big = grown;
3005+
big_cap = new_cap;
3006+
}
3007+
buf = uv_buf_init(big + big_len, std::min(kMaxChunk, big_cap - big_len));
29513008
}
29523009
FS_SYNC_TRACE_END(read);
29533010

29543011
Local<Value> val;
2955-
if (!ToV8Value(env->context(), result, isolate).ToLocal(&val)) {
3012+
const std::string_view content = big != nullptr
3013+
? std::string_view(big, big_len)
3014+
: std::string_view(result);
3015+
if (!ToV8Value(env->context(), content, isolate).ToLocal(&val)) {
29563016
return;
29573017
}
29583018

Lines changed: 98 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,98 @@
1+
'use strict';
2+
// fs.readFileSync(path, 'utf8') takes a dedicated native path. Its result must
3+
// equal fs.readFileSync(path).toString('utf8') for every file size (in
4+
// particular around its internal 8 KiB stack buffer and for multi-megabyte
5+
// files), for file descriptors positioned mid-file, and for files whose
6+
// reported size is wrong (procfs reports 0, sysfs reports a page).
7+
constcommon=require('../common');
8+
consttmpdir=require('../common/tmpdir');
9+
constassert=require('assert');
10+
constfs=require('fs');
11+
12+
tmpdir.refresh();
13+
14+
functioncontent(size){
15+
// Multi-byte characters straddling every possible chunk boundary.
16+
constunit='abcdé€\u{1F600}\n';
17+
lets=unit.repeat(Math.ceil(size/unit.length));
18+
s=s.slice(0,size);
19+
// Avoid ending on a lone surrogate produced by slice().
20+
if(/[\ud800-\udbff]$/.test(s))s=s.slice(0,-1)+'x';
21+
returns;
22+
}
23+
24+
constsizes=[0,1,8190,8191,8192,8193,8194,16383,16384,16385,
25+
65535,65536,65537,100000,(1<<20)-1,1<<20,(1<<20)+1,
26+
(8<<20)+5];
27+
for(constsizeofsizes){
28+
constfile=tmpdir.resolve(`f-${size}.txt`);
29+
conststr=content(size);
30+
fs.writeFileSync(file,str);
31+
constexpected=fs.readFileSync(file).toString('utf8');
32+
assert.strictEqual(fs.readFileSync(file,'utf8'),expected,`size ${size} by path`);
33+
assert.strictEqual(fs.readFileSync(file,{encoding: 'utf-8'}),expected,`size ${size} utf-8 alias`);
34+
// By fd: from the start (leaves the fd at EOF), then at EOF, then from a
35+
// mid-file position on a fresh fd.
36+
letfd=fs.openSync(file,'r');
37+
try{
38+
assert.strictEqual(fs.readFileSync(fd,'utf8'),expected,`size ${size} by fd`);
39+
assert.strictEqual(fs.readFileSync(fd,'utf8'),'',`size ${size} by fd at EOF`);
40+
}finally{
41+
fs.closeSync(fd);
42+
}
43+
if(size>10){
44+
fd=fs.openSync(file,'r');
45+
try{
46+
// Advance the fd 3 bytes (inside the ASCII prefix, so still valid UTF-8).
47+
assert.strictEqual(fs.readSync(fd,Buffer.alloc(3),0,3,null),3);
48+
assert.strictEqual(fs.readFileSync(fd,'utf8'),Buffer.from(expected).subarray(3).toString('utf8'),
49+
`size ${size} by fd at offset 3`);
50+
}finally{
51+
fs.closeSync(fd);
52+
}
53+
}
54+
}
55+
56+
// Binary garbage is decoded with replacement characters identically.
57+
{
58+
constfile=tmpdir.resolve('binary.bin');
59+
constbuf=Buffer.alloc(20000);
60+
for(leti=0;i<buf.length;i++)buf[i]=(i*7919)&0xff;
61+
fs.writeFileSync(file,buf);
62+
assert.strictEqual(fs.readFileSync(file,'utf8'),buf.toString('utf8'));
63+
}
64+
65+
// Files whose st_size does not describe their content.
66+
if(common.isLinux){
67+
for(constfileof['/proc/self/status','/proc/self/smaps','/proc/cpuinfo',
68+
'/proc/version','/sys/kernel/mm/transparent_hugepage/enabled']){
69+
letviaBuffer;
70+
try{
71+
viaBuffer=fs.readFileSync(file);
72+
}catch{
73+
continue;// Not available in this environment.
74+
}
75+
constviaUtf8=fs.readFileSync(file,'utf8');
76+
if(file!=='/proc/version'&&file.startsWith('/proc/')){
77+
// Content legitimately differs between two reads; compare shape instead.
78+
assert.ok(viaUtf8.length>0);
79+
assert.strictEqual(viaUtf8.split('\n').length>5,true,file);
80+
// Of these, smaps reliably exceeds the 8 KiB stack buffer.
81+
if(file==='/proc/self/smaps')assert.ok(viaUtf8.length>8192,`smaps is only ${viaUtf8.length} chars`);
82+
}else{
83+
assert.strictEqual(viaUtf8,viaBuffer.toString('utf8'),file);
84+
}
85+
}
86+
}
87+
88+
// Directory: same outcome either way (EISDIR, except on platforms where
89+
// read() accepts directories, e.g. AIX).
90+
functionoutcome(read){
91+
try{
92+
returnread();
93+
}catch(err){
94+
returnerr.code;
95+
}
96+
}
97+
assert.strictEqual(outcome(()=>fs.readFileSync(tmpdir.path,'utf8')),
98+
outcome(()=>fs.readFileSync(tmpdir.path).toString('utf8')));

0 commit comments

Comments
Β (0)