Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 8 additions & 13 deletions src/runtime/cli/pack_command.rs
Original file line number Diff line number Diff line change
Expand Up @@ -2491,9 +2491,9 @@ pub(crate) fn pack<const FOR_PUBLISH: bool>(
// default is 9
// https://github.com/npm/cli/blob/ec105f400281a5bfd17885de1ea3d54d0c231b27/node_modules/pacote/lib/util/tar-create-options.js#L12
let compression_level: &[u8] = opt_pack_gzip_level(ctx.manager).unwrap_or(b"9");
write!(&mut print_buf, "{}\x00", bstr::BStr::new(compression_level)).expect("OOM");
// SAFETY: print_buf[compression_level.len()] == 0 written above
let level_z = ZStr::from_buf(&print_buf[..], compression_level.len());
print_buf.extend_from_slice(compression_level);
print_buf.push(0);
let level_z = ZStr::from_slice_with_nul(&print_buf[..]);
match archive.write_set_filter_option(None, zstr_lit(b"compression-level\0"), level_z) {
ArchiveResult::Failed | ArchiveResult::Fatal | ArchiveResult::Warn => {
Output::err_generic(
Expand Down Expand Up @@ -3197,16 +3197,11 @@ fn add_archive_entry(
) -> Result<*mut ArchiveEntry, AllocError> {
// `entry` is the same pointer after `.clear()`.
let entry = ArchiveEntry::opaque_ref(entry);
write!(
print_buf,
"{}{}\x00",
bstr::BStr::new(PACKAGE_PREFIX),
bstr::BStr::new(filename.as_bytes())
)
.expect("OOM");
let pathname_len = PACKAGE_PREFIX.len() + filename.as_bytes().len();
// SAFETY: print_buf[pathname_len] == 0 written above
let pathname = ZStr::from_buf(&print_buf[..], pathname_len);
// Not formatted through `BStr`: its `Display` rewrites bytes that are not valid UTF-8.
print_buf.extend_from_slice(PACKAGE_PREFIX);
print_buf.extend_from_slice(filename.as_bytes());
print_buf.push(0);
let pathname = ZStr::from_slice_with_nul(&print_buf[..]);
#[cfg(windows)]
entry.set_pathname_utf8(pathname);
#[cfg(not(windows))]
Expand Down
70 changes: 69 additions & 1 deletion test/cli/install/bun-pack.test.ts
Original file line number Diff line number Diff line change
Expand Up @@ -2,7 +2,7 @@ import { file, spawn, write } from "bun";
import { readTarball } from "bun:internal-for-testing";
import { beforeEach, describe, expect, test } from "bun:test";
import { exists, mkdir, rm } from "fs/promises";
import { bunEnv, bunExe, pack, runBunInstall, tempDir, tmpdirSync } from "harness";
import { bunEnv, bunExe, isLinux, pack, runBunInstall, tempDir, tmpdirSync } from "harness";
import fs from "node:fs/promises";
import { join } from "path";

Expand Down Expand Up @@ -1616,6 +1616,74 @@ test("unicode", async () => {
expect(tarball.entries).toMatchObject([{ "pathname": "package/package.json" }, { "pathname": "package/äöüščří.js" }]);
});

// Reads the members of a .tgz without decoding their names: `readTarball` turns
// names into JS strings, which maps the bytes this test is about onto U+FFFD.
// Names are returned latin1-decoded (one char per stored byte) and collected
// from every place the archive spells a name: the ustar header `name` field
// and, when libarchive emits one, the `path` record of the pax header.
function rawTarballMembers(tgz: Uint8Array) {
const tar = Buffer.from(Bun.gunzipSync(tgz));
const members: { names: Set<string>; contents: string }[] = [];
let paxPath: string | undefined;
for (let offset = 0; offset + 512 <= tar.length; ) {
const header = tar.subarray(offset, offset + 512);
if (header.every(byte => byte === 0)) break;
const nameField = header.subarray(0, 100);
const nameLength = nameField.indexOf(0) === -1 ? 100 : nameField.indexOf(0);
const name = nameField.toString("latin1", 0, nameLength);
const size = parseInt(header.toString("latin1", 124, 136), 8);
const typeflag = header.toString("latin1", 156, 157);
const data = tar.subarray(offset + 512, offset + 512 + size);
offset += 512 + Math.ceil(size / 512) * 512;

if (typeflag === "x") {
// pax extended header data is a sequence of "<record length> <key>=<value>\n"
for (let pos = 0; pos < data.length; ) {
const space = data.indexOf(" ", pos);
const recordLength = parseInt(data.toString("latin1", pos, space), 10);
const record = data.toString("latin1", space + 1, pos + recordLength - 1);
if (record.startsWith("path=")) paxPath = record.slice("path=".length);
pos += recordLength;
}
continue;
}

const names = new Set([name]);
if (paxPath !== undefined) names.add(paxPath);
paxPath = undefined;
members.push({ names, contents: data.toString("latin1") });
}
return members.sort((a, b) => (a.contents < b.contents ? -1 : 1));
}

// Only Linux lets a filename carry bytes that are not valid UTF-8 (APFS rejects
// them and Windows filenames are UTF-16).
test.skipIf(!isLinux)("filenames that are not valid UTF-8 are stored byte for byte", async () => {
const rawPath = (byte: number) => Buffer.concat([Buffer.from(`${packageDir}/x_`), Buffer.from([byte])]);
await Promise.all([
write(
join(packageDir, "package.json"),
JSON.stringify({
name: "pack-invalid-utf8",
version: "1.0.0",
}),
),
// latin-1 "é", and a byte that is invalid in UTF-8 at any position
fs.writeFile(rawPath(0xe9), "ONE"),
fs.writeFile(rawPath(0xff), "TWO"),
]);

const { out } = await pack(packageDir, bunEnv);
expect(out).toContain("Total files: 3");

const members = rawTarballMembers(await file(join(packageDir, "pack-invalid-utf8-1.0.0.tgz")).bytes());
expect(members).toEqual([
{ names: new Set(["package/x_\xe9"]), contents: "ONE" },
{ names: new Set(["package/x_\xff"]), contents: "TWO" },
{ names: new Set(["package/package.json"]), contents: expect.stringContaining(`"pack-invalid-utf8"`) },
]);
});

test("$npm_command is accurate", async () => {
await write(
join(packageDir, "package.json"),
Expand Down