git: 9a0b9a23e0bb - main - mkimg: Add support for stream-optimized VMDK
- Go to: [ bottom of page ] [ top of archives ] [ this month ]
Date: Sat, 03 Oct 2026 15:42:43 UTC
The branch main has been updated by cperciva:
URL: https://cgit.FreeBSD.org/src/commit/?id=9a0b9a23e0bb691034a019edea756377509d0a98
commit 9a0b9a23e0bb691034a019edea756377509d0a98
Author: Colin Percival <cperciva@FreeBSD.org>
AuthorDate: 2026-09-29 19:29:22 +0000
Commit: Colin Percival <cperciva@FreeBSD.org>
CommitDate: 2026-10-03 15:42:20 +0000
mkimg: Add support for stream-optimized VMDK
FreeBSD VM/cloud images tend to be highly compressible: The generic VM
images compress roughly 3.5:1, and cloud images typically even more
since they have significant unused space in their virtual disks. This
generally doesn't matter for users who can download compressed release
images and extract them locally, but for EC2 in particular relying on
uncompressed image formats is painful: We now upload over 500 GB/week
to AWS.
This commit adds the "stream-optimized" version of the VMDK format,
which compresses each 64 kB "grain" individually (and omits grains
which are all zeroes). Compared to uncompressed formats, this can
produce much smaller images; experiments indicate that using this
format for EC2 image uploads will reduce the weekly traffic to under
100 GB/week.
Co-authored-by: Claude Opus 5.5
MFC after: 2 weeks
Sponsored by: Amazon
Differential Revision: https://reviews.freebsd.org/D60151
---
usr.bin/mkimg/Makefile | 2 +-
usr.bin/mkimg/vmdk.c | 309 +++++++++++++++++++++++++++++++++++++++++++++++++
2 files changed, 310 insertions(+), 1 deletion(-)
diff --git a/usr.bin/mkimg/Makefile b/usr.bin/mkimg/Makefile
index 9e39b7abc956..870884216679 100644
--- a/usr.bin/mkimg/Makefile
+++ b/usr.bin/mkimg/Makefile
@@ -29,7 +29,7 @@ SRCS+= \
BINDIR?=/usr/bin
-LIBADD= util
+LIBADD= util z
HAS_TESTS=
SUBDIR.${MK_TESTS}+= tests
diff --git a/usr.bin/mkimg/vmdk.c b/usr.bin/mkimg/vmdk.c
index 838cfa673ed6..8b16c6b57eae 100644
--- a/usr.bin/mkimg/vmdk.c
+++ b/usr.bin/mkimg/vmdk.c
@@ -32,6 +32,7 @@
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
+#include <zlib.h>
#include "endian.h"
#include "image.h"
@@ -41,12 +42,14 @@
#define VMDK_IMAGE_ROUND 1048576
#define VMDK_MIN_GRAIN_SIZE 8192
#define VMDK_SECTOR_SIZE 512
+#define VMDK_STREAM_GRAIN_SIZE 65536
struct vmdk_header {
uint32_t magic;
#define VMDK_MAGIC 0x564d444b
uint32_t version;
#define VMDK_VERSION 1
+#define VMDK_VERSION_STREAM 3
uint32_t flags;
#define VMDK_FLAGS_NL_TEST (1 << 0)
#define VMDK_FLAGS_RGT_USED (1 << 1)
@@ -60,6 +63,7 @@ struct vmdk_header {
#define VMDK_NGTES 512
uint64_t rgd_offset;
uint64_t gd_offset;
+#define VMDK_GD_AT_END 0xffffffffffffffffULL
uint64_t overhead;
uint8_t unclean;
uint32_t nl_test;
@@ -70,6 +74,35 @@ struct vmdk_header {
char padding[433];
} __attribute__((__packed__));
+/*
+ * In a stream-optimized VMDK, each grain is preceded by a 12-byte marker
+ * giving the grain's LBA and compressed size, and metadata (grain tables,
+ * the grain directory, and the footer) is preceded by a sector-sized
+ * marker giving the size of the metadata in sectors and its type. A
+ * sector of zeroes marks the end of the stream.
+ */
+#define VMDK_GRAIN_MARKER_SIZE 12
+
+struct vmdk_marker {
+ uint64_t val;
+ uint32_t size;
+} __attribute__((__packed__));
+_Static_assert(sizeof(struct vmdk_marker) == VMDK_GRAIN_MARKER_SIZE,
+ "Wrong size for VMDK marker");
+
+struct vmdk_metadata_marker {
+ uint64_t val;
+ uint32_t size;
+ uint32_t type;
+#define VMDK_MARKER_EOS 0
+#define VMDK_MARKER_GT 1
+#define VMDK_MARKER_GD 2
+#define VMDK_MARKER_FOOTER 3
+ char padding[496];
+} __attribute__((__packed__));
+_Static_assert(sizeof(struct vmdk_metadata_marker) == VMDK_SECTOR_SIZE,
+ "Wrong size for VMDK marker");
+
static const char desc_fmt[] =
"# Disk DescriptorFile\n"
"version=%d\n"
@@ -265,3 +298,279 @@ static struct mkimg_format vmdk_format = {
};
FORMAT_DEFINE(vmdk_format);
+
+/*
+ * Stream-optimized VMDK: grains are compressed and written sequentially,
+ * each grain table is written after the grains it describes, and the grain
+ * directory and a footer containing a copy of the header (with the grain
+ * directory offset filled in) come at the end.
+ */
+
+static int
+vmdk_stream_resize(lba_t imgsz)
+{
+ uint64_t imagesz;
+
+ imagesz = imgsz * secsz;
+ imagesz = (imagesz + VMDK_IMAGE_ROUND - 1) & ~(VMDK_IMAGE_ROUND - 1);
+ grainsz = (secsz < VMDK_STREAM_GRAIN_SIZE) ?
+ VMDK_STREAM_GRAIN_SIZE : secsz;
+
+ if (verbose)
+ fprintf(stderr, "VMDK: image size = %ju, grain size = %ju\n",
+ (uintmax_t)imagesz, (uintmax_t)grainsz);
+
+ grainsz /= VMDK_SECTOR_SIZE;
+ return (image_set_size(imagesz / secsz));
+}
+
+static int
+vmdk_stream_marker(int fd, uint64_t val, uint32_t type)
+{
+ struct vmdk_metadata_marker m;
+
+ memset(&m, 0, sizeof(m));
+ le64enc(&m.val, val);
+ le32enc(&m.type, type);
+ if (sparse_write(fd, &m, sizeof(m)) < 0)
+ return (errno);
+ return (0);
+}
+
+/*
+ * Write out the grain table gt, preceded by a marker, starting at sector
+ * *secp; record its location in the grain directory entry *gde, advance
+ * *secp past it, and clear gt for reuse.
+ */
+static int
+vmdk_stream_gt(int fd, uint32_t *gt, uint32_t *gde, uint64_t *secp)
+{
+ size_t gtsz;
+ int error;
+
+ gtsz = VMDK_NGTES * sizeof(uint32_t);
+ error = vmdk_stream_marker(fd, gtsz / VMDK_SECTOR_SIZE,
+ VMDK_MARKER_GT);
+ if (error)
+ return (error);
+ *secp += 1;
+
+ /* Grain directory entries are 32-bit sector numbers. */
+ if (*secp > UINT32_MAX)
+ return (EFBIG);
+ le32enc(gde, *secp);
+ if (sparse_write(fd, gt, gtsz) < 0)
+ return (errno);
+ *secp += gtsz / VMDK_SECTOR_SIZE;
+ memset(gt, 0, gtsz);
+ return (0);
+}
+
+static int
+vmdk_stream_iszero(const uint8_t *buf, size_t len)
+{
+ size_t i;
+
+ for (i = 0; i < len; i++) {
+ if (buf[i] != 0)
+ return (0);
+ }
+ return (1);
+}
+
+static int
+vmdk_stream_write(int fd)
+{
+ struct vmdk_header hdr;
+ struct vmdk_marker ghead;
+ uint32_t *gd, *gt;
+ uint8_t *gbuf, *zbuf;
+ char *buf, *desc;
+ uint64_t gdofs, grain, imagesz, ngrains, ngts, overhead, sec;
+ lba_t blkcnt;
+ size_t gdsz, gtsz, grainbytes, pos, len, zbufsz;
+ uLongf zlen;
+ int desc_len, error, gtused, n, zerror;
+
+ imagesz = (image_get_size() * secsz) / VMDK_SECTOR_SIZE;
+ grainbytes = grainsz * VMDK_SECTOR_SIZE;
+ blkcnt = grainbytes / secsz;
+ ngrains = imagesz / grainsz;
+ ngts = (ngrains + VMDK_NGTES - 1) / VMDK_NGTES;
+ gdsz = (ngts * sizeof(uint32_t) + VMDK_SECTOR_SIZE - 1) &
+ ~(VMDK_SECTOR_SIZE - 1);
+ gtsz = VMDK_NGTES * sizeof(uint32_t);
+
+ n = asprintf(&desc, desc_fmt, 1 /*version*/, 0 /*CID*/,
+ "streamOptimized" /*type*/, (uintmax_t)imagesz /*size*/,
+ "" /*name*/, ncyls /*cylinders*/, nheads /*heads*/,
+ nsecs /*sectors*/, "ddb.virtualHWVersion = \"4\"\n" /*extra*/);
+ if (n == -1)
+ return (ENOMEM);
+
+ desc_len = (n + VMDK_SECTOR_SIZE - 1) & ~(VMDK_SECTOR_SIZE - 1);
+ buf = realloc(desc, desc_len);
+ if (buf == NULL) {
+ free(desc);
+ return (ENOMEM);
+ }
+ desc = buf;
+ memset(desc + n, 0, desc_len - n);
+
+ /* Grains start at the first grain boundary after the descriptor. */
+ overhead = 1 + desc_len / VMDK_SECTOR_SIZE;
+ overhead = (overhead + grainsz - 1) & ~(grainsz - 1);
+
+ memset(&hdr, 0, sizeof(hdr));
+ le32enc(&hdr.magic, VMDK_MAGIC);
+ le32enc(&hdr.version, VMDK_VERSION_STREAM);
+ le32enc(&hdr.flags, VMDK_FLAGS_NL_TEST | VMDK_FLAGS_COMPRESSED |
+ VMDK_FLAGS_MARKERS);
+ le64enc(&hdr.capacity, imagesz);
+ le64enc(&hdr.grain_size, grainsz);
+ le64enc(&hdr.desc_offset, 1);
+ le64enc(&hdr.desc_size, desc_len / VMDK_SECTOR_SIZE);
+ le32enc(&hdr.ngtes, VMDK_NGTES);
+ le64enc(&hdr.gd_offset, VMDK_GD_AT_END);
+ le64enc(&hdr.overhead, overhead);
+ be32enc(&hdr.nl_test, VMDK_NL_TEST);
+ le16enc(&hdr.compress, VMDK_COMPRESS_DEFLATE);
+
+ if (verbose)
+ fprintf(stderr, "VMDK: overhead = %ju\n",
+ (uintmax_t)(overhead * VMDK_SECTOR_SIZE));
+
+ gd = calloc(1, gdsz);
+ gt = calloc(1, gtsz);
+ gbuf = malloc(grainbytes);
+ zbufsz = (VMDK_GRAIN_MARKER_SIZE + compressBound(grainbytes) +
+ VMDK_SECTOR_SIZE - 1) & ~(VMDK_SECTOR_SIZE - 1);
+ zbuf = malloc(zbufsz);
+ if (gd == NULL || gt == NULL || gbuf == NULL || zbuf == NULL) {
+ error = ENOMEM;
+ goto out;
+ }
+
+ /* Write the header and descriptor, and pad to the first grain. */
+ error = 0;
+ if (sparse_write(fd, &hdr, VMDK_SECTOR_SIZE) < 0 ||
+ sparse_write(fd, desc, desc_len) < 0) {
+ error = errno;
+ goto out;
+ }
+ error = image_copyout_zeroes(fd,
+ (overhead - 1) * VMDK_SECTOR_SIZE - desc_len);
+ if (error)
+ goto out;
+ sec = overhead;
+
+ gtused = 0;
+ for (grain = 0; grain < ngrains; grain++) {
+ /*
+ * If we're starting a new grain table, write out the
+ * previous one (but only if it has any entries).
+ */
+ if (grain % VMDK_NGTES == 0 && gtused) {
+ error = vmdk_stream_gt(fd, gt,
+ gd + grain / VMDK_NGTES - 1, &sec);
+ if (error)
+ goto out;
+ gtused = 0;
+ }
+
+ /* Skip grains which are entirely zero. */
+ if (!image_data(grain * blkcnt, blkcnt))
+ continue;
+ error = image_buffer_region((char *)gbuf, grain * blkcnt,
+ blkcnt);
+ if (error)
+ goto out;
+ if (vmdk_stream_iszero(gbuf, grainbytes))
+ continue;
+
+ /* Compress the grain, after space for the grain marker. */
+ zlen = zbufsz - VMDK_GRAIN_MARKER_SIZE;
+ zerror = compress2(zbuf + VMDK_GRAIN_MARKER_SIZE, &zlen,
+ gbuf, grainbytes, Z_DEFAULT_COMPRESSION);
+ if (zerror != Z_OK) {
+ error = (zerror == Z_MEM_ERROR) ? ENOMEM : EIO;
+ goto out;
+ }
+
+ /* Fill in the marker, and pad to a sector boundary. */
+ le64enc(&ghead.val, grain * grainsz);
+ le32enc(&ghead.size, zlen);
+ memcpy(&zbuf[0], &ghead, VMDK_GRAIN_MARKER_SIZE);
+ pos = VMDK_GRAIN_MARKER_SIZE + zlen;
+ len = (pos + VMDK_SECTOR_SIZE - 1) & ~(VMDK_SECTOR_SIZE - 1);
+ memset(&zbuf[pos], 0, len - pos);
+
+ /* Grain table entries are 32-bit sector numbers. */
+ if (sec > UINT32_MAX) {
+ error = EFBIG;
+ goto out;
+ }
+
+ /* Record the grain's location and write it out. */
+ le32enc(gt + grain % VMDK_NGTES, sec);
+ if (sparse_write(fd, zbuf, len) < 0) {
+ error = errno;
+ goto out;
+ }
+ sec += len / VMDK_SECTOR_SIZE;
+ gtused = 1;
+ }
+
+ /* Write out the final grain table, if it has any entries. */
+ if (gtused) {
+ error = vmdk_stream_gt(fd, gt, gd + (ngrains - 1) / VMDK_NGTES,
+ &sec);
+ if (error)
+ goto out;
+ }
+
+ /* Write the grain directory. */
+ error = vmdk_stream_marker(fd, gdsz / VMDK_SECTOR_SIZE,
+ VMDK_MARKER_GD);
+ if (error)
+ goto out;
+ gdofs = sec + 1;
+ if (sparse_write(fd, gd, gdsz) < 0) {
+ error = errno;
+ goto out;
+ }
+
+ /* Write the footer: a copy of the header with the GD offset. */
+ error = vmdk_stream_marker(fd, 1, VMDK_MARKER_FOOTER);
+ if (error)
+ goto out;
+ le64enc(&hdr.gd_offset, gdofs);
+ if (sparse_write(fd, &hdr, VMDK_SECTOR_SIZE) < 0) {
+ error = errno;
+ goto out;
+ }
+
+ /* Write the end-of-stream marker. */
+ error = vmdk_stream_marker(fd, 0, VMDK_MARKER_EOS);
+ if (error)
+ goto out;
+
+ error = image_copyout_done(fd);
+
+out:
+ free(zbuf);
+ free(gbuf);
+ free(gt);
+ free(gd);
+ free(desc);
+ return (error);
+}
+
+static struct mkimg_format vmdk_stream_format = {
+ .name = "vmdks",
+ .description = "Virtual Machine Disk, stream-optimized (compressed)",
+ .resize = vmdk_stream_resize,
+ .write = vmdk_stream_write,
+};
+
+FORMAT_DEFINE(vmdk_stream_format);