feat: improve office document discovery

Signed-off-by: Sebastian Krupinski <krupinski01@gmail.com>
This commit is contained in:
2026-09-07 12:55:11 -04:00
parent 625db726f6
commit fbc0559a64
+63 -8
View File
@@ -23,7 +23,10 @@ use finfo;
class Signature { class Signature {
/** Minimum bytes needed for reliable detection */ /** Minimum bytes needed for reliable detection */
public const HEADER_SIZE = 256; public const SAMPLE_SIZE = 65536;
/** Cached finfo instance */
private static ?finfo $finfo = null;
/** /**
* Fallback magic byte signatures for when finfo is unavailable * Fallback magic byte signatures for when finfo is unavailable
@@ -41,16 +44,32 @@ class Signature {
['offset' => 0, 'bytes' => '52494646', 'format' => 'riff'], // WAV/AVI/WEBP ['offset' => 0, 'bytes' => '52494646', 'format' => 'riff'], // WAV/AVI/WEBP
]; ];
/** Cached finfo instance */ /**
private static ?finfo $finfo = null; * Zip-container marker strings used to distinguish OOXML/ODF/EPUB documents
* from a generic ZIP archive. Filenames inside a ZIP's local file headers
* (and ODF's mandatory uncompressed "mimetype" entry content) are stored
* as plain text, so a simple substring search reliably identifies these
* formats without needing to parse the archive structure.
*/
private const ZIP_CONTAINER_MARKERS = [
'word/document.xml' => ['application/vnd.openxmlformats-officedocument.wordprocessingml.document', 'docx'],
'xl/workbook.xml' => ['application/vnd.openxmlformats-officedocument.spreadsheetml.sheet', 'xlsx'],
'ppt/presentation.xml' => ['application/vnd.openxmlformats-officedocument.presentationml.presentation', 'pptx'],
'application/vnd.oasis.opendocument.text' => ['application/vnd.oasis.opendocument.text', 'odt'],
'application/vnd.oasis.opendocument.spreadsheet' => ['application/vnd.oasis.opendocument.spreadsheet', 'ods'],
'application/vnd.oasis.opendocument.presentation' => ['application/vnd.oasis.opendocument.presentation', 'odp'],
'application/epub+zip' => ['application/epub+zip', 'epub'],
];
/** /**
* Detect both MIME type and format from content bytes in a single operation * Detect both MIME type and format from content bytes in a single operation
* *
* @param string $headerBytes First bytes of the file content (256 recommended) * @param string $headerBytes First bytes of the file content
* @param string|null $content Full (or larger) content, when available, used to
* distinguish OOXML/ODF/EPUB documents from a generic ZIP archive
* @return array{mime: string, format: string} Array with 'mime' and 'format' keys * @return array{mime: string, format: string} Array with 'mime' and 'format' keys
*/ */
public static function detect(string $headerBytes): array { public static function detect(string $headerBytes, ?string $content = null): array {
if (strlen($headerBytes) === 0) { if (strlen($headerBytes) === 0) {
return ['mime' => MimeTypes::MIME_BINARY, 'format' => MimeTypes::FORMAT_BINARY]; return ['mime' => MimeTypes::MIME_BINARY, 'format' => MimeTypes::FORMAT_BINARY];
} }
@@ -75,6 +94,16 @@ class Signature {
$format = self::detectFromMagicBytes($headerBytes); $format = self::detectFromMagicBytes($headerBytes);
} }
// A bare "zip" result is a generic container; when more of the file is
// available, check for OOXML/ODF/EPUB markers rather than reporting the
// misleadingly generic zip mime/format for what is really a document.
if ($format === 'zip' && $content !== null) {
$container = self::detectZipContainer($content);
if ($container !== null) {
[$mime, $format] = $container;
}
}
// Ensure MIME type is set // Ensure MIME type is set
if ($mime === null || $mime === MimeTypes::MIME_BINARY) { if ($mime === null || $mime === MimeTypes::MIME_BINARY) {
$mime = MimeTypes::toMime($format) ?? MimeTypes::MIME_BINARY; $mime = MimeTypes::toMime($format) ?? MimeTypes::MIME_BINARY;
@@ -83,6 +112,32 @@ class Signature {
return ['mime' => $mime, 'format' => $format]; return ['mime' => $mime, 'format' => $format];
} }
/**
* Look for known OOXML/ODF/EPUB entry markers within ZIP content
*
* Only a fixed-size window from the head and tail of the content is scanned,
* regardless of total length, so detection cost does not grow with the size
* of large archive uploads. Every entry name is mirrored in full in the ZIP
* central directory near the end of the archive, so sampling the head and
* tail is enough without scanning the whole file.
*
* @param string $data ZIP content to scan (filenames/mimetype entry are stored uncompressed)
* @return array{0: string, 1: string}|null [mime, format] pair, or null if no marker matched
*/
private static function detectZipContainer(string $data): ?array {
$length = strlen($data);
$sample = $length <= self::SAMPLE_SIZE * 2
? $data
: substr($data, 0, self::SAMPLE_SIZE) . substr($data, -self::SAMPLE_SIZE);
foreach (self::ZIP_CONTAINER_MARKERS as $marker => $result) {
if (str_contains($sample, $marker)) {
return $result;
}
}
return null;
}
/** /**
* Detect both MIME type and format from a stream in a single operation * Detect both MIME type and format from a stream in a single operation
* *
@@ -91,7 +146,7 @@ class Signature {
*/ */
public static function detectFromStream($stream): array { public static function detectFromStream($stream): array {
$position = ftell($stream); $position = ftell($stream);
$headerBytes = fread($stream, self::HEADER_SIZE); $headerBytes = fread($stream, self::SAMPLE_SIZE);
fseek($stream, $position); fseek($stream, $position);
if ($headerBytes === false || $headerBytes === '') { if ($headerBytes === false || $headerBytes === '') {
@@ -140,7 +195,7 @@ class Signature {
*/ */
public static function detectFormatFromStream($stream): string { public static function detectFormatFromStream($stream): string {
$position = ftell($stream); $position = ftell($stream);
$headerBytes = fread($stream, self::HEADER_SIZE); $headerBytes = fread($stream, self::SAMPLE_SIZE);
fseek($stream, $position); fseek($stream, $position);
if ($headerBytes === false || $headerBytes === '') { if ($headerBytes === false || $headerBytes === '') {
@@ -158,7 +213,7 @@ class Signature {
*/ */
public static function detectMimeTypeFromStream($stream): ?string { public static function detectMimeTypeFromStream($stream): ?string {
$position = ftell($stream); $position = ftell($stream);
$headerBytes = fread($stream, self::HEADER_SIZE); $headerBytes = fread($stream, self::SAMPLE_SIZE);
fseek($stream, $position); fseek($stream, $position);
if ($headerBytes === false || $headerBytes === '') { if ($headerBytes === false || $headerBytes === '') {