diff --git a/src/c2pa/c2pa.py b/src/c2pa/c2pa.py index fa589a6a..dec2921f 100644 --- a/src/c2pa/c2pa.py +++ b/src/c2pa/c2pa.py @@ -1196,16 +1196,21 @@ def load_settings(settings: Union[str, dict], format: str = "json") -> None: def _get_mime_type_from_path(path: Union[str, Path]) -> str: - """Attempt to guess the MIME type from a file path (with extension). + """Attempt to guess the MIME type from a file path's extension. + When the extension is missing or unrecognized, this returns an empty + string so the caller hands it to the lib for auto-detection. + A recognized-but-wrong extension still returns a mimetype: + the native layer will attempt to correct it from the bytes when + reading (the real type still needs to be supported by the lib), + and if it can't, an error will happen then. Args: path: File path as string or Path object Returns: - MIME type string - - Raises: - C2paError.NotSupported: If MIME type cannot be determined + MIME type string, or an empty string + (when it cannot be determined from the extension). + An empty string here means the native lib should attempt auto-detect. """ path_obj = Path(path) file_extension = path_obj.suffix.lower() if path_obj.suffix else "" @@ -1215,11 +1220,9 @@ def _get_mime_type_from_path(path: Union[str, Path]) -> str: # so we bypass it and set the correct type return "image/dng" else: - mime_type = mimetypes.guess_type(str(path))[0] - if not mime_type: - raise C2paError.NotSupported( - f"Could not determine MIME type for file: {path}") - return mime_type + # Fall back to an empty string for extensionless or unknown files. + # Empty string flags this as a guess-type attempt. + return mimetypes.guess_type(str(path))[0] or "" class ContextProvider(ABC): @@ -1949,7 +1952,8 @@ def _get_supported_mime_types(ffi_func, cache): try: if arr[i] is None: continue - mime_type = arr[i].decode("utf-8", errors='replace') + mime_type = arr[i].decode( + "utf-8", errors='replace').strip().lower() if mime_type: result.append(mime_type) except Exception: @@ -1968,28 +1972,48 @@ def _get_supported_mime_types(ffi_func, cache): return [], cache -def _validate_and_encode_format( - format_str: str, supported_types: list[str], class_name: str +def _encode_format( + format_str: Optional[str], + class_name: str, + allow_autodetect: bool = True, ) -> bytes: - """Validate a MIME type / format string and encode it to UTF-8 bytes. + """Normalize a MIME type/format string and encode it to UTF-8 bytes. + + The binding does not validate the format itself: the native library is the + authority on what is supported and detects the asset type from the bytes, + so the format is passed straight through after trimming whitespace. + + None, an empty string, or whitespace all mean the same thing: no format was + given. When allow_autodetect is True (default), that requests detection and + the native library guesses the type from the bytes. When allow_autodetect + is False, a missing format is rejected instead. Args: - format_str: The MIME type or format string to validate - supported_types: List of supported MIME types - class_name: Name of the calling class (for error messages) + format_str: The MIME type or format, or None. Pass None, an empty + string, or whitespace to request detection from the asset's bytes + (only when ``allow_autodetect`` is True). + class_name: Name of the calling class (for error messages). + allow_autodetect: When True (default), a missing format requests + detection. When False, a missing format raises + C2paError.NotSupported. Returns: - UTF-8 encoded format bytes + UTF-8 encoded format bytes (empty bytes for auto-detection). Raises: - C2paError.NotSupported: If the format is not supported - C2paError.Encoding: If the string contains invalid UTF-8 characters + C2paError.NotSupported: If the format is missing and + ``allow_autodetect`` is False. + C2paError.Encoding: If the format contains invalid UTF-8 characters. """ - if format_str.lower() not in supported_types: + key = (format_str or "").strip() + if not key: + if allow_autodetect: + return b"" raise C2paError.NotSupported( - f"{class_name} does not support {format_str}") + f"{class_name} requires an explicit format (MIME type)" + ) try: - return format_str.encode('utf-8') + return key.encode('utf-8') except UnicodeError as e: raise C2paError.Encoding( f"Invalid UTF-8 characters in input: {e}") @@ -2043,6 +2067,10 @@ def get_supported_mime_types(cls) -> list[str]: def _is_mime_type_supported(cls, mime_type: str) -> bool: """Check if a MIME type is supported. + Not used internally: the native library is the authority on format + support and detects the type from the bytes. Kept for callers that + want a pre-flight check before handing an asset to a Reader. + Args: mime_type: The MIME type to check @@ -2057,7 +2085,7 @@ def _is_mime_type_supported(cls, mime_type: str) -> bool: @overload def try_create( cls, - format_or_path: Union[str, Path], + format_or_path: Union[str, Path, None], stream: Optional[Any] = None, manifest_data: Optional[Any] = None, ) -> Optional["Reader"]: ... @@ -2066,7 +2094,7 @@ def try_create( @overload def try_create( cls, - format_or_path: Union[str, Path], + format_or_path: Union[str, Path, None], stream: Optional[Any], manifest_data: Optional[Any], context: 'ContextProvider', @@ -2075,7 +2103,7 @@ def try_create( @classmethod def try_create( cls, - format_or_path: Union[str, Path], + format_or_path: Union[str, Path, None], stream: Optional[Any] = None, manifest_data: Optional[Any] = None, context: Optional['ContextProvider'] = None, @@ -2090,7 +2118,8 @@ def try_create( exceptions for the expected case of no manifest. Args: - format_or_path: The format or path to read from + format_or_path: The format or path to read from. + None or an empty string requests auto-detection (best effort). stream: Optional stream to read from (Python stream-like object) manifest_data: Optional manifest data in bytes context: Optional ContextProvider for settings @@ -2113,7 +2142,7 @@ def try_create( @overload def __init__( self, - format_or_path: Union[str, Path], + format_or_path: Union[str, Path, None], stream: Optional[Any] = None, manifest_data: Optional[Any] = None, ) -> None: ... @@ -2121,7 +2150,7 @@ def __init__( @overload def __init__( self, - format_or_path: Union[str, Path], + format_or_path: Union[str, Path, None], stream: Optional[Any], manifest_data: Optional[Any], context: 'ContextProvider', @@ -2129,21 +2158,36 @@ def __init__( def __init__( self, - format_or_path: Union[str, Path], + format_or_path: Union[str, Path, None], stream: Optional[Any] = None, manifest_data: Optional[Any] = None, context: Optional['ContextProvider'] = None, ): """Create a new Reader. + The format is optional. Passing a known format gives the native + library more to work with, so prefer it. Pass None, an empty string, + or an extensionless file path to let the library detect the type from + the asset bytes. + + The library reconciles the format against the container it detects in + the bytes: when the given format disagrees with the detected container, + the library ignores the format and uses its own best guess (so reading + does not fail on a wrong file extension); when no container is + detected, the format is used as a hint and the asset is treated as a + sidecar. If the bytes are not a recognized asset and no format resolves + the type, a C2paError is raised. + Args: - format_or_path: The format or path to read from + format_or_path: The format (MIME type) or path to read from. + None or an empty string requests detection from the bytes. stream: Optional stream to read from (Python stream-like object) manifest_data: Optional manifest data in bytes context: Optional context implementing ContextProvider with settings Raises: - C2paError: If there was an error creating the reader + C2paError: If there was an error creating the reader, including + when the format cannot be detected from an unrecognized asset C2paError.Encoding: If any of the string inputs contain invalid UTF-8 characters """ @@ -2172,32 +2216,28 @@ def __init__( ) return - supported = Reader.get_supported_mime_types() - if stream is None: - # Create a stream from the file path in format_or_path + # Create a stream from the file path in format_or_path. + # Empty format = native lib will try to guess format. path = str(format_or_path) mime_type = _get_mime_type_from_path(path) - if not mime_type: - raise C2paError.NotSupported( - f"Could not determine MIME type for file: {path}") - - format_bytes = _validate_and_encode_format( - mime_type, supported, "Reader") + format_bytes = _encode_format(mime_type, "Reader") self._init_from_file(path, format_bytes) elif isinstance(stream, str): # stream is a file path, format_or_path is the format - format_bytes = _validate_and_encode_format( - str(format_or_path), supported, "Reader") + format_bytes = _encode_format( + None if format_or_path is None else str(format_or_path), + "Reader") self._init_from_file( stream, format_bytes, manifest_data) else: # format_or_path is a format string, stream is a stream object - format_bytes = _validate_and_encode_format( - str(format_or_path), supported, "Reader") + format_bytes = _encode_format( + None if format_or_path is None else str(format_or_path), + "Reader") with Stream(stream) as stream_obj: self._create_reader( @@ -2269,26 +2309,23 @@ def _init_from_context(self, context, format_or_path, raise TypeError(Reader._ERROR_MESSAGES['manifest_error']) # Determine format and open stream - supported = Reader.get_supported_mime_types() - if stream is None: + # Empty format = native lib will try to guess format. path = str(format_or_path) mime_type = _get_mime_type_from_path(path) - if not mime_type: - raise C2paError.NotSupported( - f"Could not determine MIME type for file: {path}") - format_bytes = _validate_and_encode_format( - mime_type, supported, "Reader") + format_bytes = _encode_format(mime_type, "Reader") self._backing_file = open(path, 'rb') self._own_stream = Stream(self._backing_file) elif isinstance(stream, str): - format_bytes = _validate_and_encode_format( - str(format_or_path), supported, "Reader") + format_bytes = _encode_format( + None if format_or_path is None else str(format_or_path), + "Reader") self._backing_file = open(stream, 'rb') self._own_stream = Stream(self._backing_file) else: - format_bytes = _validate_and_encode_format( - str(format_or_path), supported, "Reader") + format_bytes = _encode_format( + None if format_or_path is None else str(format_or_path), + "Reader") self._own_stream = Stream(stream) try: @@ -2394,7 +2431,7 @@ def _get_cached_manifest_data(self) -> Optional[dict]: return self._manifest_data_cache - def with_fragment(self, format: str, stream, + def with_fragment(self, format: Optional[str], stream, fragment_stream) -> "Reader": """Process a BMFF fragment stream with this reader. @@ -2402,7 +2439,9 @@ def with_fragment(self, format: str, stream, content is split into init segments and fragment files. Args: - format: MIME type of the media (e.g., "video/mp4") + format: MIME type of the media (e.g., "video/mp4"). None or an + empty string requests detection from the bytes; the native + library reconciles the format against the detected container. stream: Stream-like object with the main/init segment data fragment_stream: Stream-like object with the fragment data @@ -2414,10 +2453,7 @@ def with_fragment(self, format: str, stream, """ self._ensure_valid_state() - supported = Reader.get_supported_mime_types() - format_bytes = _validate_and_encode_format( - format, supported, "Reader" - ) + format_bytes = _encode_format(format, "Reader") with Stream(stream) as main_obj, Stream(fragment_stream) as frag_obj: new_ptr = _lib.c2pa_reader_with_fragment( @@ -3338,7 +3374,8 @@ def write_ingredient_archive(self, ingredient_id: str, stream: Any) -> None: result = _lib.c2pa_builder_write_ingredient_archive( self._handle, ingredient_id_str, stream_obj._stream) - _check_ffi_operation_result(result, + _check_ffi_operation_result( + result, Builder._ERROR_MESSAGES["archive_error"].format( "Unknown error" ), @@ -3361,7 +3398,8 @@ def add_ingredient_from_archive(self, stream: Any) -> None: result = _lib.c2pa_builder_add_ingredient_from_archive( self._handle, stream_obj._stream) - _check_ffi_operation_result(result, + _check_ffi_operation_result( + result, Builder._ERROR_MESSAGES["archive_read_error"].format( "Unknown error" ), @@ -3414,7 +3452,8 @@ def _sign_internal( are single-sign use. The Builder is closed after signing. Args: - format: The MIME type or extension of the content + format: The MIME type or extension of the content. Required for + signing: a missing format raises C2paError.NotSupported. source_stream: The source stream dest_stream: The destination stream, opened in w+b (write+read binary) mode. @@ -3433,8 +3472,8 @@ def _sign_internal( if not hasattr(signer, '_handle') or not signer._handle: raise C2paError("Invalid or closed signer") - format_bytes = _validate_and_encode_format( - format, Builder.get_supported_mime_types(), "Builder") + format_bytes = _encode_format( + format, "Builder", allow_autodetect=False) manifest_bytes_ptr = ctypes.POINTER(ctypes.c_ubyte)() try: @@ -3647,9 +3686,17 @@ def sign_file( Manifest bytes Raises: + C2paError.NotSupported: If the format cannot be determined from + the source path (extensionless or unrecognized extension), + since signing requires an explicit, resolvable format. C2paError: If there was an error during signing """ mime_type = _get_mime_type_from_path(source_path) + if not mime_type: + raise C2paError.NotSupported( + "Could not determine the format (MIME type) from " + f"'{source_path}'. Sign a stream with an explicit format " + "instead, or use a recognized file extension.") try: with ( @@ -3660,6 +3707,9 @@ def sign_file( return self.sign(signer, mime_type, source_file, dest_file) # else: return self.sign(mime_type, source_file, dest_file) + except C2paError: + # Preserve C2paError and its subtypes + raise except Exception as e: raise C2paError(f"Error signing file: {str(e)}") from e diff --git a/tests/test_unit_tests.py b/tests/test_unit_tests.py index e0580ccb..4f226c38 100644 --- a/tests/test_unit_tests.py +++ b/tests/test_unit_tests.py @@ -30,7 +30,8 @@ from c2pa import Builder, C2paError as Error, Reader, C2paSigningAlg as SigningAlg, C2paSignerInfo, Signer, sdk_version, C2paBuilderIntent, C2paDigitalSourceType from c2pa import Settings, Context, ContextBuilder, ContextProvider -from c2pa.c2pa import Stream, LifecycleState, load_settings, create_signer, create_signer_from_info, ed25519_sign, format_embeddable +from c2pa.c2pa import Stream, LifecycleState, load_settings, create_signer, create_signer_from_info, ed25519_sign, format_embeddable, _get_mime_type_from_path, _encode_format +from pathlib import Path PROJECT_PATH = os.getcwd() @@ -87,6 +88,62 @@ def test_sdk_version(self): self.assertIn(parse_native_version(), sdk_version()) +class TestFormatValidation(unittest.TestCase): + """Unit tests for _encode_format. + + The binding no longer validates the format against a supported list: the + native library is the authority and detects the type from the bytes, so a + given format is trimmed and passed straight through. + """ + + def test_strips_whitespace_from_query(self): + for value in (" image/jpeg", "image/jpeg ", "\timage/jpeg\n"): + self.assertEqual( + _encode_format(value, "Reader"), + b"image/jpeg") + + def test_case_is_preserved(self): + for value in ("IMAGE/JPEG", "Image/Jpeg"): + self.assertEqual( + _encode_format(value, "Reader"), + value.encode("utf-8")) + + def test_encodes_stripped_key(self): + self.assertEqual( + _encode_format(" IMAGE/JPEG ", "Reader"), + b"IMAGE/JPEG") + + def test_production_supported_list_is_normalized(self): + # _get_supported_mime_types strips and lowercases every entry. + for entry in Reader.get_supported_mime_types(): + self.assertEqual(entry, entry.strip().lower()) + + def test_none_autodetects(self): + self.assertEqual(_encode_format(None, "Reader"), b"") + + def test_blank_autodetects(self): + for value in ("", " ", "\t\n"): + self.assertEqual(_encode_format(value, "Reader"), b"") + + def test_missing_format_rejected_when_autodetect_disabled(self): + for value in (None, "", " "): + with self.assertRaises(Error.NotSupported): + _encode_format(value, "Builder", allow_autodetect=False) + + def test_unknown_format_passes_through(self): + # No supported-list check: the format is encoded and handed to the + # native library, which is the authority on validity. + self.assertEqual( + _encode_format("application/x-nope", "Reader"), + b"application/x-nope") + + def test_literal_none_string_passes_through(self): + # The string "None" is a plain string; + # it is trimmed and encoded like any other, + # not rejected in the binding. + self.assertEqual(_encode_format("None", "Reader"), b"None") + + class TestReader(unittest.TestCase): def setUp(self): warnings.filterwarnings("ignore", message="load_settings\\(\\) is deprecated") @@ -251,10 +308,13 @@ def test_stream_read_get_validation_results(self): active_manifest_results = validation_results["activeManifest"] self.assertIsInstance(active_manifest_results, dict) - def test_reader_detects_unsupported_mimetype_on_stream(self): + def test_wrong_mimetype_corrected_from_bytes_on_stream(self): + # The binding passes the format straight through. + # The native library detects the JPEG container and ignores the mismatched format, + # so a valid asset still reads with a wrong (or bogus) mimetype. with open(self.testPath, "rb") as file: - with self.assertRaises(Error.NotSupported): - Reader("mimetype/does-not-exist", file) + with Reader("mimetype/does-not-exist", file) as reader: + self.assertIn(DEFAULT_TEST_FILE_NAME, reader.json()) def test_stream_read_and_parse(self): with open(self.testPath, "rb") as file: @@ -297,34 +357,196 @@ def test_try_create_from_path(self): # Just run and verify there is no crash json.loads(reader.json()) - def test_stream_read_string_stream_mimetype_not_supported(self): - with self.assertRaises(Error.NotSupported): - # xyz is actually an extension that is recognized - # as mimetype chemical/x-xyz - Reader(os.path.join(FIXTURES_DIR, "C.xyz")) + def test_unreadable_asset_raises_not_supported(self): + # A plain-text file is not a recognizable asset. The binding no longer + # rejects it up front; the native library raises NotSupported when it + # fails to detect a container. + with tempfile.TemporaryDirectory() as tmp: + txt = os.path.join(tmp, "C.txt") + with open(txt, "w") as f: + f.write("just some plain text, not an asset\n") + with self.assertRaises(Error.NotSupported): + Reader(txt) - def test_try_create_raises_mimetype_not_supported(self): - with self.assertRaises(Error.NotSupported): - # xyz is actually an extension that is recognized - # as mimetype chemical/x-xyz, but we don't support it - Reader.try_create(os.path.join(FIXTURES_DIR, "C.xyz")) + def test_try_create_unreadable_asset_raises_not_supported(self): + with tempfile.TemporaryDirectory() as tmp: + txt = os.path.join(tmp, "C.txt") + with open(txt, "w") as f: + f.write("just some plain text, not an asset\n") + with self.assertRaises(Error.NotSupported): + Reader.try_create(txt) - def test_stream_read_string_stream_mimetype_not_recognized(self): - with self.assertRaises(Error.NotSupported): - Reader(os.path.join(FIXTURES_DIR, "C.test")) + def test_none_format_matches_empty_string_on_stream(self): + # None and "" both request auto-detection and read the same asset. + with open(self.testPath, "rb") as file: + with Reader(None, file) as reader: + none_json = reader.json() + with open(self.testPath, "rb") as file: + with Reader("", file) as reader: + empty_json = reader.json() + self.assertEqual(none_json, empty_json) + self.assertNotIn("None", none_json[:64]) - def test_try_create_raises_mimetype_not_recognized(self): - with self.assertRaises(Error.NotSupported): - Reader.try_create(os.path.join(FIXTURES_DIR, "C.test")) + def test_padded_format_accepted_on_stream(self): + with open(self.testPath, "rb") as file: + with Reader(" image/jpeg ", file) as reader: + self.assertIn(DEFAULT_TEST_FILE_NAME, reader.json()) + + def test_try_create_none_format_matches_empty_string(self): + with open(self.testPath, "rb") as file: + none_reader = Reader.try_create(None, file) + self.assertIsNotNone(none_reader) + with open(self.testPath, "rb") as file: + empty_reader = Reader.try_create("", file) + self.assertIsNotNone(empty_reader) + if none_reader is not None and empty_reader is not None: + self.assertEqual(none_reader.json(), empty_reader.json()) + + def test_unrecognized_extension_defers_to_detection(self): + with tempfile.TemporaryDirectory() as tmp: + asset = os.path.join(tmp, "C.test") + shutil.copyfile(self.testPath, asset) + with Reader(asset) as reader: + self.assertIn(DEFAULT_TEST_FILE_NAME, reader.json()) + + def test_try_create_unrecognized_extension_defers_to_detection(self): + with tempfile.TemporaryDirectory() as tmp: + asset = os.path.join(tmp, "C.test") + shutil.copyfile(self.testPath, asset) + reader = Reader.try_create(asset) + self.assertIsNotNone(reader) + if reader is not None: + with reader: + self.assertIn(DEFAULT_TEST_FILE_NAME, reader.json()) def test_stream_read_string_stream(self): with Reader("image/jpeg", self.testPath) as reader: json_data = reader.json() self.assertIn(DEFAULT_TEST_FILE_NAME, json_data) - def test_reader_detects_unsupported_mimetype_on_file(self): - with self.assertRaises(Error.NotSupported): - Reader("mimetype/does-not-exist", self.testPath) + def test_wrong_mimetype_corrected_from_bytes_on_file(self): + # As with a stream: a valid asset opened by path still reads when the + # given mimetype is wrong, because the native library corrects it from + # the detected container. + with Reader("mimetype/does-not-exist", self.testPath) as reader: + self.assertIn(DEFAULT_TEST_FILE_NAME, reader.json()) + + def test_read_extensionless_file_path(self): + # Extensionless file triggers type auto-detection. + with tempfile.TemporaryDirectory() as tmp: + no_ext = os.path.join(tmp, "C") + shutil.copyfile(self.testPath, no_ext) + with Reader(no_ext) as reader: + self.assertIn(DEFAULT_TEST_FILE_NAME, reader.json()) + + def test_read_extensionless_stream_empty_format(self): + # An empty format string on a stream triggers auto-detection. + with open(self.testPath, "rb") as file: + with Reader("", file) as reader: + self.assertIn(DEFAULT_TEST_FILE_NAME, reader.json()) + + def test_read_wrong_extension_corrected(self): + # A JPEG saved with a .png extension is corrected from the bytes. + # (Read is possible and does not fail due to wrong type). + with tempfile.TemporaryDirectory() as tmp: + wrong = os.path.join(tmp, "C.png") + shutil.copyfile(self.testPath, wrong) + with Reader(wrong) as reader: + self.assertIn(DEFAULT_TEST_FILE_NAME, reader.json()) + + def test_read_wrong_mime_corrected_on_stream(self): + # A wrong MIME hint on a stream is corrected from the bytes. + with open(self.testPath, "rb") as file: + with Reader("image/png", file) as reader: + self.assertIn(DEFAULT_TEST_FILE_NAME, reader.json()) + + def test_explicit_empty_format_forces_detection(self): + # Passing "" works even when the correct format is supported. + with open(self.testPath, "rb") as file: + with Reader("", file) as reader: + manifest_store = json.loads(reader.json()) + self.assertIn("active_manifest", manifest_store) + + def test_autodetect_multiformat(self): + fixtures = [ + os.path.join(FIXTURES_DIR, "files-for-reading-tests", "CA.jpg"), + os.path.join(FIXTURES_DIR, "video1.mp4"), + os.path.join(FIXTURES_DIR, "files-for-reading-tests", "pdf-file.pdf"), + os.path.join(FIXTURES_DIR, "files-for-reading-tests", "sample1_signed.wav"), + ] + for path in fixtures: + with self.subTest(path=path): + with open(path, "rb") as file: + with Reader("", file) as reader: + self.assertIn("active_manifest", json.loads(reader.json())) + + def test_dng_extension_still_special_cased(self): + dng_path = os.path.join(FIXTURES_DIR, "C.dng") + with Reader(dng_path) as reader: + self.assertIn("active_manifest", json.loads(reader.json())) + + def test_garbage_stream_empty_format_raises(self): + # Unrecognized bytes with auto-detection raise rather than crash. + with self.assertRaises(Error): + Reader("", io.BytesIO(b"not an asset at all")) + + def test_empty_stream_raises(self): + # A zero-byte stream cannot be detected and raises. + with self.assertRaises(Error): + Reader("", io.BytesIO(b"")) + + def test_truncated_corrupt_jpeg_raises(self): + with open(self.testPath, "rb") as file: + data = bytearray(file.read()) + # Corrupt the body while keeping the leading JPEG magic intact. + # This will still raise errors. + for i in range(64, min(len(data), 4096)): + data[i] ^= 0xFF + with self.assertRaises(Error): + Reader("", io.BytesIO(bytes(data))) + + def test_garbage_extensionless_file_raises(self): + # Random bytes in an extensionless file raise. + with tempfile.TemporaryDirectory() as tmp: + garbage = os.path.join(tmp, "garbage") + with open(garbage, "wb") as f: + f.write(b"this is not a media asset") + with self.assertRaises(Error): + Reader(garbage) + + def test_try_create_garbage_stream_raises(self): + with self.assertRaises(Error) as context: + Reader.try_create("", io.BytesIO(b"not an asset at all")) + self.assertNotIn("ManifestNotFound", str(context.exception)) + + def test_try_create_extensionless_no_manifest_returns_none(self): + unsigned = os.path.join( + FIXTURES_DIR, "files-for-signing-tests", "earth_apollo17.jpg") + with tempfile.TemporaryDirectory() as tmp: + no_ext = os.path.join(tmp, "unsigned") + shutil.copyfile(unsigned, no_ext) + self.assertIsNone(Reader.try_create(no_ext)) + + def test_sub_magic_length_stream_raises(self): + # Fewer bytes than a full magic signature cannot be detected. + # This will raise too. + with self.assertRaises(Error): + Reader("", io.BytesIO(b"\xff\xd8\xff")) + + def test_preseeked_stream_is_rewound(self): + with open(self.testPath, "rb") as file: + data = file.read() + stream = io.BytesIO(data) + stream.seek(len(data) // 2) + with Reader("", stream) as reader: + self.assertIn(DEFAULT_TEST_FILE_NAME, reader.json()) + + def test_dng_extension_with_jpeg_bytes_corrected(self): + with tempfile.TemporaryDirectory() as tmp: + fake_dng = os.path.join(tmp, "fake.dng") + shutil.copyfile(self.testPath, fake_dng) + with Reader(fake_dng) as reader: + self.assertIn(DEFAULT_TEST_FILE_NAME, reader.json()) def test_stream_read_filepath_as_stream_and_parse(self): with Reader("image/jpeg", self.testPath) as reader: @@ -332,6 +554,64 @@ def test_stream_read_filepath_as_stream_and_parse(self): title = manifest_store["manifests"][manifest_store["active_manifest"]]["title"] self.assertEqual(title, DEFAULT_TEST_FILE_NAME) + def test_get_mime_type_from_path_dng_special_cased(self): + self.assertEqual(_get_mime_type_from_path("photo.dng"), "image/dng") + + def test_get_mime_type_from_path_uppercase_dng(self): + self.assertEqual(_get_mime_type_from_path("PHOTO.DNG"), "image/dng") + + def test_get_mime_type_from_path_jpeg(self): + self.assertEqual(_get_mime_type_from_path("photo.jpg"), "image/jpeg") + + def test_get_mime_type_from_path_unknown_extension_returns_empty(self): + # An unrecognized extension yields "" so the caller can auto-detect. + self.assertEqual(_get_mime_type_from_path("asset.unknownext"), "") + + def test_get_mime_type_from_path_no_suffix_returns_empty(self): + self.assertEqual(_get_mime_type_from_path("asset"), "") + + def test_get_mime_type_from_path_multi_dot_uses_final_suffix(self): + # Only the final suffix determines the type. + self.assertEqual( + _get_mime_type_from_path("photo.final.jpg"), "image/jpeg") + + def test_get_mime_type_from_path_compound_extension(self): + # mimetypes treats ".gz" as an encoding, not a type suffix, + # the type is derived from ".tar" + result = _get_mime_type_from_path("archive.tar.gz") + self.assertIsInstance(result, str) + self.assertNotIn(result, ("image/jpeg", "image/dng")) + + def test_get_mime_type_from_path_hidden_dotfile_returns_empty(self): + # Check how we handle files without suffix (usually hidden files) + self.assertEqual(_get_mime_type_from_path(".gitignore"), "") + + def test_get_mime_type_from_path_accepts_path_object(self): + self.assertEqual( + _get_mime_type_from_path(Path("dir/photo.jpg")), "image/jpeg") + + def test_encode_format_blank_autodetects(self): + # Reader default: a blank format requests auto-detection. + self.assertEqual(_encode_format("", "Reader"), b"") + self.assertEqual(_encode_format(" ", "Reader"), b"") + + def test_encode_format_returns_bytes(self): + self.assertEqual( + _encode_format("image/jpeg", "Reader"), b"image/jpeg") + + def test_encode_format_case_preserved(self): + self.assertEqual( + _encode_format("IMAGE/JPEG", "Reader"), b"IMAGE/JPEG") + + def test_encode_format_unknown_passes_through(self): + # The native library decides validity. + self.assertEqual( + _encode_format("application/zip", "Reader"), b"application/zip") + + def test_reader_accepts_path_object(self): + with Reader(Path(self.testPath)) as reader: + self.assertIn(DEFAULT_TEST_FILE_NAME, reader.json()) + def test_reader_double_close(self): with open(self.testPath, "rb") as file: reader = Reader("image/jpeg", file) @@ -1162,6 +1442,22 @@ def test_builder_does_not_allow_sign_after_close(self): with self.assertRaises(Error): builder.sign(self.signer, "image/jpeg", file, output) + def test_sign_rejects_none_format(self): + # sign requires an explicit format; None must raise NotSupported. + with open(self.testPath, "rb") as file: + builder = Builder(self.manifestDefinition) + output = io.BytesIO(bytearray()) + with self.assertRaises(Error.NotSupported): + builder.sign(self.signer, None, file, output) + + def test_sign_accepts_padded_format(self): + with open(self.testPath, "rb") as file: + builder = Builder(self.manifestDefinition) + output = io.BytesIO(bytearray()) + manifest_bytes = builder.sign( + self.signer, " image/jpeg ", file, output) + self.assertTrue(len(manifest_bytes) > 0) + def test_builder_does_not_allow_archiving_after_close(self): with open(self.testPath, "rb") as file: builder = Builder(self.manifestDefinition) @@ -3331,6 +3627,44 @@ def test_sign_mov_video_file_single(self): self.assertIn("Valid", json_data) output.close() + def test_encode_format_builder_blank_raises(self): + with self.assertRaises(Error.NotSupported): + _encode_format("", "Builder", allow_autodetect=False) + + def test_sign_empty_format_raises(self): + builder = Builder(self.manifestDefinition) + with open(self.testPath, "rb") as file: + output = io.BytesIO() + with self.assertRaises(Error.NotSupported): + builder.sign(self.signer, "", file, output) + + def test_sign_file_extensionless_source_raises(self): + builder = Builder(self.manifestDefinition) + with tempfile.TemporaryDirectory() as tmp: + no_ext = os.path.join(tmp, "source") + shutil.copyfile(self.testPath, no_ext) + dest = os.path.join(tmp, "out.jpg") + with self.assertRaises(Error.NotSupported): + builder.sign_file(no_ext, dest, self.signer) + + def test_sign_file_unrecognized_extension_raises(self): + builder = Builder(self.manifestDefinition) + with tempfile.TemporaryDirectory() as tmp: + weird = os.path.join(tmp, "source.unknownextension") + shutil.copyfile(self.testPath, weird) + dest = os.path.join(tmp, "out.jpg") + with self.assertRaises(Error.NotSupported): + builder.sign_file(weird, dest, self.signer) + + def test_sign_file_exceptions_preservation(self): + builder = Builder(self.manifestDefinition) + with tempfile.TemporaryDirectory() as tmp: + src = os.path.join(tmp, "source.txt") + shutil.copyfile(self.testPath, src) + dest = os.path.join(tmp, "out.txt") + with self.assertRaises(Error.NotSupported): + builder.sign_file(src, dest, self.signer) + def test_sign_file_video(self): temp_dir = tempfile.mkdtemp() try: @@ -5997,6 +6331,16 @@ def test_reader_file_path_with_context(self): reader.close() context.close() + def test_reader_extensionless_path_with_context_autodetects(self): + context = Context() + with tempfile.TemporaryDirectory() as tmp: + no_ext = os.path.join(tmp, "asset") + shutil.copyfile(DEFAULT_TEST_FILE, no_ext) + reader = Reader(no_ext, context=context) + self.assertIn(DEFAULT_TEST_FILE_NAME, reader.json()) + reader.close() + context.close() + def test_reader_format_and_path_with_ctx(self): context = Context() reader = Reader("image/jpeg", DEFAULT_TEST_FILE, context=context)