From 1b70ac6704021e3b25cd658c45c77bb45f85f9e1 Mon Sep 17 00:00:00 2001 From: Richard Tibbles Date: Sat, 6 Jun 2026 19:15:26 -0700 Subject: [PATCH 1/2] Always generate thumbnails for content nodes and topics. Add a thumbnail extraction stage to the file pipeline (pdf, epub, html5/kpub zips, mp4, webm) that emits a PNG FileMetadata alongside the source file, skipped via the NODE_HAS_THUMBNAIL context key when the node already provides a thumbnail, and for decomposed packages whose leaves generate their own. HTML5/KPUB zips render via headless Chromium and fall back to the biggest-image heuristic, shared with the legacy extractor through create_image_from_html_zip. Thumbnail detection becomes preset-based (File.is_thumbnail, Node.has_thumbnail) instead of isinstance checks, so pipeline-generated thumbnails count. Node-level generation is extracted into generate_missing_thumbnail(), called from process_files for content nodes; topics defer to a new sequential ChannelManager.generate_deferred_thumbnails() post-pass, fixing the latent race where the concurrent executor could tile a topic before its children finished. Breaking change: the config.THUMBNAILS gate and derive_thumbnail node kwarg are removed - generation is now always on for nodes without a provided thumbnail. The --thumbnails CLI flag and the thumbnails / generate-missing-thumbnails settings remain as deprecated no-ops that emit a warning. Also: write_file no longer masks in-flight exceptions or copies partial files to storage; create_image_from_pdf_page gains max_width to cap render size (downscale-only: pages that would render narrower than the cap are not upscaled, with page width read via PyPDF2); guard get_thumbnail_preset against kind-less nodes matching the channel_thumbnail preset. Co-Authored-By: Claude Opus 4.8 (1M context) Co-Authored-By: Claude Fable 5 Co-Authored-By: Claude Opus 5.5 (1M context) Claude-Session: https://claude.ai/code/session_018JvU4HeNz2k64GnG9twYct --- docs/chefops.md | 14 +- docs/developer/design_cli.md | 2 +- docs/developer/uploadprocess.md | 9 +- docs/examples/languages.ipynb | 1 - docs/index.rst | 3 +- docs/nodes.md | 15 +- ricecooker/chefs.py | 65 +++-- ricecooker/classes/files.py | 82 ++++--- ricecooker/classes/nodes.py | 83 ++++--- ricecooker/commands.py | 3 - ricecooker/config.py | 1 - ricecooker/managers/tree.py | 29 ++- ricecooker/utils/images.py | 59 ++++- ricecooker/utils/jsontrees.py | 8 - ricecooker/utils/linecook.py | 5 - ricecooker/utils/pipeline/__init__.py | 9 +- ricecooker/utils/pipeline/context.py | 4 + ricecooker/utils/pipeline/file_handler.py | 23 +- ricecooker/utils/pipeline/thumbnails.py | 103 ++++++++ tests/media_utils/test_thumbnails.py | 39 +++ tests/pipeline/test_convert.py | 36 ++- tests/pipeline/test_file_handler.py | 27 +- tests/pipeline/test_thumbnails.py | 146 +++++++++++ tests/test_argparse.py | 10 +- tests/test_files.py | 1 + tests/test_settings.py | 28 ++- tests/test_thumbnails.py | 286 ++++++++++++++++++++-- tests/test_tree.py | 5 +- 28 files changed, 906 insertions(+), 190 deletions(-) create mode 100644 ricecooker/utils/pipeline/thumbnails.py create mode 100644 tests/pipeline/test_thumbnails.py diff --git a/docs/chefops.md b/docs/chefops.md index c562c581..4733389e 100644 --- a/docs/chefops.md +++ b/docs/chefops.md @@ -10,7 +10,7 @@ Ricecooker CLI This listing shows the `ricecooker` command line interface (CLI) arguments: usage: sushichef.py [-h] [--token TOKEN] [-u] [--debug] [-v] [--warn] - [--quiet] [--compress] [--thumbnails] + [--quiet] [--compress] [--download-attempts DOWNLOAD_ATTEMPTS] [--prompt] [--deploy] [--publish] [--sample SIZE] @@ -26,7 +26,6 @@ This listing shows the `ricecooker` command line interface (CLI) arguments: --warn Print errors and warnings. --quiet Print only errors. --compress Compress videos using ffmpeg -crf=32 -b:a 32k mono. - --thumbnails Automatically generate thumbnails for content nodes. --download-attempts N Maximum number of times to retry downloading files (default: 3). --prompt Prompt user to open the channel after the chef run. --deploy Immediately deploy changes to channel's main tree. @@ -42,12 +41,11 @@ the complete list: you'll have to run `./sushichef.py -h` to see the latest vers Below is a short guide to some of the most important and useful ones arguments. -### Compression and thumbnail globals -You can specify video compression settings (see this page) and thumbnails for -specific nodes and files in the channel, or use `--compress` and `--thumbnails` -to apply compression to ALL videos, and automatically generate thumbnails for -all the supported content kinds. **We recommend you always use the `--thumbnails`** -in order to create more colorful, lively channels that learners will want to browse. +### Compression globals +You can specify video compression settings for specific nodes and files in the +channel, or use `--compress` to apply compression to ALL videos. +Thumbnails are always generated automatically for any content node or topic that +doesn't have a thumbnail provided — no flag is needed. ### Caching diff --git a/docs/developer/design_cli.md b/docs/developer/design_cli.md index 0e980598..6f4691ac 100644 --- a/docs/developer/design_cli.md +++ b/docs/developer/design_cli.md @@ -113,7 +113,7 @@ There are three types of arguments involved in a chef run: - `args` (dict): command line args as parsed by the sushi chef class and its parents - SushiChef: the `SushiChef.__init__` method configures argparse for the following: - `compress`, `download_attempts`, `prompt`, `publish`, - `stage`, `thumbnails`, `token`, `update`, `verbose`, `warn` + `stage`, `token`, `update`, `verbose`, `warn` - MySushiChef: the chef's `__init__` method can define additional cli args - `options` (dict): additional [OPTIONS...] passed at the end of the command line diff --git a/docs/developer/uploadprocess.md b/docs/developer/uploadprocess.md index 2a76be03..9f400aa5 100644 --- a/docs/developer/uploadprocess.md +++ b/docs/developer/uploadprocess.md @@ -44,10 +44,11 @@ Each `Node` subclass implements the `process_files` method which includes the following steps: - call `process_file` on all files associated with the node (described below) - if the node has children, `process_files` is called on all child nodes - - call the node's `generate_thumbnail` method if it doesn't have a thumbnail - already, and the node has `derive_thumbnail` set to True, or if the global - command line argument `--thumbnail` (config.THUMBNAILS) is set to True. - See notes section "Node.generate_thumbnail". + - for content nodes: call `generate_missing_thumbnail` when no thumbnail has been + provided, generating one automatically from the node's content files. + Topic nodes skip this step; their thumbnails are generated in a separate + post-pass by `ChannelManager.generate_deferred_thumbnails()`, which runs after + all nodes have been processed and tiles the thumbnails of descendant content nodes. The result of the `node.process_file()` is a list of processed filenames, that reference files in the content-addressable storage directory `/content/storage/`. diff --git a/docs/examples/languages.ipynb b/docs/examples/languages.ipynb index 1b20031a..2bf5e299 100644 --- a/docs/examples/languages.ipynb +++ b/docs/examples/languages.ipynb @@ -594,7 +594,6 @@ " source_id=youtube_id,\n", " title='Youtube video',\n", " license=TE_LICENSE,\n", - " derive_thumbnail=True,\n", " uri='https://www.youtube.com/watch?v={}'.format(youtube_id),\n", " )\n", "\n", diff --git a/docs/index.rst b/docs/index.rst index 706a2532..3acd7632 100644 --- a/docs/index.rst +++ b/docs/index.rst @@ -152,8 +152,7 @@ Use the links below to jump to the specific topics that you want to learn about. `Command line interface `_ ******************************************************* -Use command line options like ``--thumbnails`` (auto-generating thumbnails), -and ``--compress`` (video compression) to create better channels. +Use command line options like ``--compress`` (video compression) to create better channels. .. rst-class:: column column3 diff --git a/docs/nodes.md b/docs/nodes.md index 31a32494..d78928a6 100644 --- a/docs/nodes.md +++ b/docs/nodes.md @@ -69,7 +69,6 @@ Each node has the following attributes: - __learner_needs__ (list[str]): Specific learning needs the content is suitable for. Expects a list of constants from `le_utils.constants.labels.needs`. Example: `from le_utils.constants.labels import needs; node.learner_needs = [needs.PRIOR_KNOWLEDGE_REMEDIATION]`. (optional) - __accessibility_labels__ (list[str]): Accessibility features provided by the content. Expects a list of constants from `le_utils.constants.labels.accessibility_categories`. Example: `from le_utils.constants.labels import accessibility_categories; node.accessibility_labels = [accessibility_categories.CAPTIONS_SUBTITLES]`. (optional) - __thumbnail__ (str or ThumbnailFile): path to thumbnail or file object (optional) - - __derive_thumbnail__ (bool): set to True to generate thumbnail from contents (optional) - __files__ ([FileObject]): list of file objects for node (optional) - __extra_fields__ (dict): any additional data needed for node (optional) - __domain_ns__ (uuid): who is providing the content (e.g. learningequality.org) (optional) @@ -141,8 +140,7 @@ See [languages][./languages.md] to read more about language codes. Thumbnails can be passed in as a local filesystem path to an image file (str), a URL (str), or a `ThumbnailFile` object. The recommended size for thumbnail images is 400px by 225px (aspect ratio 16:9). -Use the command line argument `--thumbnails` to automatically generate thumbnails -for all content node that don't have a thumbnail specified. +Thumbnails are automatically generated for all content nodes and topics that don't have a thumbnail specified. @@ -165,9 +163,8 @@ Topic nodes are folder-like containers that are used to organize the channel's c It is highly recommended to find suitable thumbnail images for topic nodes. The presence of thumbnails will make the content more appealing and easier to browse. -Set `derive_thumbnail=True` on a topic node or use the `--thumbnails` command -line argument and Ricecooker will generate thumbnails for topic nodes based on -the thumbnails of the content nodes they contain. +When no thumbnail is provided for a topic node, Ricecooker automatically generates +a tiled thumbnail by compositing the thumbnails of its descendant content nodes. @@ -187,9 +184,9 @@ For your copy-paste convenience, here is the sample code for creating a content See the [Overview](#overview) above for how the pipeline infers kind, preset, and files from `uri`. -Specify `derive_thumbnail=True` and leave thumbnail blank (`thumbnail=None`) to -let Ricecooker automatically generate a thumbnail for the node based on its content. -Thumbnail generation is supported for audio, video, PDF, and ePub, and HTML5 files. +Leaving `thumbnail=None` (the default) causes Ricecooker to automatically generate +a thumbnail for the node based on its content. +Thumbnail generation is supported for video, PDF, ePub, and HTML5 files. diff --git a/ricecooker/chefs.py b/ricecooker/chefs.py index d263b090..83658d61 100644 --- a/ricecooker/chefs.py +++ b/ricecooker/chefs.py @@ -71,20 +71,23 @@ def __init__(self, *args, **kwargs): if not hasattr(self, "SETTINGS"): self.SETTINGS = {} else: - if "generate-missing-thumbnails" in self.SETTINGS: - warning_text = "thumbnails setting is deprecated and will be replaced by thumbnails in version 0.8 please update" - config.LOGGER.warn(warning_text) - warn(warning_text, DeprecationWarning) - self.SETTINGS["thumbnails"] = self.SETTINGS[ - "generate-missing-thumbnails" - ] - if "compress-videos" in self.SETTINGS: warning_text = "compress-videos setting is deprecated and will be replaced by compress in version 0.8 please update" config.LOGGER.warn(warning_text) warn(warning_text, DeprecationWarning) self.SETTINGS["compress"] = self.SETTINGS["compress-videos"] + for thumbnail_setting in ("thumbnails", "generate-missing-thumbnails"): + if thumbnail_setting in self.SETTINGS: + warning_text = ( + "{} setting is deprecated and is ignored: thumbnails are now always " + "generated for nodes that do not provide one".format( + thumbnail_setting + ) + ) + config.LOGGER.warning(warning_text) + warn(warning_text, DeprecationWarning) + # these will be assigned to later by the argparse handling. self.args = None self.options = None @@ -138,11 +141,6 @@ def __init__(self, *args, **kwargs): action="store_true", help="Compress videos using ffmpeg -crf=32 -b:a 32k mono.", ) - parser.add_argument( - "--thumbnails", - action="store_true", - help="Automatically generate thumbnails for content nodes.", - ) parser.add_argument( "--download-attempts", type=int, @@ -186,6 +184,15 @@ def __init__(self, *args, **kwargs): " Uploading a staging tree is now the default behavior. Use --deploy to upload to the main tree." ), ) + parser.add_argument( + "--thumbnails", + dest="thumbnails_deprecated", + action="store_true", + help=( + "(deprecated) Thumbnails are now always generated" + " for content nodes and topics that do not provide one." + ), + ) parser.add_argument( "--remote", nargs="?", @@ -232,10 +239,7 @@ def get_setting(self, setting, default=None): override = None # If there is a command line flag for this setting, allow for it to override the chef - # default. Note that these are all boolean flags, so they are true if set, false if not. - if setting == "thumbnails": - override = self.args and self.args["thumbnails"] - + # default. Note that the compress flag is a boolean: true if set on the command line, false if not. if setting == "compress": override = self.args and self.args["compress"] @@ -275,15 +279,7 @@ def parse_args_and_options(self): if args["command"] == "remote": raise InvalidUsageException("remote must be the first argument.") - # Print CLI deprecation warnings info - if args["stage_deprecated"]: - config.LOGGER.warning( - "DEPRECATION WARNING: --stage is now the default bevavior. The --stage flag has been deprecated and will be removed in ricecooker 1.0." - ) - if args["reset_deprecated"]: - config.LOGGER.warning( - "DEPRECATION WARNING: --reset is now the default bevavior. The --reset flag has been deprecated and will be removed in ricecooker 1.0." - ) + self._warn_deprecated_flags(args) if args["publish"] and args["stage"]: raise InvalidUsageException( "The --publish argument must be used together with --deploy argument." @@ -318,6 +314,23 @@ def parse_args_and_options(self): return args, options + DEPRECATED_FLAG_WARNINGS = { + "stage_deprecated": "--stage is now the default bevavior. The --stage flag", + "reset_deprecated": "--reset is now the default bevavior. The --reset flag", + "thumbnails_deprecated": ( + "thumbnails are now always generated for content nodes and topics that do not provide one. The --thumbnails flag" + ), + } + + def _warn_deprecated_flags(self, args): + for arg, message in self.DEPRECATED_FLAG_WARNINGS.items(): + if args[arg]: + config.LOGGER.warning( + "DEPRECATION WARNING: {} has been deprecated and will be removed in ricecooker 1.0.".format( + message + ) + ) + def _resolve_token(self, args): remote = args["remote"] is not None if remote or args["command"] == "uploadchannel": diff --git a/ricecooker/classes/files.py b/ricecooker/classes/files.py index 26b99a4a..c1d3937e 100644 --- a/ricecooker/classes/files.py +++ b/ricecooker/classes/files.py @@ -10,15 +10,14 @@ from ricecooker.utils.caching import FILECACHE from ricecooker.utils.caching import get_cache_filename -from ricecooker.utils.images import ChromiumUnavailableError from ricecooker.utils.images import create_image_from_epub +from ricecooker.utils.images import create_image_from_html_zip from ricecooker.utils.images import create_image_from_pdf_page -from ricecooker.utils.images import create_image_from_zip -from ricecooker.utils.images import create_image_from_zip_screenshot from ricecooker.utils.images import create_tiled_image from ricecooker.utils.images import ThumbnailGenerationError from ricecooker.utils.paths import extract_path_ext from ricecooker.utils.pipeline import FilePipeline +from ricecooker.utils.pipeline.context import NODE_HAS_THUMBNAIL from ricecooker.utils.pipeline.convert import AudioCompressionHandler from ricecooker.utils.pipeline.convert import ImageConversionHandler from ricecooker.utils.pipeline.convert import SubtitleConversionHandler @@ -31,21 +30,22 @@ from ricecooker.utils.youtube import get_language_with_alpha2_fallback from .. import config -from ..exceptions import UnknownFileTypeError fallback_pipeline = FilePipeline() # Lookup table for convertible file formats for a given preset # used for converting avi/flv/etc. videos and srt subtitles CONVERTIBLE_FORMATS = {p.id: p.convertible_formats for p in format_presets.PRESETLIST} +PRESET_LOOKUP = {p.id: p for p in format_presets.PRESETLIST} class ThumbnailPresetMixin(object): def get_preset(self): - thumbnail_preset = self.node.get_thumbnail_preset() - if thumbnail_preset is None: - UnknownFileTypeError("Thumbnails are not supported for node kind.") - return thumbnail_preset + # May return None: the node's kind may not be known yet (a uri-based + # node before the pipeline has run), or may have no thumbnail preset + # (e.g. StudioContentNode, whose serialized thumbnail overrides rely + # on a None preset being resolved by Studio). + return self.node.get_thumbnail_preset() class File(object): @@ -105,6 +105,27 @@ def get_preset(self): f"object for file {self.filename} ({self.original_filename})" ) + def is_thumbnail(self): + """ + Whether this file is a thumbnail, based on its format preset. + Returns False when the preset cannot be resolved (a ThumbnailFile not yet attached to a node, or a File with no preset). + """ + try: + # For ThumbnailPresetMixin files this may resolve to None (kind + # not yet known, or no thumbnail preset for the kind), which the + # PRESET_LOOKUP check below maps to False. + preset = self.get_preset() + except NotImplementedError: + # File with no preset and no default_preset. + return False + except AttributeError: + if self.node is None: + # ThumbnailFile (ThumbnailPresetMixin) not yet attached to a node. + return False + raise + preset_obj = PRESET_LOOKUP.get(preset) + return bool(preset_obj and preset_obj.thumbnail) + def get_filename(self): return self.filename or self.process_file() @@ -216,8 +237,14 @@ def process_file(self): except ValueError as ve: raise InvalidFileException from ve pipeline = config.FILE_PIPELINE or fallback_pipeline + # This legacy path only consumes the first (source) metadata entry, + # and thumbnails for legacy nodes are generated at the node level + # (generate_missing_thumbnail), so always skip the pipeline + # thumbnail stage rather than generate an image only to discard it. + context = dict(self.context) + context[NODE_HAS_THUMBNAIL] = True metadata = pipeline.execute( - self.path, context=self.context, skip_cache=config.UPDATE + self.path, context=context, skip_cache=config.UPDATE )[0] metadata = metadata.to_dict() for key in metadata: @@ -550,7 +577,7 @@ def extractor_fun(self, fpath_in, thumbpath_out, **kwargs): class ExtractedPdfThumbnailFile(ExtractedThumbnailFile): - extractor_kwargs = {"page_number": 0, "crop": None} + extractor_kwargs = {"page_number": 0, "crop": None, "max_width": 1000} allowed_formats = DocumentFile.allowed_formats def extractor_fun(self, fpath_in, thumbpath_out, **kwargs): @@ -570,17 +597,7 @@ class ExtractedHTMLZipThumbnailFile(ExtractedThumbnailFile): allowed_formats = HTMLZipFile.allowed_formats def extractor_fun(self, fpath_in, thumbpath_out, **kwargs): - try: - create_image_from_zip_screenshot(fpath_in, thumbpath_out, **kwargs) - return - except ChromiumUnavailableError: - pass - except ThumbnailGenerationError as err: - config.LOGGER.warning( - "\t Screenshot render failed, falling back to biggest-image " - "heuristic: {}".format(err) - ) - create_image_from_zip(fpath_in, thumbpath_out, **kwargs) + create_image_from_html_zip(fpath_in, thumbpath_out, **kwargs) class ExtractedKPUBThumbnailFile(ExtractedHTMLZipThumbnailFile): @@ -601,16 +618,25 @@ class TiledThumbnailFile(ThumbnailPresetMixin, File): def __init__(self, source_nodes, **kwargs): self.sources = [] for n in source_nodes: - images = [ - f for f in n.files if isinstance(f, ThumbnailFile) and f.get_filename() - ] - if len(images) > 0: - self.sources.append(images[0]) + # Check f.filename directly (not get_filename()) so a thumbnail + # that failed to process is skipped rather than re-processed here. + image = next((f for f in n.files if f.is_thumbnail() and f.filename), None) + if image: + self.sources.append(image) super(TiledThumbnailFile, self).__init__(**kwargs) def process_file(self): - self.filename = self.generate_tiled_image() - config.LOGGER.info("\t--- Tiled image {}".format(self.filename)) + try: + self.filename = self.generate_tiled_image() + if self.filename: + config.LOGGER.info("\t--- Tiled image {}".format(self.filename)) + else: + config.LOGGER.info("\t--- No source thumbnails to tile") + except ThumbnailGenerationError as err: + config.LOGGER.warning("\t Failed to generate tiled image {}".format(err)) + self.filename = None + self.error = str(err) + config.FAILED_FILES.append(self) return self.filename def generate_tiled_image(self): diff --git a/ricecooker/classes/nodes.py b/ricecooker/classes/nodes.py index d7c49ed5..078f0a0a 100644 --- a/ricecooker/classes/nodes.py +++ b/ricecooker/classes/nodes.py @@ -19,6 +19,7 @@ from le_utils.constants.labels import resource_type from le_utils.constants.labels import subjects +from ricecooker.utils.pipeline.context import NODE_HAS_THUMBNAIL from ricecooker.utils.pipeline.exceptions import ExpectedFileException from ricecooker.utils.pipeline.exceptions import InvalidFileException @@ -33,7 +34,9 @@ from .files import ExtractedKPUBThumbnailFile from .files import ExtractedPdfThumbnailFile from .files import File +from .files import PRESET_LOOKUP from .files import SubtitleFile +from .files import ThumbnailFile from .files import YouTubeSubtitleFile from .licenses import License from .questions import QTIQuestion @@ -43,7 +46,6 @@ MASTERY_MODELS = [id for id, name in exercises.MASTERY_MODELS] ROLES = [id for id, name in roles.choices] EXERCISE_SETTING_KEYS = ("mastery_model", "m", "n", "randomize", "options") -PRESET_LOOKUP = {p.id: p for p in format_presets.PRESETLIST} # Used to map content kind to learning activity when a list of learning_activiies is not provided @@ -102,7 +104,6 @@ class Node(object): description (str): description of the content (optional) thumbnail (str or ThumbnailFile): local path, url, or ThumbnailFile for thumbnail (optional) files ([File]): list of File objects for the node (optional) - derive_thumbnail (bool): whether to auto-generate thumbnail from content (default False) node_modifications (dict): modifications passed in by CSV import (optional) extra_fields (dict): additional data needed for the node (optional) suggested_duration (int): suggested duration in seconds (optional) @@ -112,6 +113,10 @@ class Node(object): license = None language = None valid = False + # When True, thumbnail generation is deferred to the tree manager's + # post-pass, after all descendants have been processed (topics need + # their children's thumbnails to build a tiled image). + defer_thumbnail_generation = False # Nodes with questions (ExerciseNode, UnitNode) set this True. # This gates Node._validate() to allow a non-empty `questions` list, # which TreeNode.to_dict() serializes for the Studio API. @@ -140,7 +145,6 @@ def __init__( description=None, thumbnail=None, files=None, - derive_thumbnail=False, node_modifications={}, extra_fields=None, suggested_duration=None, @@ -157,7 +161,6 @@ def __init__( self.title = title self.set_language(language) self.description = description or "" - self.derive_thumbnail = derive_thumbnail self.extra_fields = extra_fields or {} for f in files or []: @@ -322,10 +325,13 @@ def generate_thumbnail(self): return None def has_thumbnail(self): - from .files import ThumbnailFile - - return any(f for f in self.files if isinstance(f, ThumbnailFile)) - # TODO deep check: f.process_file() and check f.filename is not None + """ + Whether this node already has a thumbnail: either one was provided + explicitly (self.thumbnail), or one of its files has a thumbnail + format preset. The explicit self.thumbnail check handles uri-based nodes, + which have no kind until the pipeline runs. + """ + return self.thumbnail is not None or any(f.is_thumbnail() for f in self.files) def set_thumbnail(self, thumbnail): """set_thumbnail: Set node's thumbnail @@ -334,8 +340,6 @@ def set_thumbnail(self, thumbnail): """ self.thumbnail = thumbnail if isinstance(self.thumbnail, str): - from .files import ThumbnailFile - self.thumbnail = ThumbnailFile(path=self.thumbnail) if self.thumbnail: @@ -348,6 +352,11 @@ def get_thumbnail_preset(self): """ if isinstance(self, ChannelNode): return format_presets.CHANNEL_THUMBNAIL + if self.kind is None: + # A kind-less node (e.g. a uri-based node before the pipeline has + # run) would otherwise wrongly match the channel_thumbnail preset, + # whose kind is also None. + return None try: preset = next( filter( @@ -359,29 +368,36 @@ def get_thumbnail_preset(self): except StopIteration: return None + def generate_missing_thumbnail(self): + """ + Generate and attach a thumbnail if this node doesn't already have one. + Returns: list of generated thumbnail filenames (empty if nothing was + generated - node already has a thumbnail, not implemented for this + kind, no suitable source file, or generation failed). + """ + filenames = [] + if not self.has_thumbnail(): + thumbnail_file = self.generate_thumbnail() + if thumbnail_file: + thumbnail_filename = thumbnail_file.process_file() + if thumbnail_filename: + self.set_thumbnail(thumbnail_file) + filenames.append(thumbnail_filename) + return filenames + def process_files(self): """Processes all the files associated with this Node, including: - download files if not present in the local storage - convert and compress video files - - (optionally) generate thumbnail file from the node's content + - generate a thumbnail from the node's content when none is provided Returns: content-hash based filenames of all the files for this node """ filenames = [] for file in self.files: filenames.append(file.process_file()) - # Auto-generation of thumbnails happens here if derive_thumbnail or config.THUMBNAILS is set - if not self.has_thumbnail() and (config.THUMBNAILS or self.derive_thumbnail): - thumbnail_file = self.generate_thumbnail() - if thumbnail_file: - thumbnail_filename = thumbnail_file.process_file() - if thumbnail_filename: - self.set_thumbnail(thumbnail_file) - filenames.append(thumbnail_filename) - else: - pass # failed to generate thumbnail - else: - pass # method generate_thumbnail is not implemented or no suitable source file found + if not self.defer_thumbnail_generation: + filenames.extend(self.generate_missing_thumbnail()) return filenames @@ -968,13 +984,15 @@ def gather_ancestor_metadata(self): class TopicNode(TreeNode): """Model representing topic nodes for organizing channel content. - Topic nodes create the folder structure of a channel. When derive_thumbnail - is True, generates a tiled thumbnail from descendant content nodes. + Topic nodes create the folder structure of a channel. A tiled thumbnail is + generated from descendants after all of them have been processed, unless a + thumbnail was provided. See TreeNode and Node for inherited attributes. """ kind = content_kinds.TOPIC + defer_thumbnail_generation = True def generate_thumbnail(self): """Generate a ``TiledThumbnailFile`` based on the descendants. @@ -1195,7 +1213,13 @@ def _apply_exercise_settings(self, settings): self.extra_fields.update(settings, options=options) def _process_uri(self): - context = self.context + # has_thumbnail() cannot resolve a ThumbnailFile's kind-dependent + # preset before the pipeline has set this node's kind, so also count + # provided ThumbnailFile instances (e.g. passed via files=[...]). + has_thumbnail = self.has_thumbnail() or any( + isinstance(f, ThumbnailFile) for f in self.files + ) + context = {**self.context, NODE_HAS_THUMBNAIL: has_thumbnail} if type(self).kind is not None: # A typed node cannot become a folder or change kind. context = {"preserve_kind": True, **context} @@ -1279,9 +1303,6 @@ class VideoNode(ContentNode): Videos must be mp4 or webm format - Attributes: - derive_thumbnail (bool): set to generate thumbnail from video (optional) - See ContentNode for inherited attributes. """ @@ -1348,9 +1369,6 @@ class DocumentNode(ContentNode): Documents must be in PDF, ePub, Bloom, or KPUB format - Attributes: - derive_thumbnail (bool): automatically generate thumbnail (optional) - See ContentNode for inherited attributes. """ @@ -1410,7 +1428,6 @@ class HTML5AppNode(ContentNode): Attributes: entrypoint (str): custom entry point file (optional, defaults to index.html) - derive_thumbnail (bool): generate thumbnail from largest image inside zip (optional) See ContentNode for inherited attributes. """ diff --git a/ricecooker/commands.py b/ricecooker/commands.py index eacac0bc..c94e8cbc 100644 --- a/ricecooker/commands.py +++ b/ricecooker/commands.py @@ -30,7 +30,6 @@ def uploadchannel( # noqa: C901 chef, command="uploadchannel", update=False, - thumbnails=False, download_attempts=3, token="#", prompt=False, @@ -44,7 +43,6 @@ def uploadchannel( # noqa: C901 chef (SushiChef subclass): class that implements the construct_channel method command (str): the action we want to perform in this run update (bool): indicates whether to re-download files (optional) - thumbnails (bool): indicates whether to automatically derive thumbnails from content (optional) download_attempts (int): number of times to retry downloading files (optional) token (str): content server authorization token prompt (bool): indicates whether to prompt user to open channel when done (optional) @@ -58,7 +56,6 @@ def uploadchannel( # noqa: C901 # Set configuration settings config.UPDATE = update config.VIDEO_HEIGHT = chef.get_setting("video-height", None) - config.THUMBNAILS = chef.get_setting("thumbnails", False) config.STAGE = stage config.PUBLISH = publish config.FILE_PIPELINE = chef.file_pipeline diff --git a/ricecooker/config.py b/ricecooker/config.py index 6f386541..3818e0f3 100644 --- a/ricecooker/config.py +++ b/ricecooker/config.py @@ -19,7 +19,6 @@ UPDATE = False VIDEO_HEIGHT = None -THUMBNAILS = False PUBLISH = False SUSHI_BAR_CLIENT = None FILE_PIPELINE = None diff --git a/ricecooker/managers/tree.py b/ricecooker/managers/tree.py index 965d5a12..3e8db612 100644 --- a/ricecooker/managers/tree.py +++ b/ricecooker/managers/tree.py @@ -8,6 +8,7 @@ from ricecooker.exceptions import ChannelIncompleteError from ricecooker.exceptions import InvalidNodeException +from ricecooker.utils.images import ThumbnailGenerationError from .. import config @@ -81,6 +82,7 @@ def process_tree(self): for node in self.all_nodes: if id(node) not in processed: self.file_map.update(self.node_files(node)) + self.generate_deferred_thumbnails() return list(self.file_map.keys()) def deduplicate_shared_nodes(self): @@ -115,7 +117,7 @@ def gather_tree_recur(self, nodes, node): for child_node in node.children: self.gather_tree_recur( nodes, child_node - ) # Defer insert until after all descendants in case a tiled thumbnail is needed + ) # Insert after all descendants so generate_deferred_thumbnails can rely on children-first order nodes.append(node) return nodes @@ -148,6 +150,31 @@ def node_files(self, node): output[question_file.get_filename()] = question_file return output + def generate_deferred_thumbnails(self): + """ + Generate thumbnails for nodes that defer generation until all their + descendants have been processed (topics). Runs sequentially after the + concurrent processing pass: self.all_nodes is ordered children-first, + so every descendant (and its generated thumbnail) is complete by the + time each ancestor is reached. + """ + for node in self.all_nodes: + if not node.defer_thumbnail_generation: + continue + try: + new_filenames = node.generate_missing_thumbnail() + except ThumbnailGenerationError as e: + # Safety net: generation failures are normally recorded in + # config.FAILED_FILES by the file object and returned as None. + config.LOGGER.warning( + "\tFailed to generate thumbnail for {}: {}".format(node.title, e) + ) + continue + # Register the generated thumbnail (attached as node.thumbnail) + # so the upload pass knows about it. + for filename in new_filenames: + self.file_map[filename] = node.thumbnail + def check_for_files_failed(self): """check_for_files_failed: print any files that failed during download process Args: None diff --git a/ricecooker/utils/images.py b/ricecooker/utils/images.py index bf1392d4..141f9921 100644 --- a/ricecooker/utils/images.py +++ b/ricecooker/utils/images.py @@ -9,7 +9,9 @@ import ebooklib.epub from pdf2image import convert_from_path from PIL import Image +from PyPDF2 import PdfFileReader +from .. import config from .thumbscropping import scale_and_crop from .zip import find_html_entrypoint @@ -182,14 +184,67 @@ def create_image_from_zip_screenshot(htmlfile, fpath_out, crop="smart"): raise ThumbnailGenerationError("Fail on zip {} {}".format(htmlfile, e)) -def create_image_from_pdf_page(fpath_in, fpath_out, page_number=0, crop=None): +def create_image_from_html_zip(htmlfile, fpath_out, crop="smart"): + """Render a thumbnail for the html5 zip, falling back to its biggest image when Chromium is absent or fails.""" + try: + create_image_from_zip_screenshot(htmlfile, fpath_out, crop=crop) + return + except ChromiumUnavailableError: + pass + except ThumbnailGenerationError as err: + config.LOGGER.warning( + "\t Screenshot render failed, falling back to biggest-image " + "heuristic: {}".format(err) + ) + create_image_from_zip(htmlfile, fpath_out, crop=crop) + + +def _pdf_page_width_px(fpath_in, page_number, dpi): + """ + Native render width in pixels of the given pdf page at `dpi`, accounting + for page rotation, or None when the page dimensions cannot be read. + """ + try: + # Open the file ourselves: PyPDF2 never closes streams it opens from + # a path string. + with open(fpath_in, "rb") as f: + # convert_from_path's page numbering is 1-based (0 is clamped to 1). + page = PdfFileReader(f, strict=False).getPage(max(page_number - 1, 0)) + media_box = page.mediaBox + width_pts = float(media_box.getWidth()) + if (page.get("/Rotate") or 0) % 180: + width_pts = float(media_box.getHeight()) + return width_pts / 72 * dpi + except Exception: + return None + + +def create_image_from_pdf_page( + fpath_in, fpath_out, page_number=0, crop=None, max_width=None +): """ Create an image from the pdf at fpath_in and write result to fpath_out. + + Pass max_width to cap the rendered width (preserving aspect ratio); + this avoids enormous intermediate images for large-format pages. + Pages that would render narrower than max_width are not upscaled. """ try: assert fpath_in.endswith("pdf"), "File must be in pdf format" + size = None + if max_width: + native_width = _pdf_page_width_px(fpath_in, page_number, 500) + # Cap, don't force: poppler's -scale-to-x renders at exactly the + # given width, so only pass it for pages that would render wider. + # When the page size can't be read, apply the cap defensively. + if native_width is None or native_width > max_width: + size = (max_width, None) pages = convert_from_path( - fpath_in, 500, first_page=page_number, last_page=page_number + 1 + fpath_in, + 500, + first_page=page_number, + last_page=page_number + 1, + size=size, ) page = pages[0] # resize diff --git a/ricecooker/utils/jsontrees.py b/ricecooker/utils/jsontrees.py index 0694382e..cd6714f1 100644 --- a/ricecooker/utils/jsontrees.py +++ b/ricecooker/utils/jsontrees.py @@ -125,7 +125,6 @@ def build_tree_from_json(parent_node, sourcetree): # no role for topics (computed dynaically from descendants) language=source_node.get("language"), thumbnail=source_node.get("thumbnail"), - derive_thumbnail=source_node.get("derive_thumbnail", False), tags=source_node.get("tags"), ) parent_node.add_child(child_node) @@ -144,7 +143,6 @@ def build_tree_from_json(parent_node, sourcetree): role=source_node.get("role", roles.LEARNER), language=source_node.get("language"), thumbnail=source_node.get("thumbnail"), - derive_thumbnail=source_node.get("derive_thumbnail", False), tags=source_node.get("tags"), ) add_files(child_node, source_node.get("files") or []) @@ -162,7 +160,6 @@ def build_tree_from_json(parent_node, sourcetree): role=source_node.get("role", roles.LEARNER), language=source_node.get("language"), thumbnail=source_node.get("thumbnail"), - derive_thumbnail=source_node.get("derive_thumbnail", False), tags=source_node.get("tags"), ) add_files(child_node, source_node.get("files") or []) @@ -180,9 +177,6 @@ def build_tree_from_json(parent_node, sourcetree): role=source_node.get("role", roles.LEARNER), language=source_node.get("language"), thumbnail=source_node.get("thumbnail"), - derive_thumbnail=source_node.get( - "derive_thumbnail", False - ), # not supported yet tags=source_node.get("tags"), exercise_data=source_node.get("exercise_data"), questions=[], @@ -219,7 +213,6 @@ def build_tree_from_json(parent_node, sourcetree): role=source_node.get("role", roles.LEARNER), language=source_node.get("language"), thumbnail=source_node.get("thumbnail"), - derive_thumbnail=source_node.get("derive_thumbnail", False), tags=source_node.get("tags"), ) @@ -238,7 +231,6 @@ def build_tree_from_json(parent_node, sourcetree): role=source_node.get("role", roles.LEARNER), language=source_node.get("language"), thumbnail=source_node.get("thumbnail"), - derive_thumbnail=source_node.get("derive_thumbnail", False), tags=source_node.get("tags"), ) add_files(child_node, source_node.get("files") or []) diff --git a/ricecooker/utils/linecook.py b/ricecooker/utils/linecook.py index 8ed66845..95484611 100644 --- a/ricecooker/utils/linecook.py +++ b/ricecooker/utils/linecook.py @@ -332,7 +332,6 @@ def make_content_node(channeldir, rel_path, filename, metadata): # noqa: C901 description=description, language=lang, license=license_dict, - derive_thumbnail=True, thumbnail=thumbnail_rel_path, files=[ {"file_type": VIDEO_FILE, "path": filepath, "language": lang} @@ -349,7 +348,6 @@ def make_content_node(channeldir, rel_path, filename, metadata): # noqa: C901 language=lang, license=license_dict, thumbnail=thumbnail_rel_path, - derive_thumbnail=True, files=[{"file_type": AUDIO_FILE, "path": filepath, "language": lang}], ) @@ -363,7 +361,6 @@ def make_content_node(channeldir, rel_path, filename, metadata): # noqa: C901 language=lang, license=license_dict, thumbnail=thumbnail_rel_path, - derive_thumbnail=True, files=[], ) if ext == "pdf": @@ -385,7 +382,6 @@ def make_content_node(channeldir, rel_path, filename, metadata): # noqa: C901 language=lang, license=license_dict, thumbnail=thumbnail_rel_path, - derive_thumbnail=True, files=[{"file_type": HTML5_FILE, "path": filepath, "language": lang}], ) @@ -401,7 +397,6 @@ def make_content_node(channeldir, rel_path, filename, metadata): # noqa: C901 exercise_data=metadata["exercise_data"], questions=metadata["questions"], thumbnail=thumbnail_rel_path, - derive_thumbnail=False, files=[], ) diff --git a/ricecooker/utils/pipeline/__init__.py b/ricecooker/utils/pipeline/__init__.py index 3c56cbd3..696dbcdc 100644 --- a/ricecooker/utils/pipeline/__init__.py +++ b/ricecooker/utils/pipeline/__init__.py @@ -9,6 +9,7 @@ from .convert import ConversionStageHandler from .extract_metadata import ExtractMetadataStageHandler from .file_handler import CompositeHandler +from .thumbnails import ThumbnailStageHandler from .transfer import DownloadStageHandler # Do this to prevent import of broken Windows filetype registry that makes guesstype not work. @@ -37,12 +38,17 @@ class FilePipeline(CompositeHandler): from ricecooker.utils.pipeline import FilePipeline from ricecooker.utils.pipeline.convert import ConversionStageHandler from ricecooker.utils.pipeline.extract_metadata import ExtractMetadataStageHandler + from ricecooker.utils.pipeline.thumbnails import ThumbnailStageHandler from ricecooker.utils.pipeline.transfer import DownloadStageHandler from ricecooker.utils.pipeline.transfer import DiskResourceHandler download_stage = DownloadStageHandler(children=[DiskResourceHandler()]) - pipeline = FilePipeline(children=[download_stage, ConversionStageHandler(), ExtractMetadataStageHandler()]) + pipeline = FilePipeline( + children=[download_stage, ConversionStageHandler(), ThumbnailStageHandler(), ExtractMetadataStageHandler()] + ) ``` + Note: omitting ThumbnailStageHandler disables automatic thumbnail generation + for files processed by the pipeline. This will replace the default `DownloadStageHandler` with a new one that has a `DiskResourceHandler` as its only child. Instead of a combined `default_context`, a stage's handler can be configured directly with init-time context: @@ -61,6 +67,7 @@ class FilePipeline(CompositeHandler): DEFAULT_CHILDREN = [ DownloadStageHandler, ConversionStageHandler, + ThumbnailStageHandler, ExtractMetadataStageHandler, ] diff --git a/ricecooker/utils/pipeline/context.py b/ricecooker/utils/pipeline/context.py index a4e97df2..29dee563 100644 --- a/ricecooker/utils/pipeline/context.py +++ b/ricecooker/utils/pipeline/context.py @@ -3,6 +3,10 @@ from typing import Optional from typing import Type +# Context key used by producers (nodes, chefs) to tell the thumbnail stage +# that the node already has a thumbnail, so generation can be skipped. +NODE_HAS_THUMBNAIL = "node_has_thumbnail" + class AutoDataClassMetaClass(type): def __new__(mcs, name: str, bases: tuple, namespace: dict) -> Type: diff --git a/ricecooker/utils/pipeline/file_handler.py b/ricecooker/utils/pipeline/file_handler.py index 232fafa0..009892d6 100644 --- a/ricecooker/utils/pipeline/file_handler.py +++ b/ricecooker/utils/pipeline/file_handler.py @@ -169,18 +169,21 @@ def _get_context(self, context: Optional[Dict] = None): def write_file(self, extension: str): """ Context manager that provides a file handle to write to and handles copying to storage. + + Copying to storage only happens when the body completes without raising: + an exception from the body propagates unchanged (rather than being masked + by the empty-file check) and a partially written file is discarded + instead of being copied to storage. """ with DualModeTemporaryFile(ext=extension) as tempf: - try: - yield tempf - finally: - tempf.flush() - if not tempf.file_not_empty(): - raise InvalidFileException( - f"File with extension {extension} failed to write (corrupted)." - ) - filename = copy_file_to_storage(tempf.name, ext=extension) - self._output_path = config.get_storage_path(filename) + yield tempf + tempf.flush() + if not tempf.file_not_empty(): + raise InvalidFileException( + f"File with extension {extension} failed to write (corrupted)." + ) + filename = copy_file_to_storage(tempf.name, ext=extension) + self._output_path = config.get_storage_path(filename) @abstractmethod def handle_file(self, path, **kwargs) -> Union[None, FileMetadata]: diff --git a/ricecooker/utils/pipeline/thumbnails.py b/ricecooker/utils/pipeline/thumbnails.py new file mode 100644 index 00000000..40edb6d9 --- /dev/null +++ b/ricecooker/utils/pipeline/thumbnails.py @@ -0,0 +1,103 @@ +""" +Pipeline stage that extracts thumbnail images from processed files. + +Runs after conversion, so it sees files in their final formats. For each +supported file it emits an additional FileMetadata for the generated PNG +with the appropriate thumbnail preset, alongside the source file. +Generation is skipped when the node already provides a thumbnail (see the +NODE_HAS_THUMBNAIL context key). + +Thumbnail generation is best-effort: if extraction fails, the source +file passes through unaffected. +""" + +from typing import Dict +from typing import Optional + +from le_utils.constants import file_formats +from le_utils.constants import format_presets + +from ricecooker import config +from ricecooker.utils.images import create_image_from_epub +from ricecooker.utils.images import create_image_from_html_zip +from ricecooker.utils.images import create_image_from_pdf_page +from ricecooker.utils.images import ThumbnailGenerationError +from ricecooker.utils.paths import extract_path_ext +from ricecooker.utils.pipeline.context import ContextMetadata +from ricecooker.utils.pipeline.context import FileMetadata +from ricecooker.utils.pipeline.exceptions import ExpectedFileException +from ricecooker.utils.pipeline.exceptions import InvalidFileException +from ricecooker.utils.videos import extract_thumbnail_from_video + +from .file_handler import ExtensionMatchingHandler +from .file_handler import StageHandler + +THUMBNAIL_PRESETS = { + file_formats.PDF: format_presets.DOCUMENT_THUMBNAIL, + file_formats.EPUB: format_presets.DOCUMENT_THUMBNAIL, + file_formats.HTML5_ARTICLE: format_presets.DOCUMENT_THUMBNAIL, + file_formats.HTML5: format_presets.HTML5_THUMBNAIL, + file_formats.MP4: format_presets.VIDEO_THUMBNAIL, + file_formats.WEBM: format_presets.VIDEO_THUMBNAIL, +} + + +class ThumbnailContextMetadata(ContextMetadata): + node_has_thumbnail: bool = False + content_node_metadata: Optional[Dict] = None + + +class ThumbnailExtractionHandler(ExtensionMatchingHandler): + """Generates a PNG thumbnail from a processed file.""" + + EXTENSIONS = set(THUMBNAIL_PRESETS) + + HANDLED_EXCEPTIONS = [ThumbnailGenerationError] + + CONTEXT_CLASS = ThumbnailContextMetadata + + def get_file_kwargs(self, context): + decomposed = (context.content_node_metadata or {}).get("children") is not None + if context.node_has_thumbnail or decomposed: + # The node already has a thumbnail; nothing to generate. + # Returning [] bypasses both cache lookup and generation + # entirely - any cached thumbnail for this path is not consulted. + return [] + return [{}] + + def handle_file(self, path): + ext = extract_path_ext(path) + with self.write_file(file_formats.PNG) as fh: + if ext == file_formats.PDF: + create_image_from_pdf_page(path, fh.name, max_width=1000) + elif ext == file_formats.EPUB: + create_image_from_epub(path, fh.name) + elif ext in (file_formats.HTML5, file_formats.HTML5_ARTICLE): + create_image_from_html_zip(path, fh.name) + else: + extract_thumbnail_from_video(path, fh.name, overwrite=True) + return FileMetadata(preset=THUMBNAIL_PRESETS[ext]) + + +class ThumbnailStageHandler(StageHandler): + STAGE = "THUMBNAIL" + DEFAULT_CHILDREN = [ThumbnailExtractionHandler] + + def execute( + self, + path: str, + context: Optional[Dict] = None, + skip_cache: Optional[bool] = False, + ) -> list[FileMetadata]: + """ + Pass the source file through unchanged, plus the extracted + thumbnail if generation succeeded. + """ + try: + thumbnails = super().execute(path, context=context, skip_cache=skip_cache) + except (ExpectedFileException, InvalidFileException) as e: + # InvalidFileException covers an extractor that completes without + # writing any bytes; a failed thumbnail must never fail the node. + config.LOGGER.warning(f"\tFailed to extract thumbnail from {path}: {e}") + thumbnails = [] + return [FileMetadata(path=path)] + thumbnails diff --git a/tests/media_utils/test_thumbnails.py b/tests/media_utils/test_thumbnails.py index e88c67db..212fc825 100644 --- a/tests/media_utils/test_thumbnails.py +++ b/tests/media_utils/test_thumbnails.py @@ -1,6 +1,7 @@ import os import zipfile from io import BytesIO +from unittest.mock import patch import PIL import pytest @@ -130,6 +131,44 @@ def test_generates_16_9_thumbnail(self, tmpdir): images.create_image_from_pdf_page(input_file, output_file, crop="smart") self.check_16_9_format(output_file) + def _render_mocked_page(self, tmpdir, native_width_px=None, **kwargs): + output_file = tmpdir.join("thumbnail.png").strpath + page = PIL.Image.new("RGB", (160, 90)) + with patch( + "ricecooker.utils.images.convert_from_path", return_value=[page] + ) as mock_convert: + with patch( + "ricecooker.utils.images._pdf_page_width_px", + return_value=native_width_px, + ): + images.create_image_from_pdf_page("fake.pdf", output_file, **kwargs) + assert os.path.exists(output_file) + return mock_convert.call_args.kwargs + + def test_max_width_caps_oversized_page(self, tmpdir): + kwargs = self._render_mocked_page(tmpdir, native_width_px=4250, max_width=1000) + assert kwargs["size"] == (1000, None) + + def test_max_width_does_not_upscale_small_page(self, tmpdir): + kwargs = self._render_mocked_page(tmpdir, native_width_px=600, max_width=1000) + assert kwargs["size"] is None + + def test_max_width_applied_when_page_size_unreadable(self, tmpdir): + kwargs = self._render_mocked_page(tmpdir, native_width_px=None, max_width=1000) + assert kwargs["size"] == (1000, None) + + def test_no_max_width_renders_at_full_size(self, tmpdir): + kwargs = self._render_mocked_page(tmpdir) + assert kwargs["size"] is None + + def test_pdf_page_width_px_reads_real_pdf(self): + input_file = os.path.join(files_dir, "generate_thumbnail", "sample.pdf") + width = images._pdf_page_width_px(input_file, page_number=0, dpi=500) + assert width is not None and width > 0 + + def test_pdf_page_width_px_returns_none_for_unreadable(self): + assert images._pdf_page_width_px("fake.pdf", page_number=0, dpi=500) is None + def test_raises_for_missing_file(self, tmpdir): input_file = os.path.join(files_dir, "file_that_does_not_exist.pdf") assert not os.path.exists(input_file) diff --git a/tests/pipeline/test_convert.py b/tests/pipeline/test_convert.py index 1f8ca402..218111e3 100644 --- a/tests/pipeline/test_convert.py +++ b/tests/pipeline/test_convert.py @@ -1473,14 +1473,23 @@ def _leaves_by_title(tree): return {leaf["title"]: leaf for leaf in _tree_dict_leaves(tree)} +_THUMBNAIL_PRESETS = {p.id for p in format_presets.PRESETLIST if p.thumbnail} + + +def _content_files(files): + return [f for f in files if f["preset"] not in _THUMBNAIL_PRESETS] + + def _filenames(leaf): - return {f["filename"] for f in leaf["files"]} + return {f["filename"] for f in _content_files(leaf["files"])} def _primary_file(leaf): """The file dict of the zip ``leaf`` is sealed into.""" (primary,) = [ - f for f in leaf["files"] if f["preset"] != format_presets.HTML5_DEPENDENCY_ZIP + f + for f in _content_files(leaf["files"]) + if f["preset"] != format_presets.HTML5_DEPENDENCY_ZIP ] return primary @@ -1490,7 +1499,9 @@ def _primary_members(leaf): def _tree_files(tree): - return [f for leaf in _tree_dict_leaves(tree) for f in leaf["files"]] + return [ + f for leaf in _tree_dict_leaves(tree) for f in _content_files(leaf["files"]) + ] def _tree_zips(tree): @@ -1924,8 +1935,11 @@ def test_end_to_end_node_expansion(self): # eXe's residual jQuery/effects and .js/.css members keep pages off KPUB, # and those shared assets ship once, in a dependency zip every leaf carries. presets = {format_presets.HTML5_ZIP, format_presets.HTML5_DEPENDENCY_ZIP} - assert all({f.get_preset() for f in leaf.files} == presets for leaf in leaves) - files = [f for leaf in leaves for f in leaf.files] + files = [f for leaf in leaves for f in leaf.files if not f.is_thumbnail()] + assert all( + {f.get_preset() for f in leaf.files if not f.is_thumbnail()} == presets + for leaf in leaves + ) # Each leaf is backed by its own sealed zip, not the shared package. filenames = {preset: [] for preset in presets} for f in files: @@ -1933,7 +1947,8 @@ def test_end_to_end_node_expansion(self): leaf_filenames = filenames[format_presets.HTML5_ZIP] assert len(leaf_filenames) == len(set(leaf_filenames)) assert len(set(filenames[format_presets.HTML5_DEPENDENCY_ZIP])) == 1 - assert {f.get_filename() for f in files} <= set(files_to_upload) + all_files = [f for leaf in leaves for f in leaf.files] + assert {f.get_filename() for f in all_files} <= set(files_to_upload) def test_shared_assets_move_to_one_dependency_zip(self): tree = _decompose_package(_SHARED_RESOURCES, _SHARED_FILES) @@ -2028,9 +2043,12 @@ def test_wrapped_media_gets_package_compression_settings(self, video_file): ) (leaf,) = _tree_dict_leaves(tree) (expected,) = [ - f.filename - for f in FilePipeline().execute( - video_file.path, context=context, skip_cache=True + f["filename"] + for f in _content_files( + f.to_dict() + for f in FilePipeline().execute( + video_file.path, context=context, skip_cache=True + ) ) ] assert _filenames(leaf) == {expected} diff --git a/tests/pipeline/test_file_handler.py b/tests/pipeline/test_file_handler.py index 26daf385..33051eb5 100644 --- a/tests/pipeline/test_file_handler.py +++ b/tests/pipeline/test_file_handler.py @@ -24,23 +24,28 @@ def handle_file(self, path, **kwargs): return None -def test_write_file_with_exception_still_checks_file_not_empty(): +def test_write_file_exception_propagates_unchanged(): """ - Test that file_not_empty check runs even when an exception is caught - within the context manager. This verifies the try/finally behavior. - - Without the try/finally block, this test would fail because the exception - would prevent the file_not_empty check from running. + An exception raised inside the write_file body must propagate unchanged - + not be masked by the empty-file check - and nothing is copied to storage. + This lets handlers raise their typed exceptions (caught via + HANDLED_EXCEPTIONS) while writing directly into write_file's temp path. """ handler = TestFileHandler() - # This should raise InvalidFileException from the finally block, - # not the RuntimeError from the try block + with pytest.raises(RuntimeError, match="boom"): + with handler.write_file("txt"): + raise RuntimeError("boom") + assert handler._output_path is None, "no partial file should reach storage" + + +def test_write_file_empty_file_raises(): + """A body that completes without writing anything is treated as corrupt.""" + handler = TestFileHandler() + with pytest.raises(InvalidFileException, match="failed to write \\(corrupted\\)"): with handler.write_file("txt"): - # Don't write anything to file (will make it empty) - # Then raise an exception that would normally prevent cleanup - raise RuntimeError("This exception should be caught by try/finally") + pass class ThreadRaceTestHandler(FileHandler): diff --git a/tests/pipeline/test_thumbnails.py b/tests/pipeline/test_thumbnails.py new file mode 100644 index 00000000..0edb237d --- /dev/null +++ b/tests/pipeline/test_thumbnails.py @@ -0,0 +1,146 @@ +"""Tests for the thumbnail extraction pipeline stage.""" + +import os +import shutil +import tempfile +from unittest.mock import patch + +from le_utils.constants import format_presets + +from ricecooker.utils.pipeline.context import NODE_HAS_THUMBNAIL +from ricecooker.utils.pipeline.thumbnails import ThumbnailStageHandler + + +def _assert_source_and_thumbnail(results, source_path, expected_preset): + assert len(results) == 2, "expected source pass-through plus thumbnail" + source, thumb = results + assert source.path == source_path + assert thumb.preset == expected_preset + assert thumb.path.endswith(".png") + assert os.path.getsize(thumb.path) > 0 + + +def test_generates_thumbnail_from_pdf(document_file): + results = ThumbnailStageHandler().execute(document_file.path, skip_cache=True) + _assert_source_and_thumbnail( + results, document_file.path, format_presets.DOCUMENT_THUMBNAIL + ) + + +def test_generates_thumbnail_from_epub(epub_file): + results = ThumbnailStageHandler().execute(epub_file.path, skip_cache=True) + _assert_source_and_thumbnail( + results, epub_file.path, format_presets.DOCUMENT_THUMBNAIL + ) + + +def test_generates_thumbnail_from_html_zip(html_file): + results = ThumbnailStageHandler().execute(html_file.path, skip_cache=True) + _assert_source_and_thumbnail( + results, html_file.path, format_presets.HTML5_THUMBNAIL + ) + + +def test_generates_thumbnail_from_mp4(video_file): + # webm is also a supported extension but shares the + # extract_thumbnail_from_video code path with mp4; no webm fixture + # exists, so mp4 covers the video path. + results = ThumbnailStageHandler().execute(video_file.path, skip_cache=True) + _assert_source_and_thumbnail( + results, video_file.path, format_presets.VIDEO_THUMBNAIL + ) + + +def test_generates_thumbnail_from_kpub(html_file): + # kpub (HTML5_ARTICLE) zips share the zip extraction path with html5, + # but map to the document thumbnail preset. + tempdir = tempfile.mkdtemp() + try: + path = os.path.join(tempdir, "test.kpub") + shutil.copy(html_file.path, path) + results = ThumbnailStageHandler().execute(path, skip_cache=True) + _assert_source_and_thumbnail(results, path, format_presets.DOCUMENT_THUMBNAIL) + finally: + shutil.rmtree(tempdir) + + +def test_unsupported_format_passes_through(audio_file): + results = ThumbnailStageHandler().execute(audio_file.path, skip_cache=True) + assert len(results) == 1 + assert results[0].path == audio_file.path + + +def test_invalid_pdf_passes_source_through(invalid_document_file): + results = ThumbnailStageHandler().execute( + invalid_document_file.path, skip_cache=True + ) + assert len(results) == 1 + assert results[0].path == invalid_document_file.path + + +def test_empty_thumbnail_output_passes_source_through(document_file): + # An extractor that completes without writing any bytes triggers + # write_file's InvalidFileException; the stage treats it as best-effort + # and passes the source through rather than failing the node. + with patch( + "ricecooker.utils.pipeline.thumbnails.create_image_from_pdf_page" + ) as mock_create: + results = ThumbnailStageHandler().execute(document_file.path, skip_cache=True) + assert mock_create.called + assert len(results) == 1 + assert results[0].path == document_file.path + + +def test_skips_generation_when_node_has_thumbnail(document_file): + with patch( + "ricecooker.utils.pipeline.thumbnails.create_image_from_pdf_page" + ) as mock_create: + results = ThumbnailStageHandler().execute( + document_file.path, + context={NODE_HAS_THUMBNAIL: True}, + skip_cache=True, + ) + # The length check below is the real guard: when generation is skipped, + # handle_file is never invoked, so the mock can never be reached. + assert not mock_create.called + assert len(results) == 1 + assert results[0].path == document_file.path + + +def test_skips_generation_for_decomposed_package(html_file): + results = ThumbnailStageHandler().execute( + html_file.path, + context={"content_node_metadata": {"children": [{"title": "Leaf"}]}}, + skip_cache=True, + ) + assert len(results) == 1 + assert results[0].path == html_file.path + + +def test_generates_when_node_has_thumbnail_is_false(document_file): + results = ThumbnailStageHandler().execute( + document_file.path, + context={NODE_HAS_THUMBNAIL: False}, + skip_cache=True, + ) + assert len(results) == 2 + assert results[1].preset == format_presets.DOCUMENT_THUMBNAIL + + +def test_thumbnail_cached_on_second_run(document_file): + # Copy to a unique path so this test never collides with cache entries + # written by other tests or earlier runs. + tempdir = tempfile.mkdtemp() + try: + path = os.path.join(tempdir, "cached_thumbnail_test.pdf") + shutil.copy(document_file.path, path) + stage = ThumbnailStageHandler() + first = stage.execute(path, skip_cache=True) + with patch( + "ricecooker.utils.pipeline.thumbnails.create_image_from_pdf_page" + ) as mock_create: + second = stage.execute(path) + assert not mock_create.called, "second run should be served from cache" + assert second[1].filename == first[1].filename + finally: + shutil.rmtree(tempdir) diff --git a/tests/test_argparse.py b/tests/test_argparse.py index c265c90b..817a58ea 100644 --- a/tests/test_argparse.py +++ b/tests/test_argparse.py @@ -17,12 +17,12 @@ def cli_args_and_expected(): "warn": False, "quiet": False, "compress": False, - "thumbnails": False, "download_attempts": 3, "prompt": False, "reset_deprecated": False, "stage": True, "stage_deprecated": False, + "thumbnails_deprecated": False, "publish": False, "sample": None, } @@ -37,6 +37,14 @@ def cli_args_and_expected(): "expected_args": dict(defaults, token="letoken"), "expected_options": {}, }, + { # --thumbnails is deprecated (generation is always on) but must + # still parse as a no-op rather than crash options parsing + "cli_input": "./sushichef.py --token=letoken --thumbnails", + "expected_args": dict( + defaults, token="letoken", thumbnails_deprecated=True + ), + "expected_options": {}, + }, { "cli_input": "./sushichef.py --token=letoken lang=fr", "expected_args": dict(defaults, token="letoken"), diff --git a/tests/test_files.py b/tests/test_files.py index 25cc135b..4dadd34b 100644 --- a/tests/test_files.py +++ b/tests/test_files.py @@ -1462,6 +1462,7 @@ def test_html5_zip_cache_keys(mock_filecache, html_file, html_filename): assert f"DOWNLOAD:{path}" in keys assert f"CONVERT:{html_filename}" in keys assert f"EXTRACT_METADATA:{html.filename}" in keys + assert not any(k.startswith("THUMBNAIL:") for k in keys) gif_convert_keys = { k for k in keys if k.startswith("CONVERT:") and k.endswith(".gif") } diff --git a/tests/test_settings.py b/tests/test_settings.py index b16daec7..5492c287 100644 --- a/tests/test_settings.py +++ b/tests/test_settings.py @@ -13,7 +13,7 @@ from ricecooker.utils import metadata_provider from ricecooker.utils.request_utils import DomainSpecificAuth -settings = {"thumbnails": True, "compress": True} +settings = {"compress": True} def test_settings_unset_default(): @@ -40,32 +40,27 @@ def test_cli_args_override_settings(): takes precedence over the default setting. """ - test_argv = ["sushichef.py", "--compress", "--thumbnails", "--token", "12345"] + test_argv = ["sushichef.py", "--compress", "--token", "12345"] with patch.object(sys, "argv", test_argv): chef = chefs.SushiChef() - chef.SETTINGS["thumbnails"] = False chef.SETTINGS["compress"] = False - assert chef.get_setting("thumbnails") is False assert chef.get_setting("compress") is False chef.parse_args_and_options() - assert chef.get_setting("thumbnails") is True assert chef.get_setting("compress") is True - test_argv = ["sushichef.py", "--compress", "--thumbnails", "--token", "12345"] + test_argv = ["sushichef.py", "--compress", "--token", "12345"] with patch.object(sys, "argv", test_argv): chef = chefs.SushiChef() assert len(chef.SETTINGS) == 0 - assert chef.get_setting("thumbnails") is None assert chef.get_setting("compress") is None chef.parse_args_and_options() - assert chef.get_setting("thumbnails") is True assert chef.get_setting("compress") is True # now test without setting the flags @@ -73,17 +68,28 @@ def test_cli_args_override_settings(): with patch.object(sys, "argv", test_argv): chef = chefs.SushiChef() - chef.SETTINGS["thumbnails"] = False chef.SETTINGS["compress"] = False - assert chef.get_setting("thumbnails") is False assert chef.get_setting("compress") is False chef.parse_args_and_options() - assert chef.get_setting("thumbnails") is False assert chef.get_setting("compress") is False +def test_deprecated_thumbnail_settings_warn(): + """ + Thumbnail generation is always on, so the old opt-in SETTINGS keys are + obsolete; instantiation must warn rather than silently ignore them. + """ + for key in ("thumbnails", "generate-missing-thumbnails"): + + class ThumbnailSettingsChef(chefs.SushiChef): + SETTINGS = {key: True} + + with pytest.warns(DeprecationWarning, match=key): + ThumbnailSettingsChef() + + # Domain-specific authentication tests def test_domain_auth_with_valid_environment_variables(): """Test DomainSpecificAuth initialization and header application with valid environment variables.""" diff --git a/tests/test_thumbnails.py b/tests/test_thumbnails.py index ea9b8aa0..9bea65ed 100644 --- a/tests/test_thumbnails.py +++ b/tests/test_thumbnails.py @@ -1,4 +1,6 @@ import os +from unittest.mock import MagicMock +from unittest.mock import patch import PIL import pytest # noqa F401 @@ -17,13 +19,18 @@ from ricecooker.classes.files import ExtractedKPUBThumbnailFile from ricecooker.classes.files import ExtractedPdfThumbnailFile from ricecooker.classes.files import ExtractedVideoThumbnailFile +from ricecooker.classes.files import File from ricecooker.classes.files import ThumbnailFile from ricecooker.classes.files import TiledThumbnailFile from ricecooker.classes.files import VideoFile +from ricecooker.classes.nodes import ContentNode from ricecooker.classes.nodes import DocumentNode from ricecooker.classes.nodes import HTML5AppNode from ricecooker.classes.nodes import TopicNode from ricecooker.classes.nodes import VideoNode +from ricecooker.managers.tree import ChannelManager +from ricecooker.utils.images import ThumbnailGenerationError +from ricecooker.utils.pipeline.context import NODE_HAS_THUMBNAIL SHOW_THUMBS = False # set to True to show outputs when running tests locally @@ -168,6 +175,134 @@ def test_set_thumbnail_from_bad_path( self.assert_failed_thumbnail(video_node) +class TestIsThumbnail(object): + """File.is_thumbnail() detects thumbnails by format preset, not class.""" + + def test_plain_file_with_thumbnail_preset(self): + f = File(preset=format_presets.DOCUMENT_THUMBNAIL, filename="abc.png") + assert f.is_thumbnail() is True + + def test_plain_file_with_content_preset(self): + f = File(preset=format_presets.DOCUMENT, filename="abc.pdf") + assert f.is_thumbnail() is False + + def test_plain_file_without_preset(self): + f = File(filename="abc.pdf") + assert f.is_thumbnail() is False + + def test_thumbnail_file_attached_to_node(self, thumbnail_path): # noqa F811 + node = DocumentNode( + "doc-src-id", "Document", licenses.PUBLIC_DOMAIN, thumbnail=thumbnail_path + ) + assert any(f.is_thumbnail() for f in node.files) + + def test_unattached_thumbnail_file_is_not_thumbnail( + self, + thumbnail_path, # noqa F811 + ): + # A ThumbnailFile with no node cannot resolve its preset. + f = ThumbnailFile(thumbnail_path) + assert f.is_thumbnail() is False + + def test_thumbnail_file_on_kindless_node_is_not_thumbnail( + self, + thumbnail_path, # noqa F811 + ): + # Attached to a node that has no kind yet (a uri-based node before + # the pipeline has run), so the preset is unresolvable. + node = ContentNode( + "src-id", "Title", licenses.PUBLIC_DOMAIN, uri="/tmp/doc.pdf" + ) + f = ThumbnailFile(thumbnail_path) + node.add_file(f) + assert f.is_thumbnail() is False + + +class TestHasThumbnail(object): + """Node.has_thumbnail() detects provided and preset-based thumbnails.""" + + def test_pipeline_generated_thumbnail_counts(self): + node = DocumentNode("doc-src-id", "Document", licenses.PUBLIC_DOMAIN) + node.add_file( + File(preset=format_presets.DOCUMENT_THUMBNAIL, filename="abc.png") + ) + assert node.has_thumbnail() is True + + def test_content_file_does_not_count(self): + node = DocumentNode("doc-src-id", "Document", licenses.PUBLIC_DOMAIN) + node.add_file(File(preset=format_presets.DOCUMENT, filename="abc.pdf")) + assert node.has_thumbnail() is False + + def test_provided_thumbnail_counts_before_kind_is_known( + self, + thumbnail_path, # noqa F811 + ): + # uri-based ContentNode has no kind until the pipeline has run, so the + # ThumbnailFile preset is unresolvable - self.thumbnail must count. + node = ContentNode( + "src-id", + "Title", + licenses.PUBLIC_DOMAIN, + uri="/tmp/does-not-matter.pdf", + thumbnail=thumbnail_path, + ) + assert node.has_thumbnail() is True + + def test_process_uri_passes_node_has_thumbnail_context( + self, + thumbnail_path, # noqa F811 + ): + pipeline = MagicMock() + pipeline.execute.return_value = [] + + node = ContentNode( + "src-id", + "Title", + licenses.PUBLIC_DOMAIN, + uri="/tmp/doc.pdf", + pipeline=pipeline, + ) + node._process_uri() + pipeline.execute.assert_called_once_with( + "/tmp/doc.pdf", + context={NODE_HAS_THUMBNAIL: False}, + skip_cache=config.UPDATE, + ) + + pipeline.reset_mock() + node.set_thumbnail(thumbnail_path) + node._process_uri() + pipeline.execute.assert_called_once_with( + "/tmp/doc.pdf", + context={NODE_HAS_THUMBNAIL: True}, + skip_cache=config.UPDATE, + ) + + def test_process_uri_counts_thumbnail_file_passed_in_files( + self, + thumbnail_path, # noqa F811 + ): + # A ThumbnailFile in files=[...] cannot resolve its preset before the + # node has a kind, but it must still suppress pipeline generation. + pipeline = MagicMock() + pipeline.execute.return_value = [] + + node = ContentNode( + "src-id", + "Title", + licenses.PUBLIC_DOMAIN, + uri="/tmp/doc.pdf", + pipeline=pipeline, + files=[ThumbnailFile(thumbnail_path)], + ) + node._process_uri() + pipeline.execute.assert_called_once_with( + "/tmp/doc.pdf", + context={NODE_HAS_THUMBNAIL: True}, + skip_cache=config.UPDATE, + ) + + class TestThumbnailGeneration(object): def setup_method(self, test_method): """ @@ -175,7 +310,6 @@ def setup_method(self, test_method): """ _clear_ricecookerfilecache() config.FAILED_FILES = [] - config.THUMBNAILS = False def check_has_thumbnail(self, node): thumbnail_files = [ @@ -217,7 +351,6 @@ def test_generate_thumbnail_from_pdf(self, document_file): "doc-src-id", "Document", licenses.PUBLIC_DOMAIN, thumbnail=None ) node.add_file(document_file) - config.THUMBNAILS = True filenames = node.process_files() assert len(filenames) == 2, "expected two filenames" self.check_has_thumbnail(node) @@ -227,7 +360,6 @@ def test_generate_thumbnail_from_epub(self, epub_file): "doc-src-id", "Document", licenses.PUBLIC_DOMAIN, thumbnail=None ) node.add_file(epub_file) - config.THUMBNAILS = True filenames = node.process_files() assert len(filenames) == 2, "expected two filenames" self.check_has_thumbnail(node) @@ -237,7 +369,6 @@ def test_generate_thumbnail_from_html(self, html_file): "html-src-id", "HTML5 App", licenses.PUBLIC_DOMAIN, thumbnail=None ) node.add_file(html_file) - config.THUMBNAILS = True filenames = node.process_files() assert len(filenames) == 2, "expected two filenames" self.check_has_thumbnail(node) @@ -254,23 +385,44 @@ def test_generate_thumbnail_from_kpub(self): def test_generate_thumbnail_from_video(self, video_file): node = VideoNode("vid-src-id", "Video", licenses.PUBLIC_DOMAIN, thumbnail=None) node.add_file(video_file) - config.THUMBNAILS = True filenames = node.process_files() assert len(filenames) == 2, "expected two filenames" self.check_has_thumbnail(node) - def test_generate_tiled_thumbnail(self, document, html, video, audio): + def test_topic_does_not_generate_thumbnail_in_process_files( + self, document, html, video, audio + ): topic = TopicNode("test-topic", "Topic") - topic.add_child(document) - topic.add_child(html) - topic.add_child(video) - topic.add_child(audio) - config.THUMBNAILS = True - for child in topic.children: # must process children before topic node + # Children are processed first so their thumbnails would be available; + # the topic must still defer (the tree manager post-pass tiles later). + for child in (document, html, video, audio): + topic.add_child(child) child.process_files() filenames = topic.process_files() - assert len(filenames) == 1, "expected one filename" - self.check_has_thumbnail(topic) + assert filenames == [], "topic thumbnail generation must be deferred" + assert not topic.has_thumbnail() + + def test_legacy_file_processing_skips_pipeline_thumbnail_stage(self, document_file): + # The legacy DownloadFile path only consumes the source metadata, so + # the pipeline thumbnail stage must not generate an image for it + # (node-level generate_missing_thumbnail handles legacy thumbnails). + with patch( + "ricecooker.utils.pipeline.thumbnails.create_image_from_pdf_page" + ) as mock_create: + filename = document_file.process_file() + assert filename is not None + assert not mock_create.called + + def test_generate_missing_thumbnail_noop_when_thumbnail_present( + self, + document_file, + thumbnail_path, # noqa F811 + ): + node = DocumentNode( + "doc-src-id", "Document", licenses.PUBLIC_DOMAIN, thumbnail=thumbnail_path + ) + node.add_file(document_file) + assert node.generate_missing_thumbnail() == [] # ERROR PATHS ############################################################################ @@ -342,3 +494,109 @@ def test_invalid_mp4_fails(self, invalid_video_file): assert result is None, "expected None result for invalid MP4" assert len(config.FAILED_FILES) == 1, "expected one failed file" assert thumbnail_file.filename is None, "filename should remain None" + + +class TestDeferredTopicThumbnails(object): + def setup_method(self, test_method): + _clear_ricecookerfilecache() + config.FAILED_FILES = [] + + def _content_node(self, cls, source_id, file_obj): + node = cls(source_id, source_id, licenses.PUBLIC_DOMAIN) + node.add_file(file_obj) + return node + + def _build_topic(self, channel, children, thumbnail=None): + topic = TopicNode("test-topic", "Topic", thumbnail=thumbnail) + channel.add_child(topic) + for child in children: + topic.add_child(child) + return topic + + def test_topic_gets_tiled_thumbnail_via_post_pass( + self, channel, document_file, html_file, video_file + ): + children = [ + self._content_node(DocumentNode, "doc-src-id", document_file), + self._content_node(HTML5AppNode, "html-src-id", html_file), + self._content_node(VideoNode, "vid-src-id", video_file), + ] + topic = self._build_topic(channel, children) + + manager = ChannelManager(channel) + filenames = manager.process_tree() + + assert topic.has_thumbnail(), "topic should have a generated tile" + tile_files = [f for f in topic.files if f.is_thumbnail()] + assert len(tile_files) == 1 + tile_filename = tile_files[0].get_filename() + assert tile_filename in filenames, "tile must be registered for upload" + assert os.path.exists(config.get_storage_path(tile_filename)) + + def test_topic_with_provided_thumbnail_is_untouched( + self, + channel, + document_file, + thumbnail_path, # noqa F811 + ): + children = [self._content_node(DocumentNode, "doc-src-id", document_file)] + topic = self._build_topic(channel, children, thumbnail=thumbnail_path) + + manager = ChannelManager(channel) + manager.process_tree() + + thumb_files = [f for f in topic.files if f.is_thumbnail()] + assert len(thumb_files) == 1, "no second thumbnail should be generated" + assert isinstance(thumb_files[0], ThumbnailFile) + assert not isinstance(thumb_files[0], TiledThumbnailFile) + + def test_tile_sources_include_generated_child_thumbnails( + self, channel, document_file + ): + # The child has no provided thumbnail: it generates its own during + # process_files, and the topic tile is built from that generated + # thumbnail afterward. + children = [self._content_node(DocumentNode, "doc-src-id", document_file)] + topic = self._build_topic(channel, children) + + manager = ChannelManager(channel) + manager.process_tree() + assert topic.has_thumbnail() + + def test_topic_with_no_eligible_descendants_stays_thumbnail_less(self, channel): + topic = self._build_topic(channel, []) + + manager = ChannelManager(channel) + filenames = manager.process_tree() + + assert not topic.has_thumbnail() + assert filenames == [] + + def test_failed_tile_generation_is_recorded( + self, + channel, + document_file, + monkeypatch, + ): + children = [self._content_node(DocumentNode, "doc-src-id", document_file)] + topic = self._build_topic(channel, children) + + def boom(*args, **kwargs): + raise ThumbnailGenerationError("tile failure") + + monkeypatch.setattr("ricecooker.classes.files.create_tiled_image", boom) + manager = ChannelManager(channel) + manager.process_tree() + + assert not topic.has_thumbnail() + failed = [f for f in config.FAILED_FILES if isinstance(f, TiledThumbnailFile)] + assert len(failed) == 1 + assert failed[0].error == "tile failure" + + def test_tile_sources_skip_unprocessed_thumbnails(self): + # A thumbnail file that failed to process (filename is None) must be + # skipped as a tile source, not re-processed by the tile build. + node = DocumentNode("doc-src-id", "Document", licenses.PUBLIC_DOMAIN) + node.add_file(File(preset=format_presets.DOCUMENT_THUMBNAIL, filename=None)) + tiled = TiledThumbnailFile([node]) + assert tiled.sources == [] diff --git a/tests/test_tree.py b/tests/test_tree.py index 6960da9c..885d1043 100644 --- a/tests/test_tree.py +++ b/tests/test_tree.py @@ -49,6 +49,7 @@ from ricecooker.managers.tree import InsufficientStorageException from ricecooker.utils.jsontrees import build_tree_from_json from ricecooker.utils.pipeline import FilePipeline +from ricecooker.utils.pipeline.context import NODE_HAS_THUMBNAIL from ricecooker.utils.zip import create_predictable_zip """ *********** TOPIC FIXTURES *********** """ @@ -814,7 +815,9 @@ def test_content_node_passes_context_to_pipeline(): ) node._process_uri() mock_pipeline.execute.assert_called_once_with( - node.uri, context={"subtitle_languages": ["en", "es"]}, skip_cache=False + node.uri, + context={"subtitle_languages": ["en", "es"], NODE_HAS_THUMBNAIL: False}, + skip_cache=False, ) From f6825ff5ac214b97c45cbec3afd3e0ca91390c97 Mon Sep 17 00:00:00 2001 From: Richard Tibbles Date: Sat, 26 Sep 2026 15:16:29 -0700 Subject: [PATCH 2/2] Render HTML5 thumbnails with Playwright. Replace the shelled-out `chrome --headless --screenshot` render with Playwright's bundled Chromium, installed via the new `screenshots` extra plus `playwright install chromium`. The PATH search is dropped; RICECOOKER_CHROMIUM_PATH still overrides the executable. Without Playwright or its browser, HTML5/KPUB thumbnails fall back to the biggest image as before. Tests disable real rendering by default; `real_browser`-marked tests opt back in, and ubuntu CI installs Playwright's Chromium to run them. Co-Authored-By: Claude Opus 5.5 (1M context) --- .github/workflows/pythontest.yml | 5 +- docs/installation.md | 12 +++ pyproject.toml | 4 + ricecooker/utils/images.py | 80 ++++++++--------- tests/conftest.py | 12 +++ tests/media_utils/test_thumbnails.py | 124 +++++++++++++++++++-------- uv.lock | 103 +++++++++++++++++++++- 7 files changed, 255 insertions(+), 85 deletions(-) diff --git a/.github/workflows/pythontest.yml b/.github/workflows/pythontest.yml index 5558602a..a44a6274 100644 --- a/.github/workflows/pythontest.yml +++ b/.github/workflows/pythontest.yml @@ -134,8 +134,11 @@ jobs: Add-Content -Path $env:GITHUB_PATH -Value "$env:GITHUB_WORKSPACE\tools\ffmpeg-master-latest-win64-gpl\bin" -Encoding utf8 Add-Content -Path $env:GITHUB_PATH -Value "$env:GITHUB_WORKSPACE\tools\poppler-21.11.0\Library\bin" -Encoding utf8 Add-Content -Path $env:GITHUB_PATH -Value "$env:GITHUB_WORKSPACE\tools\pandoc-3.1.11" -Encoding utf8 + - name: Install Playwright Chromium + run: uv run --extra screenshots playwright install --with-deps chromium + if: ${{ startsWith(matrix.os, 'ubuntu') }} - name: Run tests - run: uv run --group test --extra google_drive pytest + run: uv run --group test --extra google_drive --extra screenshots pytest # Single stable required check: a skipped matrix job reports no per-combination checks. required_checks: name: Python tests diff --git a/docs/installation.md b/docs/installation.md index 4c527ebc..59490c33 100644 --- a/docs/installation.md +++ b/docs/installation.md @@ -116,6 +116,18 @@ both: *Checklist*: run `single-file --help` to confirm the binary is available. +### Optional: rendered HTML5 thumbnails (Playwright) + +HTML5 and KPUB thumbnails are screenshots of the rendered page when Playwright's +Chromium is available; otherwise the largest image in the zip is used. Install +the extra and its browser: + + uv add "ricecooker[screenshots]" + uv run playwright install chromium + +To use a different Chromium/Chrome, set `RICECOOKER_CHROMIUM_PATH` to its executable. + + Installing Ricecooker --------------------- Create a `pyproject.toml` for your chef project (or use an existing one), then run: diff --git a/pyproject.toml b/pyproject.toml index 42da8b52..2499375b 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -50,6 +50,7 @@ dependencies = [ [project.optional-dependencies] google_drive = ["google-api-python-client", "google-auth"] sentry = ["sentry-sdk>=2.32.0"] +screenshots = ["playwright<2"] [project.scripts] corrections = "ricecooker.utils.corrections:correctionsmain" @@ -97,6 +98,9 @@ docs = { requires-python = ">=3.11" } [tool.pytest.ini_options] testpaths = ["tests"] norecursedirs = ["docs", "examples", "resources"] +markers = [ + "real_browser: use the real Playwright renderer instead of the test default that disables it", +] env = [ "RICECOOKER_STORAGE=./.pytest_storage", "RICECOOKER_FILECACHE=./.pytest_filecache", diff --git a/ricecooker/utils/images.py b/ricecooker/utils/images.py index 141f9921..d03048ec 100644 --- a/ricecooker/utils/images.py +++ b/ricecooker/utils/images.py @@ -1,7 +1,5 @@ import os import pathlib -import shutil -import subprocess import tempfile import zipfile from io import BytesIO @@ -113,35 +111,46 @@ def create_image_from_zip(htmlfile, fpath_out, crop="smart"): raise ThumbnailGenerationError("Fail on zip {} {}".format(htmlfile, e)) -CHROMIUM_BINARY_NAMES = [ - "chromium", - "chromium-browser", - "google-chrome", - "google-chrome-stable", - "chrome", -] +SCREENSHOT_VIEWPORT = {"width": 1600, "height": 900} +SCREENSHOT_TIMEOUT_MS = 30000 -def find_chromium_binary(): - """Return a Chromium/Chrome binary path, or None. Honors ``RICECOOKER_CHROMIUM_PATH``, else searches PATH.""" - env_path = os.environ.get("RICECOOKER_CHROMIUM_PATH") - if env_path: - return env_path - for name in CHROMIUM_BINARY_NAMES: - path = shutil.which(name) - if path: - return path - return None +def render_html_screenshot(url, shot_path): + """Load ``url`` in headless Chromium via Playwright and save a viewport screenshot to ``shot_path``. + + Raises ChromiumUnavailableError when Playwright or its browser is missing. ``RICECOOKER_CHROMIUM_PATH`` overrides the browser executable. + """ + try: + from playwright.sync_api import Error as PlaywrightError + from playwright.sync_api import sync_playwright + except ImportError as e: + raise ChromiumUnavailableError( + "Playwright is not installed; install ricecooker[screenshots]." + ) from e + with sync_playwright() as p: + try: + browser = p.chromium.launch( + executable_path=os.environ.get("RICECOOKER_CHROMIUM_PATH") or None + ) + except PlaywrightError as e: + raise ChromiumUnavailableError( + "Could not launch Chromium; run `playwright install chromium`: {}".format( + e + ) + ) from e + try: + page = browser.new_page(viewport=SCREENSHOT_VIEWPORT) + page.goto(url, wait_until="networkidle", timeout=SCREENSHOT_TIMEOUT_MS) + page.screenshot(path=shot_path) + finally: + browser.close() def create_image_from_zip_screenshot(htmlfile, fpath_out, crop="smart"): """Render the html5 zip's entry point in headless Chromium and write a screenshot thumbnail to ``fpath_out``. - Raises ChromiumUnavailableError if no binary is found (callers fall back), ThumbnailGenerationError on render failure. + Raises ChromiumUnavailableError if no browser is available (callers fall back), ThumbnailGenerationError on render failure. """ - binary = find_chromium_binary() - if binary is None: - raise ChromiumUnavailableError("No Chromium/Chrome binary found.") try: with ( tempfile.TemporaryDirectory() as temp_dir, @@ -154,31 +163,12 @@ def create_image_from_zip_screenshot(htmlfile, fpath_out, crop="smart"): ) zf.extractall(temp_dir) entry_abspath = os.path.join(temp_dir, *entry.split("/")) - profile_dir = os.path.join(temp_dir, "_chromium_profile") shot_path = os.path.join(temp_dir, "shot.png") - url = pathlib.Path(entry_abspath).as_uri() - cmd = [ - binary, - "--headless=new", - "--no-sandbox", - "--disable-gpu", - "--hide-scrollbars", - "--force-device-scale-factor=1", - "--window-size=1600,900", - "--virtual-time-budget=8000", - "--user-data-dir={}".format(profile_dir), - "--screenshot={}".format(shot_path), - url, - ] - subprocess.run(cmd, capture_output=True, timeout=60, check=True) - if not os.path.exists(shot_path): - raise ThumbnailGenerationError( - "Chromium produced no screenshot for {}.".format(htmlfile) - ) + render_html_screenshot(pathlib.Path(entry_abspath).as_uri(), shot_path) img = Image.open(shot_path) img = scale_and_crop_thumbnail(img, crop=crop) img.save(fpath_out) - except ThumbnailGenerationError: + except (ThumbnailGenerationError, ChromiumUnavailableError): raise except Exception as e: raise ThumbnailGenerationError("Fail on zip {} {}".format(htmlfile, e)) @@ -336,4 +326,4 @@ class ThumbnailGenerationError(Exception): class ChromiumUnavailableError(Exception): - """Raised when no Chromium/Chrome binary is available, so callers can fall back to a non-render thumbnail path.""" + """Raised when no Chromium browser is available, so callers can fall back to a non-render thumbnail path.""" diff --git a/tests/conftest.py b/tests/conftest.py index d0019ab6..2f0b9353 100644 --- a/tests/conftest.py +++ b/tests/conftest.py @@ -42,6 +42,7 @@ from ricecooker.classes.nodes import VideoNode from ricecooker.classes.questions import InputQuestion from ricecooker.classes.questions import SingleSelectQuestion +from ricecooker.utils import images # GLOBAL TEST SETUP/TEARDOWN UTILS ################################################################################ @@ -83,6 +84,17 @@ def global_fixture(): pass +@pytest.fixture(autouse=True) +def no_real_browser(request, monkeypatch): + if request.node.get_closest_marker("real_browser"): + return + + def unavailable(url, shot_path): + raise images.ChromiumUnavailableError("Browser rendering is disabled in tests.") + + monkeypatch.setattr(images, "render_html_screenshot", unavailable) + + # Monkey patch VCRHTTPResponse to handle kwargs that are not compatible with BufferIO # This can be removed when this issue has been resolved: https://github.com/kevin1024/vcrpy/issues/902 def _new_read(self, *args, **kwargs): diff --git a/tests/media_utils/test_thumbnails.py b/tests/media_utils/test_thumbnails.py index 212fc825..b7df3517 100644 --- a/tests/media_utils/test_thumbnails.py +++ b/tests/media_utils/test_thumbnails.py @@ -1,6 +1,9 @@ import os +import sys import zipfile +from contextlib import contextmanager from io import BytesIO +from types import SimpleNamespace from unittest.mock import patch import PIL @@ -42,24 +45,17 @@ def make_no_image_zip(path, entry="index.html"): return path -def make_fake_chromium_run(png_bytes): - """ - Return a fake ``subprocess.run`` that writes ``png_bytes`` to the path given by - the ``--screenshot=`` argument in ``cmd`` and records the invoked command. - """ +def make_fake_render(png_bytes): + """Return a fake ``render_html_screenshot`` that writes ``png_bytes`` and records each rendered URL in ``.calls``.""" calls = [] - def fake_run(cmd, *args, **kwargs): - calls.append(cmd) - for arg in cmd: - if arg.startswith("--screenshot="): - shot_path = arg[len("--screenshot=") :] - with open(shot_path, "wb") as f: - f.write(png_bytes) - return None + def fake_render(url, shot_path): + calls.append(url) + with open(shot_path, "wb") as f: + f.write(png_bytes) - fake_run.calls = calls - return fake_run + fake_render.calls = calls + return fake_render def make_png_bytes(size=(1600, 900)): @@ -68,13 +64,42 @@ def make_png_bytes(size=(1600, 900)): return buf.getvalue() +def make_fake_playwright(launches): + """A stand-in ``playwright.sync_api`` module that records each ``chromium.launch`` kwargs in ``launches``.""" + + class Page: + def goto(self, url, **kwargs): + pass + + def screenshot(self, path): + with open(path, "wb") as f: + f.write(make_png_bytes()) + + class Browser: + def new_page(self, **kwargs): + return Page() + + def close(self): + pass + + class Chromium: + def launch(self, **kwargs): + launches.append(kwargs) + return Browser() + + @contextmanager + def sync_playwright(): + yield SimpleNamespace(chromium=Chromium()) + + return SimpleNamespace(Error=Exception, sync_playwright=sync_playwright) + + @pytest.fixture def fake_chromium(monkeypatch): - """Patch the render path to a fake Chromium that writes a PNG; returns the fake run (with ``.calls``).""" - fake_run = make_fake_chromium_run(make_png_bytes()) - monkeypatch.setattr(images, "find_chromium_binary", lambda: "/fake/chromium") - monkeypatch.setattr(images.subprocess, "run", fake_run) - return fake_run + """Patch the render path to a fake browser that writes a PNG; returns the fake render (with ``.calls``).""" + fake_render = make_fake_render(make_png_bytes()) + monkeypatch.setattr(images, "render_html_screenshot", fake_render) + return fake_render # TESTS @@ -214,21 +239,47 @@ def test_raises_for_invalid_zip(self, tmpdir, bad_zip_file): class Test_html_zip_screenshot_thumbnail_generation(BaseThumbnailGeneratorTestCase): - def test_find_chromium_binary_env_override(self, tmpdir, monkeypatch): - fake_binary = tmpdir.join("chromium").strpath - with open(fake_binary, "w") as f: - f.write("") - monkeypatch.setenv("RICECOOKER_CHROMIUM_PATH", fake_binary) - assert images.find_chromium_binary() == fake_binary - - monkeypatch.delenv("RICECOOKER_CHROMIUM_PATH", raising=False) - monkeypatch.setattr(images.shutil, "which", lambda *a, **k: None) - assert images.find_chromium_binary() is None - - def test_screenshot_raises_when_chromium_absent(self, tmpdir, monkeypatch): + @pytest.mark.real_browser + def test_render_unavailable_without_playwright(self, tmpdir, monkeypatch): + monkeypatch.setitem(sys.modules, "playwright.sync_api", None) + with pytest.raises(images.ChromiumUnavailableError): + images.render_html_screenshot( + "file:///index.html", tmpdir.join("out.png").strpath + ) + + @pytest.mark.real_browser + def test_render_launches_env_executable(self, tmpdir, monkeypatch): + launches = [] + monkeypatch.setitem( + sys.modules, "playwright.sync_api", make_fake_playwright(launches) + ) + monkeypatch.setenv("RICECOOKER_CHROMIUM_PATH", "/opt/chrome") + images.render_html_screenshot( + "file:///index.html", tmpdir.join("out.png").strpath + ) + monkeypatch.delenv("RICECOOKER_CHROMIUM_PATH") + images.render_html_screenshot( + "file:///index.html", tmpdir.join("out.png").strpath + ) + assert [launch["executable_path"] for launch in launches] == [ + "/opt/chrome", + None, + ] + + @pytest.mark.real_browser + def test_render_with_real_browser(self, tmpdir): + input_file = make_no_image_zip(tmpdir.join("noimg.zip").strpath) + output_file = tmpdir.join("out.png").strpath + try: + images.create_image_from_zip_screenshot(input_file, output_file) + except images.ChromiumUnavailableError as e: + pytest.skip(str(e)) + self.check_is_png_file(output_file) + self.check_16_9_format(output_file) + + def test_screenshot_raises_when_chromium_absent(self, tmpdir): input_file = make_no_image_zip(tmpdir.join("noimg.zip").strpath) output_file = tmpdir.join("out.png").strpath - monkeypatch.setattr(images, "find_chromium_binary", lambda: None) with pytest.raises(images.ChromiumUnavailableError): images.create_image_from_zip_screenshot(input_file, output_file) @@ -245,19 +296,16 @@ def test_screenshot_resolves_nonroot_entrypoint(self, tmpdir, fake_chromium): ) output_file = tmpdir.join("out.png").strpath images.create_image_from_zip_screenshot(input_file, output_file) - url_arg = fake_chromium.calls[0][-1] + url_arg = fake_chromium.calls[0] assert url_arg.startswith("file://") assert url_arg.endswith("reader/start.html") class Test_html_zip_extractor_fun(BaseThumbnailGeneratorTestCase): - def test_htmlzip_extractor_falls_back_when_chromium_absent( - self, tmpdir, monkeypatch - ): + def test_htmlzip_extractor_falls_back_when_chromium_absent(self, tmpdir): input_file = os.path.join(files_dir, "generate_thumbnail", "sample.zip") assert os.path.exists(input_file) output_file = tmpdir.join("out.png").strpath - monkeypatch.setattr(images, "find_chromium_binary", lambda: None) ExtractedHTMLZipThumbnailFile(input_file).extractor_fun(input_file, output_file) self.check_is_png_file(output_file) diff --git a/uv.lock b/uv.lock index 181fa5da..36e52903 100644 --- a/uv.lock +++ b/uv.lock @@ -607,6 +607,72 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/69/28/23eea8acd65972bbfe295ce3666b28ac510dfcb115fac089d3edb0feb00a/googleapis_common_protos-1.73.0-py3-none-any.whl", hash = "sha256:dfdaaa2e860f242046be561e6d6cb5c5f1541ae02cfbcb034371aadb2942b4e8", size = 297578, upload-time = "2026-03-06T21:52:33.933Z" }, ] +[[package]] +name = "greenlet" +version = "3.5.6" +source = { registry = "https://pypi.org/simple" } +sdist = { url = "https://files.pythonhosted.org/packages/3e/6e/0091f175ccd02b02bc8811bbcbcc6ac2e980be116e3b2f7a736ca322bf84/greenlet-3.5.6.tar.gz", hash = "sha256:8e67c43bdfc88d5fee6db0d3e40175b362fc95fb85f0412d233b9b203c53a575", size = 207653, upload-time = "2026-09-14T15:42:51.806Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a7/4f/4af258e1ce388eb64be82d6466b7d0b3bc3eede154e14c3360a76a759b71/greenlet-3.5.6-cp310-cp310-macosx_11_0_universal2.whl", hash = "sha256:95e7c44d072db623a1aab04ce488cf9533294a77ed9d072cd503a3596f4106ac", size = 292967, upload-time = "2026-09-14T14:26:34.475Z" }, + { url = "https://files.pythonhosted.org/packages/1d/05/dc0d54e90af2f192724936799cee990687792ff54a4638d3ae0e0a0145c0/greenlet-3.5.6-cp310-cp310-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:b7d501d5eb5d4f67207df364752ad697465b834268744be7581c18d81d35d41d", size = 609284, upload-time = "2026-09-14T15:11:58.49Z" }, + { url = "https://files.pythonhosted.org/packages/c1/51/826b4fe7bc39c3f910c95e889a981c827f72a611fc035f31816a256cf043/greenlet-3.5.6-cp310-cp310-manylinux_2_24_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:a364c1ea75dc51b83a17f52fe0c79cf8bc4ddf740403bebd4581c7666eea017d", size = 622648, upload-time = "2026-09-14T15:20:39.198Z" }, + { url = "https://files.pythonhosted.org/packages/55/3b/f2d36fc38934588dff6d6ec9288934c9611abac7a79c866c2002fc707698/greenlet-3.5.6-cp310-cp310-manylinux_2_24_s390x.manylinux_2_28_s390x.whl", hash = "sha256:5599b380c1f28efeb724e81569eac80cd92f99a85bd9775456caaf3225d40b11", size = 629557, upload-time = "2026-09-14T15:25:03.036Z" }, + { url = "https://files.pythonhosted.org/packages/94/5c/092682ae7ca1aadd44aa36b0bc35c48c9ed55f9ffebb5794980b1e2ae74b/greenlet-3.5.6-cp310-cp310-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:eed88b64a5e5da72d6a71cdc5aaeefaa5ced9b748f8d19f89800b339961dad39", size = 622813, upload-time = "2026-09-14T14:35:54.845Z" }, + { url = "https://files.pythonhosted.org/packages/d6/f6/4167f0e44e795840cbc219b2e5d8f576492192cd9e3e17e827db20287daa/greenlet-3.5.6-cp310-cp310-manylinux_2_39_riscv64.whl", hash = "sha256:5bbda3c70dd35d60671bc33b01916802707a052130d9e50cdb871d34594d35cb", size = 425482, upload-time = "2026-09-14T15:28:34.144Z" }, + { url = "https://files.pythonhosted.org/packages/07/59/9c6b723a559b53bee0980a5dc57577d434891dd454f0fe54ed6e22d4e4ce/greenlet-3.5.6-cp310-cp310-musllinux_1_2_aarch64.whl", hash = "sha256:874cea8bb1ec1ddccbacbd027856f6bf496f6bc18aba97a918c20e067edab236", size = 1585817, upload-time = "2026-09-14T15:10:03.922Z" }, + { url = "https://files.pythonhosted.org/packages/de/d4/2bceaa305ff0ecf99511480da8f0a866eb2469a68da8715008a0d3e8f65f/greenlet-3.5.6-cp310-cp310-musllinux_1_2_x86_64.whl", hash = "sha256:128813fc29f2336a21b4d06eedd5e16bcc7ea46f59e9ff1cb30ea70e48195d88", size = 1650177, upload-time = "2026-09-14T14:35:46.736Z" }, + { url = "https://files.pythonhosted.org/packages/7a/33/c57855a6abada0c7bfb9c6c0f33df1fbed8e5224708478dbcdff6c5db490/greenlet-3.5.6-cp310-cp310-win_amd64.whl", hash = "sha256:dad3d233d441a022c1f7155f0fb9d5aff7b97c1ea8c7dfa02cce586b16ab2d0b", size = 322896, upload-time = "2026-09-14T14:24:52.308Z" }, + { url = "https://files.pythonhosted.org/packages/f1/d7/41511ee2696f14be4200b524d9553dc4295e2bdeb20aa8962c3cb25e71c6/greenlet-3.5.6-cp311-cp311-macosx_11_0_universal2.whl", hash = "sha256:a6a4b98a9132e0f45c9fc245a63894cfd8c45fb7a0d6bffc5eab3ec327cf7324", size = 294075, upload-time = "2026-09-14T14:25:16.922Z" }, + { url = "https://files.pythonhosted.org/packages/f8/7b/b509624970909294cd064ff7346148ca9941c21bec9026d7873dd254e9fa/greenlet-3.5.6-cp311-cp311-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:45bfd2b51e38aaa5f9849f114d9c7c1d75f69187c849b3549cd64c465283abfa", size = 613429, upload-time = "2026-09-14T15:12:00.454Z" }, + { url = "https://files.pythonhosted.org/packages/2b/5c/d2eb503067f9ba20875ef8c87681f29a64f53bbbbe4059a5d7c53179d442/greenlet-3.5.6-cp311-cp311-manylinux_2_24_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:3c6dede9133e1da41d561bc3fb14e92b47e2ce39ae60edefaad145658ea7c5e2", size = 625405, upload-time = "2026-09-14T15:20:41.053Z" }, + { url = "https://files.pythonhosted.org/packages/1b/24/9b071d11c8bb9f5f38cccacc38fcc234d91997a4c395cc2bf43ecae89642/greenlet-3.5.6-cp311-cp311-manylinux_2_24_s390x.manylinux_2_28_s390x.whl", hash = "sha256:4fb8e59f68845d56c23c031dcd79c329f345e4a9d2ffac91c3d1ab366bdc457b", size = 633270, upload-time = "2026-09-14T15:25:04.864Z" }, + { url = "https://files.pythonhosted.org/packages/ec/d3/63d4477ce31dff2fd802a9a20240f6606aac85977e0fb18443aae33de3f6/greenlet-3.5.6-cp311-cp311-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:1c20ea32a73d17b9b60e3371240e17b0068120c98a5ec01a224a7dd8c89733ba", size = 624428, upload-time = "2026-09-14T14:35:56.895Z" }, + { url = "https://files.pythonhosted.org/packages/88/17/ac11883ecc9da19c681c8b763ee39e6f7dca2aa81874eb11a075d3cbeb00/greenlet-3.5.6-cp311-cp311-manylinux_2_39_riscv64.whl", hash = "sha256:d701eab36200c36224833d07dbdb709adb7fd4253429548ddb5e547b8ed40586", size = 428068, upload-time = "2026-09-14T15:28:35.872Z" }, + { url = "https://files.pythonhosted.org/packages/ad/aa/9cde4e00688eaa2a03b91d12e4681439a87e6aad860399e0847af6a014ca/greenlet-3.5.6-cp311-cp311-musllinux_1_2_aarch64.whl", hash = "sha256:5a0b2791239c99992a86c1b635b787fe2a877d9eaaa26f8891ce943832b585ae", size = 1588385, upload-time = "2026-09-14T15:10:05.386Z" }, + { url = "https://files.pythonhosted.org/packages/5c/01/24632b5ec186b64e21e07a8f53ce5e15a7e9cb33eddee99a5fe16379afa5/greenlet-3.5.6-cp311-cp311-musllinux_1_2_x86_64.whl", hash = "sha256:188bf333769b7145e2b0b4a7f09615ec550ed44d3a2a8395fb7b36f0e9901e13", size = 1653119, upload-time = "2026-09-14T14:35:48.275Z" }, + { url = "https://files.pythonhosted.org/packages/ce/6c/019d2ef898f4b9ac845167f1c6f73229e9a4e2439362a5e2ce50205a19b0/greenlet-3.5.6-cp311-cp311-win_amd64.whl", hash = "sha256:a6b4ff33f7e011bbaa148238d131c4fd4f8afbab3c104ddfbdb2b12b74ff7016", size = 323317, upload-time = "2026-09-14T14:22:38.836Z" }, + { url = "https://files.pythonhosted.org/packages/5a/7a/439df999455e3bdf02b1c68f3848d4020385ef0a01f89f706b07bf148a65/greenlet-3.5.6-cp311-cp311-win_arm64.whl", hash = "sha256:59deccd347735a7774223b05a93773fddbb298aba3cea21be4337fb4752dbe32", size = 307739, upload-time = "2026-09-14T14:23:40.469Z" }, + { url = "https://files.pythonhosted.org/packages/72/18/3fc6d951466ae9a2a688edcddde3b2e388da0a8244e0caf7117bbeb0eb95/greenlet-3.5.6-cp312-cp312-macosx_11_0_universal2.whl", hash = "sha256:a5876d0a60355af98d535c47f6cd6eb0f8a432396dab26845d380b92f8412422", size = 295668, upload-time = "2026-09-14T14:22:33.241Z" }, + { url = "https://files.pythonhosted.org/packages/27/89/366d2af5061eeefa5012f510d95a99c8620dcc457609838db4d538820318/greenlet-3.5.6-cp312-cp312-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:e85880b538e59a59f55117b81f208a6660ad5ac328aad9305f812d9b8bc67a0f", size = 611700, upload-time = "2026-09-14T15:12:01.962Z" }, + { url = "https://files.pythonhosted.org/packages/54/1c/07f133f865fd58ae593dd2bbec3144acaee9b04ffe2eb48c6e121747ceef/greenlet-3.5.6-cp312-cp312-manylinux_2_24_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:f0ba7c2a329d650628f4c8572fd1db29f0a59dd70a3e3e0710dcf18a35cce9d8", size = 624223, upload-time = "2026-09-14T15:20:42.459Z" }, + { url = "https://files.pythonhosted.org/packages/a7/f2/844dc823ff2752ad049caa6b59d57e4572f9c445934b02d3518f4c67197c/greenlet-3.5.6-cp312-cp312-manylinux_2_24_s390x.manylinux_2_28_s390x.whl", hash = "sha256:ee7d9da3bf493909cf811a3f038840cb34fab5ae2956b8a263919f6e289ab188", size = 629529, upload-time = "2026-09-14T15:25:06.354Z" }, + { url = "https://files.pythonhosted.org/packages/66/6a/1594f3869c57c149abdb380492529e04d4c0229b5e4d79572c5bd0aaa673/greenlet-3.5.6-cp312-cp312-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:975736b002ed080d124cf81a79cb7e05cb26d6b3f5c7a7b651c0fcce70353aa1", size = 621404, upload-time = "2026-09-14T14:35:59.027Z" }, + { url = "https://files.pythonhosted.org/packages/c0/42/b1f8dbc89a53b9e77859fc1ad1627d106fc361daa3ea4bdf43a91ebb4338/greenlet-3.5.6-cp312-cp312-manylinux_2_39_riscv64.whl", hash = "sha256:71890d5247020c25c21a6b65202782bfc281d4e6e244842419d30e3492bb6dcc", size = 432385, upload-time = "2026-09-14T15:28:37.369Z" }, + { url = "https://files.pythonhosted.org/packages/a2/f5/33e5c9e48178b9259fd000f8f45caa4a65036f65d3d0c06a602f570f025d/greenlet-3.5.6-cp312-cp312-musllinux_1_2_aarch64.whl", hash = "sha256:0616b8f878098c5681fd8f0dc92d887551717402342a70f0abcbfea5f5ad8a44", size = 1584998, upload-time = "2026-09-14T15:10:06.653Z" }, + { url = "https://files.pythonhosted.org/packages/ef/31/9b4e140bc24d0ad7927ebd651f5608b0acc2334d061748c3b6ad19085cfa/greenlet-3.5.6-cp312-cp312-musllinux_1_2_x86_64.whl", hash = "sha256:3dbb4596a6a4e5d47121a33ff20533a81e60f302d9e67b69909a8bc21a43f0a7", size = 1647568, upload-time = "2026-09-14T14:35:49.787Z" }, + { url = "https://files.pythonhosted.org/packages/c3/71/d79f1791f824f8ff15c2978746640467ae932a2365e0201069f7f272395f/greenlet-3.5.6-cp312-cp312-win_amd64.whl", hash = "sha256:7ac4abb3877c43af320392c664774eef6fa2cc063c79a55fc02d844a3cbe7395", size = 324203, upload-time = "2026-09-14T14:22:54.504Z" }, + { url = "https://files.pythonhosted.org/packages/63/af/42aca4d56e8cb321912203069d8d34734cb288222f10ad2ae102718cc577/greenlet-3.5.6-cp312-cp312-win_arm64.whl", hash = "sha256:301102a49120b095e72a7838792b41233975fc1c155daec6d98f81c00c9280e0", size = 308310, upload-time = "2026-09-14T14:24:03.008Z" }, + { url = "https://files.pythonhosted.org/packages/f1/a1/e720a38852366c589e1a46cf570b886507ad2cf591050c203365638baab0/greenlet-3.5.6-cp313-cp313-macosx_11_0_universal2.whl", hash = "sha256:f96f0e30b5a95c7631b12bfe214cbc90ec8fe8cfa36920596c10514a65743519", size = 294627, upload-time = "2026-09-14T14:24:40.102Z" }, + { url = "https://files.pythonhosted.org/packages/eb/c3/58187858df41354a11e6a55b421e7af9059798abdab3a384cc51b8567c38/greenlet-3.5.6-cp313-cp313-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c75116c9de79949de23006e2d9b35ee82874c594fcf5c0311b439acaa14b8441", size = 614356, upload-time = "2026-09-14T15:12:03.399Z" }, + { url = "https://files.pythonhosted.org/packages/ce/b9/3a7e67d5f05c9760b1ad411fa52264bd69cc08e22a2ebfb4018b90628ced/greenlet-3.5.6-cp313-cp313-manylinux_2_24_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:cad5782f93f7f738b62c6527b6f32a60694d924029f299a8b524758cfa53d815", size = 626756, upload-time = "2026-09-14T15:20:44.269Z" }, + { url = "https://files.pythonhosted.org/packages/c6/7c/40400455f5b5a65bb83e94fde66d1be9e5ec518638113f8083ace746c309/greenlet-3.5.6-cp313-cp313-manylinux_2_24_s390x.manylinux_2_28_s390x.whl", hash = "sha256:a93ee7c6e8fd0f8a83525a51bd777be57ee17787e91d805bd8d6faf9dcada18e", size = 632632, upload-time = "2026-09-14T15:25:07.813Z" }, + { url = "https://files.pythonhosted.org/packages/85/cb/ab0c123c514ed4e94c0dc9ee2e86362633e6b998cfc05de7fc9ac2eb9690/greenlet-3.5.6-cp313-cp313-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:f98e8215e172f567ce80eeaed9107fb4d32b6c44f26983d9b8334658136a205a", size = 623779, upload-time = "2026-09-14T14:36:01.104Z" }, + { url = "https://files.pythonhosted.org/packages/f9/67/1f35cff30a6c51c3f23b63d4afcc7313ab4f97490ba3676fa78178984b27/greenlet-3.5.6-cp313-cp313-manylinux_2_39_riscv64.whl", hash = "sha256:7f731ebac68ea06d628658295cb2d217b10186329fcf9a3b6a149045059bf92e", size = 434933, upload-time = "2026-09-14T15:28:38.858Z" }, + { url = "https://files.pythonhosted.org/packages/a5/26/fda8a5a06e7073333ccb038133c5893b9e0c4fe29d5992a17e83c241bc6e/greenlet-3.5.6-cp313-cp313-musllinux_1_2_aarch64.whl", hash = "sha256:df19e2d0b1620039af5102563fbd96e8938c7f5c3f5828528d641d9fc585525e", size = 1584930, upload-time = "2026-09-14T15:10:08.234Z" }, + { url = "https://files.pythonhosted.org/packages/2f/37/50f8813163148d6234e08b23dcad6a9e37f01d148c8ec976e4c44ea2d918/greenlet-3.5.6-cp313-cp313-musllinux_1_2_x86_64.whl", hash = "sha256:06c0e933290fba8ffe53ead4ae1b8044b0e9754b75cebf381aa2bc3e50d82fac", size = 1647590, upload-time = "2026-09-14T14:35:51.173Z" }, + { url = "https://files.pythonhosted.org/packages/86/da/b7669b09586365654083a62bd0724cf06cb74bd5085a15cdd161271f992f/greenlet-3.5.6-cp313-cp313-win_amd64.whl", hash = "sha256:5b602b4201b965a8354d74e232364a66ff243dd142e350d035f46169bb36e13d", size = 324086, upload-time = "2026-09-14T14:23:48.428Z" }, + { url = "https://files.pythonhosted.org/packages/e5/5d/c9663cfe84a2a9e0aa96f066f5b0594c227ea4c647511e087e2e11d4ac0a/greenlet-3.5.6-cp313-cp313-win_arm64.whl", hash = "sha256:876077e7ebb8c84ed068e2b23d4c62ebb010d60df84b9591af1be2f39010ffb2", size = 308211, upload-time = "2026-09-14T14:28:01.634Z" }, + { url = "https://files.pythonhosted.org/packages/66/c0/d254544ae2b8bdd311aef000fafc02828c2771b17d994b3075620ea7cc6e/greenlet-3.5.6-cp314-cp314-macosx_11_0_universal2.whl", hash = "sha256:8cddea1b8339451c2fb3388e138347b6126744f33b611bdb55b7357361cfef46", size = 295221, upload-time = "2026-09-14T14:25:11.583Z" }, + { url = "https://files.pythonhosted.org/packages/18/18/eb54be16b9cc3971e09ca5b73334e1b8c804a4630d9addaaf218a4fe300f/greenlet-3.5.6-cp314-cp314-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:c59acfa8eb73a1e0d484392dc002bdf001fd4ce73394e0132df3d1ab6093d7cb", size = 660992, upload-time = "2026-09-14T15:12:04.876Z" }, + { url = "https://files.pythonhosted.org/packages/8f/b4/e193efe65671dcf294bc51fcc59efb52d154adf8612c4ea016da0d2c486c/greenlet-3.5.6-cp314-cp314-manylinux_2_24_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:a3b4a01c6da07ef9f80d4fe8933b994bc99747bcea3eab0330a9c34d3c12655b", size = 673428, upload-time = "2026-09-14T15:20:45.756Z" }, + { url = "https://files.pythonhosted.org/packages/fd/21/631bb45fafde1dca782152377c0676d182ec924820064047f533a3627b28/greenlet-3.5.6-cp314-cp314-manylinux_2_24_s390x.manylinux_2_28_s390x.whl", hash = "sha256:dd0b83bed3405b586a3133629f1d1a5bc7bfd64822a3b7ab342bdc68e6dbc61b", size = 677688, upload-time = "2026-09-14T15:25:09.279Z" }, + { url = "https://files.pythonhosted.org/packages/45/ac/28fa7a9e50f2859466214c4ac584d776db52c1604ad4dd158960a5af2a1f/greenlet-3.5.6-cp314-cp314-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:9a09d59bef1db94f384b5bcc2d523694d338f3df6b757aeeaf7baca5d0c0be88", size = 670773, upload-time = "2026-09-14T14:36:02.577Z" }, + { url = "https://files.pythonhosted.org/packages/40/30/2b0a73e68e1e18e30b601d0d183cfdfc2beca4de5a6843c630f0fc9fb90c/greenlet-3.5.6-cp314-cp314-manylinux_2_39_riscv64.whl", hash = "sha256:fdacf26402389bdd89857ad3c045a26fe8f3314f9a8b28226f82f88463a65b77", size = 480475, upload-time = "2026-09-14T15:28:40.741Z" }, + { url = "https://files.pythonhosted.org/packages/c3/cd/fb7d6cdd86ff3427c1494854f0e35437eba05142be91f530f6da75e09e19/greenlet-3.5.6-cp314-cp314-musllinux_1_2_aarch64.whl", hash = "sha256:8b7c73d1cef3d9ae963e9ff03f6222df43efbb9054ffd2f1969c935b7fc84c02", size = 1631900, upload-time = "2026-09-14T15:10:09.745Z" }, + { url = "https://files.pythonhosted.org/packages/f6/40/143bdbb20a516628cb15074ae52ed17d850b450292609c7a6fccac6dbece/greenlet-3.5.6-cp314-cp314-musllinux_1_2_x86_64.whl", hash = "sha256:8b27df301f56e3b3d2298095c8f7d6b68f2521f6b1693e901fa039bdbae34424", size = 1693740, upload-time = "2026-09-14T14:35:52.959Z" }, + { url = "https://files.pythonhosted.org/packages/c9/9e/019642432e6ae283301df1361227d47610709d2dc69a38f95edef266d713/greenlet-3.5.6-cp314-cp314-win_amd64.whl", hash = "sha256:f8f0bd690e1a41294ac87905e8121c81a3761ec2583c768f13467428606c8c7a", size = 327473, upload-time = "2026-09-14T14:28:12.948Z" }, + { url = "https://files.pythonhosted.org/packages/e9/7f/8aafc7bf70c948786dba7221d0dc0838e5329bebc6d434ef2208b4f0e760/greenlet-3.5.6-cp314-cp314-win_arm64.whl", hash = "sha256:8cda13494d86a4f12429641117cb6ac4bbbc9c30a33f711f7d3a2e5fbe4b0b7e", size = 311095, upload-time = "2026-09-14T14:28:00.7Z" }, + { url = "https://files.pythonhosted.org/packages/14/7e/7a205688a5b3074933b18a906608d46d106e9a79d776bdab5a4abf4b4feb/greenlet-3.5.6-cp314-cp314t-macosx_11_0_universal2.whl", hash = "sha256:97c5a53e8c1754df58e73f047a99e287d4da1bdfe64b0072fb25c87000897951", size = 305352, upload-time = "2026-09-14T14:21:31.962Z" }, + { url = "https://files.pythonhosted.org/packages/78/cb/9c4a57a9d9dd0256e20b8f7f4f06554c2c92badebf0ab73ce344321b78b9/greenlet-3.5.6-cp314-cp314t-manylinux_2_24_aarch64.manylinux_2_28_aarch64.whl", hash = "sha256:fea4427d1ffdb3b523d7daa6712038428a4c16c450b9777bdd1221cfee0eab49", size = 672671, upload-time = "2026-09-14T15:12:06.347Z" }, + { url = "https://files.pythonhosted.org/packages/97/52/c6729681ebbd298f4decd28746815acc8a0b0a0fde21d2df33776fd4d042/greenlet-3.5.6-cp314-cp314t-manylinux_2_24_ppc64le.manylinux_2_28_ppc64le.whl", hash = "sha256:73a29b5ba642e35433166a03a3e02935e7238c4b3467fbd77523b99edea23e5b", size = 679489, upload-time = "2026-09-14T15:20:47.291Z" }, + { url = "https://files.pythonhosted.org/packages/71/76/3c11c21e0716b1f1dc7c1a4b3d690abb1d3b448c69a9d32049fecb64010a/greenlet-3.5.6-cp314-cp314t-manylinux_2_24_s390x.manylinux_2_28_s390x.whl", hash = "sha256:61a61b4a95a4f97922c3a6f5606d3e360851584bd47e500a5161373c53810e3d", size = 681303, upload-time = "2026-09-14T15:25:11.088Z" }, + { url = "https://files.pythonhosted.org/packages/58/c5/2b6c721ba8b8963da42d5a0f57f25b8aaeb1fe9bdd156875e57f3be648a2/greenlet-3.5.6-cp314-cp314t-manylinux_2_24_x86_64.manylinux_2_28_x86_64.whl", hash = "sha256:460e70b033aba8ed47e2ac9b5d0d2157b05a34fbfa30a241400aef4118902cdc", size = 676608, upload-time = "2026-09-14T14:36:03.959Z" }, + { url = "https://files.pythonhosted.org/packages/3f/26/3ae402202452cd5941bbbd483e5a74297e2397e7aa3182c2a5e3ab7d5666/greenlet-3.5.6-cp314-cp314t-manylinux_2_39_riscv64.whl", hash = "sha256:fe3170a69fe039b18ad18171e66faa9a75f6fe9d78f968fd9b54e09fbd714d81", size = 510112, upload-time = "2026-09-14T15:28:42.112Z" }, + { url = "https://files.pythonhosted.org/packages/b2/04/0d018e0d05bcdde19a0fcb907834155f1fc853a9bedd3f3f5e6acadcae19/greenlet-3.5.6-cp314-cp314t-musllinux_1_2_aarch64.whl", hash = "sha256:ca80a49b53ed1d22f7282da7255f7bb2fd1935fd0f623d8613fda38745f18961", size = 1641479, upload-time = "2026-09-14T15:10:11.216Z" }, + { url = "https://files.pythonhosted.org/packages/59/bb/f02ef9073919158f6403fe3701d4ed4403d646720e7201dfc6e9d264bac3/greenlet-3.5.6-cp314-cp314t-musllinux_1_2_x86_64.whl", hash = "sha256:916f92f2a8db10508f739d0b5e00b83defe5d1115a997c54532a6d7cf8c95404", size = 1698758, upload-time = "2026-09-14T14:35:54.336Z" }, + { url = "https://files.pythonhosted.org/packages/08/a5/1f48fe647473a2dcccfd1839b2ff2c78eb57009be776b4da071e901c9bff/greenlet-3.5.6-cp314-cp314t-win_amd64.whl", hash = "sha256:886bcf1870af74c32bc310fd00a6b803445e17e51b7d5a107c7b35c0f362cc16", size = 331574, upload-time = "2026-09-14T14:27:18.451Z" }, +] + [[package]] name = "h11" version = "0.16.0" @@ -1679,6 +1745,25 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/63/d7/97f7e3a6abb67d8080dd406fd4df842c2be0efaf712d1c899c32a075027c/platformdirs-4.9.4-py3-none-any.whl", hash = "sha256:68a9a4619a666ea6439f2ff250c12a853cd1cbd5158d258bd824a7df6be2f868", size = 21216, upload-time = "2026-03-05T18:34:12.172Z" }, ] +[[package]] +name = "playwright" +version = "1.63.0" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "greenlet" }, + { name = "pyee" }, +] +wheels = [ + { url = "https://files.pythonhosted.org/packages/19/dd/fbb3d34228ad753bc14464d5f2252585b73367539775a88d0a3949cf5a45/playwright-1.63.0-py3-none-macosx_10_13_x86_64.whl", hash = "sha256:84c540759e8e7f01e690e197e04060f48899d37cea322299f255843273d3385d", size = 44278407, upload-time = "2026-09-15T16:49:06.045Z" }, + { url = "https://files.pythonhosted.org/packages/f5/9a/948b930b1a8c4ee869e5a139a2b7747caa06aab56a3f09a2f0abdcbda221/playwright-1.63.0-py3-none-macosx_11_0_arm64.whl", hash = "sha256:fd1aa00631d44d55e56e0975bf3f3f285fac4a9fd2813183f0c12a488f1a1b24", size = 42944679, upload-time = "2026-09-15T16:49:10.039Z" }, + { url = "https://files.pythonhosted.org/packages/94/11/dc5c13fa1602371603acd461be47529c1b3513815d3a0dc98f642c291a10/playwright-1.63.0-py3-none-macosx_11_0_universal2.whl", hash = "sha256:c89fc4736502a1f0fac2c8ca5d10c0cbc1c669f1f4774a2d8507a43140e4d53f", size = 44278410, upload-time = "2026-09-15T16:49:13.861Z" }, + { url = "https://files.pythonhosted.org/packages/27/9c/103a5037789062bdab27c7dca53f3ca6b075b572ab2cd96eec825b3aec4e/playwright-1.63.0-py3-none-manylinux1_x86_64.whl", hash = "sha256:ad21bc07516b187965a7521c5cf0df0bd657b17482eaad74335272d35a2b07de", size = 48217159, upload-time = "2026-09-15T16:49:17.404Z" }, + { url = "https://files.pythonhosted.org/packages/f3/82/3d85505284c5a210f2da6c07b8f757524e79d1fba9cfdafe1eafb766ae59/playwright-1.63.0-py3-none-manylinux_2_17_aarch64.manylinux2014_aarch64.whl", hash = "sha256:354e15b29503565fc598b89f16fbe070459343bef9d7498a93e304864000c6a7", size = 47902708, upload-time = "2026-09-15T16:49:21.055Z" }, + { url = "https://files.pythonhosted.org/packages/69/8d/f74ff6b52751859f4b69caf66aa7d4a3c8d6c1f0c7dc9920da103612ee39/playwright-1.63.0-py3-none-win32.whl", hash = "sha256:660c00c62639e31b16700ba5456b351ddba55bb766b7ce8261223aa34928e482", size = 38606075, upload-time = "2026-09-15T16:49:29.546Z" }, + { url = "https://files.pythonhosted.org/packages/76/eb/d6b8d92658038e260dbc7dd69fb3fdbe245aac283953de8b50b02bfe5f61/playwright-1.63.0-py3-none-win_amd64.whl", hash = "sha256:2f9a707a6c6c91157ed77bff2b8caeb04b3c8d46e70d585fc134298cbe4b5cc6", size = 38606082, upload-time = "2026-09-15T16:49:32.838Z" }, + { url = "https://files.pythonhosted.org/packages/be/75/2432b3e3c7c62103b72d4c4cc8b16a56383ada372bbb0d1278591e988f71/playwright-1.63.0-py3-none-win_arm64.whl", hash = "sha256:1e4a3a838ce22fb68ad17193fcd142a19610d9d70fb9966d2239f5dddc0cc05b", size = 34523554, upload-time = "2026-09-15T16:49:36.455Z" }, +] + [[package]] name = "pluggy" version = "1.6.0" @@ -1908,6 +1993,18 @@ wheels = [ { url = "https://files.pythonhosted.org/packages/36/c7/cfc8e811f061c841d7990b0201912c3556bfeb99cdcb7ed24adc8d6f8704/pydantic_core-2.41.5-pp311-pypy311_pp73-win_amd64.whl", hash = "sha256:56121965f7a4dc965bff783d70b907ddf3d57f6eba29b6d2e5dabfaf07799c51", size = 2145302, upload-time = "2025-11-04T13:43:46.64Z" }, ] +[[package]] +name = "pyee" +version = "13.0.1" +source = { registry = "https://pypi.org/simple" } +dependencies = [ + { name = "typing-extensions" }, +] +sdist = { url = "https://files.pythonhosted.org/packages/8b/04/e7c1fe4dc78a6fdbfd6c337b1c3732ff543b8a397683ab38378447baa331/pyee-13.0.1.tar.gz", hash = "sha256:0b931f7c14535667ed4c7e0d531716368715e860b988770fc7eb8578d1f67fc8", size = 31655, upload-time = "2026-02-14T21:12:28.044Z" } +wheels = [ + { url = "https://files.pythonhosted.org/packages/a0/c4/b4d4827c93ef43c01f599ef31453ccc1c132b353284fc6c87d535c233129/pyee-13.0.1-py3-none-any.whl", hash = "sha256:af2f8fede4171ef667dfded53f96e2ed0d6e6bd7ee3bb46437f77e3b57689228", size = 15659, upload-time = "2026-02-14T21:12:26.263Z" }, +] + [[package]] name = "pygments" version = "2.20.0" @@ -2210,6 +2307,9 @@ google-drive = [ { name = "google-api-python-client" }, { name = "google-auth" }, ] +screenshots = [ + { name = "playwright" }, +] sentry = [ { name = "sentry-sdk" }, ] @@ -2256,6 +2356,7 @@ requires-dist = [ { name = "lxml", specifier = ">=4.9" }, { name = "pdf2image", specifier = "==1.17.0" }, { name = "pillow", specifier = "==11.3.0" }, + { name = "playwright", marker = "extra == 'screenshots'", specifier = "<2" }, { name = "pypdf2", specifier = "==1.26.0" }, { name = "requests", specifier = ">=2.11.1" }, { name = "requests-file" }, @@ -2264,7 +2365,7 @@ requires-dist = [ { name = "urllib3", specifier = "==2.6.3" }, { name = "yt-dlp", specifier = ">=2024.12.23" }, ] -provides-extras = ["google-drive", "sentry"] +provides-extras = ["google-drive", "sentry", "screenshots"] [package.metadata.requires-dev] dev = [