Source code for scrapy.utils.project

from __future__ import annotations

import os
import warnings
from importlib import import_module
from pathlib import Path
from typing import TYPE_CHECKING

from scrapy.exceptions import NotConfigured
from scrapy.settings import Settings
from scrapy.utils.conf import closest_scrapy_cfg, get_config, init_env

if TYPE_CHECKING:
    from collections.abc import Iterable, Iterator

ENVVAR = "SCRAPY_SETTINGS_MODULE"
DATADIR_CFG_SECTION = "datadir"


[docs] def find_projects( path: str | os.PathLike[str] = ".", *, max_depth: int | None = None, ignored_dirs: Iterable[str] = (), ) -> Iterator[Path]: """Yield the root directory of every Scrapy project found in *path* or, recursively, in any of its subdirectories. .. versionadded:: 2.19.0 Once a project is found, its subdirectories are not searched. Hidden directories and any directory named in *ignored_dirs* are not searched either, and neither are virtual environments, identified by a :file:`pyvenv.cfg` file, regardless of their directory name. *max_depth* limits how deep to search: 0 only checks *path* itself, 1 also checks its direct subdirectories, and so on. By default, the whole directory tree is searched. """ root = Path(path) ignored = frozenset(ignored_dirs) for dirpath, dirnames, filenames in os.walk(root): current = Path(dirpath) if "scrapy.cfg" in filenames: dirnames.clear() yield current continue if "pyvenv.cfg" in filenames: dirnames.clear() continue if max_depth is not None and len(current.relative_to(root).parts) >= max_depth: dirnames.clear() continue dirnames[:] = sorted( dirname for dirname in dirnames if not dirname.startswith(".") and dirname not in ignored )
def inside_project() -> bool: scrapy_module = os.environ.get(ENVVAR) if scrapy_module: try: import_module(scrapy_module) except ImportError as exc: warnings.warn( f"Cannot import scrapy settings module {scrapy_module}: {exc}", stacklevel=2, ) else: return True return bool(closest_scrapy_cfg()) def project_data_dir(project: str = "default") -> str: """Return the current project data dir, creating it if it doesn't exist""" if not inside_project(): raise NotConfigured("Not inside a project") cfg = get_config() if cfg.has_option(DATADIR_CFG_SECTION, project): d = Path(cfg.get(DATADIR_CFG_SECTION, project)) else: scrapy_cfg = closest_scrapy_cfg() if not scrapy_cfg: raise NotConfigured( "Unable to find scrapy.cfg file to infer project data dir" ) d = (Path(scrapy_cfg).parent / ".scrapy").resolve() if not d.exists(): d.mkdir(parents=True) return str(d) def data_path(path: str | os.PathLike[str], createdir: bool = False) -> str: """ Return the given path joined with the .scrapy data directory. If given an absolute path, return it unmodified. """ path_obj = Path(path) if not path_obj.is_absolute(): if inside_project(): path_obj = Path(project_data_dir(), path) else: path_obj = Path(".scrapy", path) if createdir and not path_obj.exists(): path_obj.mkdir(parents=True) return str(path_obj)
[docs] def get_project_settings() -> Settings: """Return a :class:`~scrapy.settings.Settings` object with the :ref:`project settings <project-settings>` of the current project. Settings from sources with a higher precedence, such as :ref:`spider settings <spider-settings>`, are applied when a crawl starts. """ init_env(os.environ.get("SCRAPY_PROJECT", "default")) settings = Settings() settings_module_path = os.environ.get(ENVVAR) if settings_module_path: settings.setmodule(settings_module_path, priority="project") valid_envvars = { "CHECK", "PROJECT", "PYTHON_SHELL", "SETTINGS_MODULE", } scrapy_envvars = { k[7:]: v for k, v in os.environ.items() if k.startswith("SCRAPY_") and k.replace("SCRAPY_", "") in valid_envvars } settings.setdict(scrapy_envvars, priority="project") return settings