from __future__ import annotations
import os
import warnings
from importlib import import_module
from pathlib import Path
from typing import TYPE_CHECKING
from scrapy.exceptions import NotConfigured
from scrapy.settings import Settings
from scrapy.utils.conf import closest_scrapy_cfg, get_config, init_env
if TYPE_CHECKING:
from collections.abc import Iterable, Iterator
ENVVAR = "SCRAPY_SETTINGS_MODULE"
DATADIR_CFG_SECTION = "datadir"
[docs]
def find_projects(
path: str | os.PathLike[str] = ".",
*,
max_depth: int | None = None,
ignored_dirs: Iterable[str] = (),
) -> Iterator[Path]:
"""Yield the root directory of every Scrapy project found in *path* or,
recursively, in any of its subdirectories.
.. versionadded:: 2.19.0
Once a project is found, its subdirectories are not searched.
Hidden directories and any directory named in *ignored_dirs* are not
searched either, and neither are virtual environments, identified by a
:file:`pyvenv.cfg` file, regardless of their directory name.
*max_depth* limits how deep to search: 0 only checks *path* itself, 1 also
checks its direct subdirectories, and so on. By default, the whole
directory tree is searched.
"""
root = Path(path)
ignored = frozenset(ignored_dirs)
for dirpath, dirnames, filenames in os.walk(root):
current = Path(dirpath)
if "scrapy.cfg" in filenames:
dirnames.clear()
yield current
continue
if "pyvenv.cfg" in filenames:
dirnames.clear()
continue
if max_depth is not None and len(current.relative_to(root).parts) >= max_depth:
dirnames.clear()
continue
dirnames[:] = sorted(
dirname
for dirname in dirnames
if not dirname.startswith(".") and dirname not in ignored
)
def inside_project() -> bool:
scrapy_module = os.environ.get(ENVVAR)
if scrapy_module:
try:
import_module(scrapy_module)
except ImportError as exc:
warnings.warn(
f"Cannot import scrapy settings module {scrapy_module}: {exc}",
stacklevel=2,
)
else:
return True
return bool(closest_scrapy_cfg())
def project_data_dir(project: str = "default") -> str:
"""Return the current project data dir, creating it if it doesn't exist"""
if not inside_project():
raise NotConfigured("Not inside a project")
cfg = get_config()
if cfg.has_option(DATADIR_CFG_SECTION, project):
d = Path(cfg.get(DATADIR_CFG_SECTION, project))
else:
scrapy_cfg = closest_scrapy_cfg()
if not scrapy_cfg:
raise NotConfigured(
"Unable to find scrapy.cfg file to infer project data dir"
)
d = (Path(scrapy_cfg).parent / ".scrapy").resolve()
if not d.exists():
d.mkdir(parents=True)
return str(d)
def data_path(path: str | os.PathLike[str], createdir: bool = False) -> str:
"""
Return the given path joined with the .scrapy data directory.
If given an absolute path, return it unmodified.
"""
path_obj = Path(path)
if not path_obj.is_absolute():
if inside_project():
path_obj = Path(project_data_dir(), path)
else:
path_obj = Path(".scrapy", path)
if createdir and not path_obj.exists():
path_obj.mkdir(parents=True)
return str(path_obj)
[docs]
def get_project_settings() -> Settings:
"""Return a :class:`~scrapy.settings.Settings` object with the
:ref:`project settings <project-settings>` of the current project.
Settings from sources with a higher precedence, such as :ref:`spider
settings <spider-settings>`, are applied when a crawl starts.
"""
init_env(os.environ.get("SCRAPY_PROJECT", "default"))
settings = Settings()
settings_module_path = os.environ.get(ENVVAR)
if settings_module_path:
settings.setmodule(settings_module_path, priority="project")
valid_envvars = {
"CHECK",
"PROJECT",
"PYTHON_SHELL",
"SETTINGS_MODULE",
}
scrapy_envvars = {
k[7:]: v
for k, v in os.environ.items()
if k.startswith("SCRAPY_") and k.replace("SCRAPY_", "") in valid_envvars
}
settings.setdict(scrapy_envvars, priority="project")
return settings