archivebox.crawls.models

Module Contents

Classes

CrawlSchedule

Crawl

API

class archivebox.crawls.models.CrawlSchedule[source]

Bases: archivebox.base_models.models.ModelWithUUID, archivebox.base_models.models.ModelWithNotes

id[source]

‘CompactUUIDField(…)’

created_at[source]

‘DateTimeField(…)’

created_by[source]

‘ForeignKey(…)’

modified_at[source]

‘DateTimeField(…)’

template: Crawl[source]

‘ForeignKey(…)’

schedule[source]

‘CharField(…)’

is_enabled[source]

‘BooleanField(…)’

config[source]

‘JSONField(…)’

label[source]

‘CharField(…)’

notes[source]

‘TextField(…)’

crawl_set: django.db.models.Manager[Crawl][source]

None

class Meta[source]

Bases: archivebox.base_models.models.ModelWithUUID.Meta, archivebox.base_models.models.ModelWithNotes.Meta

app_label[source]

‘crawls’

verbose_name[source]

‘Scheduled Crawl’

verbose_name_plural[source]

‘Scheduled Crawls’

__str__() → str[source]
property api_url: str[source]
save(*args, **kwargs)[source]
property last_run_at[source]
property next_run_at[source]
is_due(now=None) → bool[source]
property kind: str[source]
dispatch(queued_at=None) → Crawl | None[source]

Run maintenance directly or enqueue one ordinary Crawl.

enqueue(queued_at=None) → archivebox.crawls.models.Crawl[source]
class archivebox.crawls.models.Crawl[source]

Bases: archivebox.base_models.models.ModelWithDeleteAfter, archivebox.base_models.models.ModelWithOutputDir, archivebox.base_models.models.ModelWithConfig, archivebox.base_models.models.ModelWithHealthStats, archivebox.workers.models.ModelWithQueue

objects[source]

‘as_manager(…)’

id[source]

‘CompactUUIDField(…)’

created_at[source]

‘DateTimeField(…)’

created_by[source]

‘ForeignKey(…)’

modified_at[source]

‘DateTimeField(…)’

urls[source]

‘TextField(…)’

config[source]

‘JSONField(…)’

permissions[source]

‘GeneratedField(…)’

max_depth[source]

‘PositiveSmallIntegerField(…)’

tags_str[source]

‘CharField(…)’

persona[source]

‘ForeignKey(…)’

label[source]

‘CharField(…)’

notes[source]

‘TextField(…)’

schedule[source]

‘ForeignKey(…)’

status[source]

‘StatusField(…)’

retry_at[source]

‘RetryAtField(…)’

retry_at_field_name[source]

‘retry_at’

state_field_name[source]

‘status’

StatusChoices[source]

None

INITIAL_STATE[source]

None

ACTIVE_STATE[source]

None

FINAL_STATES[source]

()

FINAL_OR_ACTIVE_STATES[source]

()

active_state[source]

None

delete_after_final_statuses[source]

()

RUNNABLE_STATES[source]

()

INACTIVE_STATES[source]

()

schedule_id: uuid.UUID | None[source]

None

snapshot_set: django.db.models.Manager[archivebox.core.models.Snapshot][source]

None

class Meta[source]

Bases: archivebox.base_models.models.ModelWithDeleteAfter.Meta, archivebox.base_models.models.ModelWithOutputDir.Meta, archivebox.base_models.models.ModelWithConfig.Meta, archivebox.base_models.models.ModelWithHealthStats.Meta, archivebox.workers.models.ModelWithQueue.Meta

app_label[source]

‘crawls’

verbose_name[source]

‘Crawl’

verbose_name_plural[source]

‘Crawls’

indexes[source]

None

__str__()[source]
get_delete_after_config_value()[source]
pause(*, save: bool = True) → bool[source]
resume(*, when=None, save: bool = True) → bool[source]
cancel() → None[source]
schedule_child_snapshots_for_sealing() → int[source]
schedule_child_snapshots_for_pause() → int[source]
classmethod missing_delete_at_candidates()[source]
save(*args, **kwargs)[source]
update_child_snapshot_permissions(old_permissions: str | None, new_permissions: str | None) → int[source]
property api_url: str[source]
static parse_tag_names(tags: collections.abc.Iterable[str] | str, *, pattern: str = ',') → list[str][source]
current_tag_names() → list[str][source]
apply_snapshot_tag_diff(*, added_tag_names: collections.abc.Iterable[str], removed_tag_names: collections.abc.Iterable[str]) → None[source]
to_json() → dict[source]

Convert Crawl model instance to a JSON-serializable dict.

static from_json(record: dict, overrides: dict | None = None)[source]

Create or get a Crawl from a JSON dict.

Args: record: Dict with ‘urls’ (required), optional ‘max_depth’, ‘tags_str’, ‘label’ overrides: Dict of field overrides (e.g., created_by_id)

Returns: Crawl instance or None if invalid

property output_dir: pathlib.Path[source]
get_urls_list() → list[str][source]

Get list of URLs from urls field, filtering out comments and empty lines.

static normalize_domain(value: str) → str[source]
static split_filter_patterns(value) → list[str][source]
classmethod _pattern_matches_url(url: str, pattern: str) → bool[source]
get_current_config(*, refresh: bool = False) → dict[str, Any][source]
get_url_allowlist(*, use_effective_config: bool = False, snapshot=None) → list[str][source]
get_url_denylist(*, use_effective_config: bool = False, snapshot=None) → list[str][source]
url_passes_filters(url: str, *, snapshot=None, use_effective_config: bool = True) → bool[source]
url_passes_compiled_filters(url: str, *, allowlist: list[str], denylist: list[str]) → bool[source]
set_url_filters(allowlist, denylist) → None[source]
apply_crawl_config_filters() → dict[str, int][source]
_iter_url_lines() → list[tuple[str, str]][source]
count_urls_for_limit() → int[source]

Count unique URLs already queued or snapshotted for this crawl.

max_urls is a crawl-wide cap on snapshots, so direct URL entries and recursively discovered snapshots both have to consume the same budget.

remaining_url_capacity() → int | None[source]
has_remaining_url_capacity() → bool[source]
remaining_snapshot_capacity() → int | None[source]
has_remaining_snapshot_capacity() → bool[source]
prune_urls(predicate) → list[str][source]
prune_url(url: str) → int[source]
exclude_domain(domain: str) → dict[str, int | str | bool][source]
resolve_persona()[source]
static _config_value(config: collections.abc.Mapping[str, Any] | Any, key: str, default: Any = None) → Any[source]
classmethod create_scheduler_row(**kwargs) → archivebox.crawls.models.Crawl[source]
limit_stop_reason(*, config: collections.abc.Mapping[str, Any] | Any | None = None, output_dir: pathlib.Path | None = None, num_snapshots: int | None = None) → str[source]
lifecycle_stop_reason(*, num_snapshots: int | None = None, num_sealed_snapshots: int | None = None) → str[source]
stop_reason(*, config: collections.abc.Mapping[str, Any] | Any | None = None, output_dir: pathlib.Path | None = None, num_snapshots: int | None = None, num_sealed_snapshots: int | None = None) → str[source]
add_url(entry: dict) → bool[source]

Add a URL to the crawl queue if not already present.

Args: entry: dict with ‘url’, optional ‘depth’, ‘title’, ‘timestamp’, ‘tags’, ‘via_snapshot’, ‘plugin’

Returns: True if URL was added, False if skipped (duplicate or depth exceeded)

create_snapshots_from_urls() → list[archivebox.core.models.Snapshot][source]

Create Snapshot objects for each URL in self.urls that doesn’t already exist.

Returns: List of newly created Snapshot objects

create_discovered_snapshot(parent_snapshot, *, url: str, depth: int, title: str = '', tags: str = '', created_by_id: int | None = None)[source]

Create one child snapshot if it passes crawl filters and limits.

create_discovered_snapshots(parent_snapshot, records: collections.abc.Iterable[collections.abc.Mapping[str, Any]], *, depth: int, created_by_id: int | None = None) → list[archivebox.core.models.Snapshot][source]

Create child snapshots from discovered URL records after filtering and deduping once.

is_finished() → bool[source]

Check if crawl is finished (all snapshots sealed or no snapshots exist).

can_start() → bool[source]
has_finished_snapshots() → bool[source]
mark_started() → bool[source]
seal() → bool[source]

Finalize a runner-owned Crawl without dispatching hooks directly.

advance_lifecycle() → bool[source]

Advance one explicit lifecycle step after the runner claims this row.

cleanup_runtime() → None[source]

Remove runner-owned runtime artifacts after abx-dl cleanup hooks finish.