browsertrix/backend/btrixcloud/operator/cronjobs.py
Tessa Walsh 9140dd75bc
Add and enforce readOnly field in Organization (#1886)
Fixes https://github.com/webrecorder/browsertrix/issues/1883
Backend work for https://github.com/webrecorder/browsertrix/issues/1876

- If readOnly is set true, disallow crawls and QA analysis runs
- If readOnly is set to true, skip scheduled crawls
- Add endpoint to set `readOnly` with optional `readOnlyReason` (which
is automatically set back to an empty string when `readOnly` is being
set to false), which can be displayed in banner
- Operator: ensures cronjobs that are skipped due to internal logic (eg. readonly mode) simply succeed right away and do not leave a k8s job dangling.

---------
Co-authored-by: Ilya Kreymer <ikreymer@users.noreply.github.com>
Co-authored-by: Ilya Kreymer <ikreymer@gmail.com>
2024-06-25 19:30:53 -07:00

155 lines
5.1 KiB
Python

""" Operator handler for crawl CronJobs """
from uuid import UUID
from typing import Optional
import yaml
from btrixcloud.utils import to_k8s_date, dt_now
from .models import MCBaseRequest, MCDecoratorSyncData, CJS, CMAP
from .baseoperator import BaseOperator
# pylint: disable=too-many-locals
# ============================================================================
class CronJobOperator(BaseOperator):
"""CronJob Operator"""
def init_routes(self, app):
"""init routes for crawl CronJob decorator"""
@app.post("/op/cronjob/sync")
async def mc_sync_cronjob_crawls(data: MCDecoratorSyncData):
return await self.sync_cronjob_crawl(data)
@app.post("/op/cronjob/customize")
async def mc_cronjob_related(data: MCBaseRequest):
return self.get_cronjob_crawl_related(data)
def get_cronjob_crawl_related(self, data: MCBaseRequest):
"""return configmap related to crawl"""
labels = data.parent.get("metadata", {}).get("labels", {})
cid = labels.get("btrix.crawlconfig")
return {
"relatedResources": [
{
"apiVersion": "v1",
"resource": "configmaps",
"labelSelector": {"matchLabels": {"btrix.crawlconfig": cid}},
}
]
}
def get_finished_response(
self, metadata: dict[str, str], set_status=True, finished: Optional[str] = None
):
"""get final response to indicate cronjob created job is finished"""
if not finished:
finished = to_k8s_date(dt_now())
status = None
# set status on decorated job to indicate that its finished
if set_status:
status = {
"succeeded": 1,
"startTime": metadata.get("creationTimestamp"),
"completionTime": finished,
}
return {
"attachments": [],
# set on job to match default behavior when job finishes
"annotations": {"finished": finished},
"status": status,
}
async def sync_cronjob_crawl(self, data: MCDecoratorSyncData):
"""create crawljobs from a job object spawned by cronjob"""
metadata = data.object["metadata"]
labels = metadata.get("labels", {})
cid = labels.get("btrix.crawlconfig")
name = metadata.get("name")
crawl_id = name
actual_state, finished = await self.crawl_ops.get_crawl_state(
crawl_id, is_qa=False
)
if finished:
finished_str = to_k8s_date(finished)
set_status = False
# mark job as completed
if not data.object["status"].get("succeeded"):
print("Cron Job Complete!", finished)
set_status = True
return self.get_finished_response(metadata, set_status, finished_str)
configmap = data.related[CMAP][f"crawl-config-{cid}"]["data"]
oid = configmap.get("ORG_ID")
userid = configmap.get("USER_ID")
crawljobs = data.attachments[CJS]
org = await self.org_ops.get_org_by_id(UUID(oid))
warc_prefix = None
if not actual_state:
# cronjob doesn't exist yet
crawlconfig = await self.crawl_config_ops.get_crawl_config(
UUID(cid), UUID(oid)
)
if not crawlconfig:
print(
f"error: no crawlconfig {cid}. skipping scheduled job. old cronjob left over?"
)
return self.get_finished_response(metadata)
# db create
user = await self.user_ops.get_by_id(UUID(userid))
if not user:
print(f"error: missing user for id {userid}")
return self.get_finished_response(metadata)
warc_prefix = self.crawl_config_ops.get_warc_prefix(org, crawlconfig)
if org.readOnly:
print(
f"org {org.id} set to read-only. skipping scheduled crawl for workflow {cid}"
)
return self.get_finished_response(metadata)
await self.crawl_config_ops.add_new_crawl(
crawl_id,
crawlconfig,
user,
manual=False,
)
print("Scheduled Crawl Created: " + crawl_id)
crawl_id, crawljob = self.k8s.new_crawl_job_yaml(
cid,
userid=userid,
oid=oid,
storage=org.storage,
crawler_channel=configmap.get("CRAWLER_CHANNEL", "default"),
scale=int(configmap.get("INITIAL_SCALE", 1)),
crawl_timeout=int(configmap.get("CRAWL_TIMEOUT", 0)),
max_crawl_size=int(configmap.get("MAX_CRAWL_SIZE", "0")),
manual=False,
crawl_id=crawl_id,
warc_prefix=warc_prefix,
)
attachments = list(yaml.safe_load_all(crawljob))
if crawl_id in crawljobs:
attachments[0]["status"] = crawljobs[CJS][crawl_id]["status"]
return {
"attachments": attachments,
}