Skip to content
Merged
10 changes: 5 additions & 5 deletions backend/btrixcloud/colls.py
Original file line number Diff line number Diff line change
Expand Up @@ -442,7 +442,7 @@ async def get_collection_out(
thumbnail = result.get("thumbnail")
if thumbnail:
image_file = UserFile(**thumbnail)
result["thumbnail"] = await image_file.get_file_out(
result["thumbnail"], _ = await image_file.get_file_out(
org, self.storage_ops, headers
)

Expand Down Expand Up @@ -490,7 +490,7 @@ async def get_public_collection_out(
thumbnail = result.get("thumbnail")
if thumbnail:
image_file = UserFile(**thumbnail)
result["thumbnail"] = await image_file.get_public_file_out(
result["thumbnail"], _ = await image_file.get_public_file_out(
org, self.storage_ops, headers
)

Expand All @@ -513,7 +513,7 @@ async def get_public_thumbnail(
raise HTTPException(status_code=404, detail="thumbnail_not_found")

image_file = UserFile(**thumbnail)
image_file_out = await image_file.get_public_file_out(
image_file_out, _ = await image_file.get_public_file_out(
org, self.storage_ops, headers
)

Expand Down Expand Up @@ -630,11 +630,11 @@ async def list_collections(
image_file = UserFile(**thumbnail)

if public_colls_out:
res["thumbnail"] = await image_file.get_public_file_out(
res["thumbnail"], _ = await image_file.get_public_file_out(
org, self.storage_ops, headers
)
else:
res["thumbnail"] = await image_file.get_file_out(
res["thumbnail"], _ = await image_file.get_file_out(
org, self.storage_ops, headers
)

Expand Down
10 changes: 3 additions & 7 deletions backend/btrixcloud/crawlconfigs.py
Original file line number Diff line number Diff line change
Expand Up @@ -1379,7 +1379,6 @@ async def run_now_internal(
crawlconfig.crawlFilenameTemplate or self.default_filename_template
)

seed_file_url = ""
if crawlconfig.config.seedFileId:
crawler_image = self.get_channel_crawler_image(crawlconfig.crawlerChannel)
if (
Expand All @@ -1393,11 +1392,6 @@ async def run_now_internal(
status_code=400, detail="seed_file_not_supported_by_crawler"
)

seed_file_out = await self.file_ops.get_seed_file_out(
crawlconfig.config.seedFileId, org
)
seed_file_url = seed_file_out.path

try:
crawl_id = await self.crawl_manager.create_crawl_job(
crawlconfig,
Expand All @@ -1408,7 +1402,9 @@ async def run_now_internal(
profile_filename=profile_filename or "",
profileid=str(crawlconfig.profileid) if crawlconfig.profileid else "",
is_single_page=self.is_single_page(crawlconfig.config),
seed_file_url=seed_file_url,
seed_file_id=str(crawlconfig.config.seedFileId)
if crawlconfig.config.seedFileId
else "",
)
await self.add_new_crawl(crawl_id, crawlconfig, user, org, manual=True)
return crawl_id
Expand Down
4 changes: 2 additions & 2 deletions backend/btrixcloud/crawlmanager.py
Original file line number Diff line number Diff line change
Expand Up @@ -423,7 +423,7 @@ async def create_crawl_job(
profile_filename: str,
profileid: str,
is_single_page: bool,
seed_file_url: str,
seed_file_id: str,
) -> str:
"""create new crawl job from config"""
cid = str(crawlconfig.id)
Expand Down Expand Up @@ -453,7 +453,7 @@ async def create_crawl_job(
str(crawlconfig.dedupeCollId) if crawlconfig.dedupeCollId else ""
),
is_single_page=is_single_page,
seed_file_url=seed_file_url,
seed_file_id=seed_file_id,
)

async def reload_running_crawl_config(self, crawl_id: str):
Expand Down
18 changes: 12 additions & 6 deletions backend/btrixcloud/file_uploads.py
Original file line number Diff line number Diff line change
Expand Up @@ -3,7 +3,7 @@
import os
import tempfile
from collections.abc import AsyncGenerator, Callable
from datetime import timedelta
from datetime import datetime, timedelta
from typing import TYPE_CHECKING, Any
from uuid import UUID, uuid4

Expand Down Expand Up @@ -110,10 +110,13 @@ async def get_seed_file_out(
org: Organization | None = None,
type_: str | None = None,
headers: dict | None = None,
) -> SeedFileOut:
force_update_presigned: bool = False,
) -> tuple[SeedFileOut, datetime]:
"""Get file output model by UUID"""
user_file = await self.get_seed_file(file_id, org, type_)
return await user_file.get_file_out(org, self.storage_ops, headers)
return await user_file.get_file_out(
org, self.storage_ops, headers, force_update_presigned
)

async def list_seed_files(
self,
Expand Down Expand Up @@ -178,7 +181,7 @@ async def list_seed_files(
user_files = []
for res in items:
file_ = SeedFile.from_dict(res)
file_out = await file_.get_file_out(org, self.storage_ops, headers)
file_out, _ = await file_.get_file_out(org, self.storage_ops, headers)
user_files.append(file_out)

return user_files, total
Expand Down Expand Up @@ -311,7 +314,7 @@ async def _parse_seed_info_from_file(
first_seed = ""
seed_count = 0

file_url = await file_obj.get_absolute_presigned_url(
file_url, _ = await file_obj.get_absolute_presigned_url(
org, self.storage_ops, None
)

Expand Down Expand Up @@ -527,7 +530,10 @@ async def list_seed_files(
async def get_seed_file(
file_id: UUID, request: Request, org: Organization = Depends(org_viewer_dep)
):
return await ops.get_seed_file_out(file_id, org, headers=dict(request.headers))
seed_file, _ = await ops.get_seed_file_out(
file_id, org, headers=dict(request.headers)
)
return seed_file

@router.delete("/{file_id}", response_model=SuccessResponse)
async def delete_user_file(
Expand Down
8 changes: 4 additions & 4 deletions backend/btrixcloud/k8sapi.py
Original file line number Diff line number Diff line change
Expand Up @@ -123,7 +123,7 @@ def new_crawl_job_yaml(
proxy_id: str = "",
dedupe_coll_id: str = "",
is_single_page: bool = False,
seed_file_url: str = "",
seed_file_id: str = "",
):
"""load job template from yaml"""
if not crawl_id:
Expand Down Expand Up @@ -151,7 +151,7 @@ def new_crawl_job_yaml(
"proxy_id": proxy_id,
"dedupe_coll_id": dedupe_coll_id,
"is_single_page": "1" if is_single_page else "0",
"seed_file_url": seed_file_url,
"seed_file_id": seed_file_id,
}

data = self.templates.env.get_template("crawl_job.yaml").render(params)
Expand All @@ -178,7 +178,7 @@ async def new_crawl_job(
proxy_id: str = "",
dedupe_coll_id: str = "",
is_single_page: bool = False,
seed_file_url: str = "",
seed_file_id: str = "",
) -> str:
"""load and init crawl job via k8s api"""
crawl_id, data = self.new_crawl_job_yaml(
Expand All @@ -201,7 +201,7 @@ async def new_crawl_job(
proxy_id=proxy_id,
dedupe_coll_id=dedupe_coll_id,
is_single_page=is_single_page,
seed_file_url=seed_file_url,
seed_file_id=seed_file_id,
)

# create job directly
Expand Down
51 changes: 36 additions & 15 deletions backend/btrixcloud/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -1390,39 +1390,53 @@ class UserFile(BaseFile):
created: datetime

async def get_absolute_presigned_url(
self, org, storage_ops, headers: dict | None
) -> str:
self, org, storage_ops, headers: dict | None, force_update: bool = False
) -> tuple[str, datetime]:
"""Get presigned URL as absolute URL"""
presigned_url, _ = await storage_ops.get_presigned_url(org, self)
return storage_ops.resolve_relative_access_path(presigned_url, headers) or ""
presigned_url, expire_at = await storage_ops.get_presigned_url(
org, self, force_update
)
return storage_ops.resolve_relative_access_path(
presigned_url, headers
) or "", expire_at

async def get_file_out(
self, org, storage_ops, headers: dict | None = None
) -> UserFileOut:
self,
org,
storage_ops,
headers: dict | None = None,
force_update_presigned: bool = False,
) -> tuple[UserFileOut, datetime]:
"""Get UserFileOut with new presigned url"""
path, expire_at = await self.get_absolute_presigned_url(
org, storage_ops, headers, force_update_presigned
)
return UserFileOut(
name=self.filename,
path=await self.get_absolute_presigned_url(org, storage_ops, headers),
path=path,
hash=self.hash,
size=self.size,
originalFilename=self.originalFilename,
mime=self.mime,
userid=self.userid,
userName=self.userName,
created=self.created,
)
), expire_at

async def get_public_file_out(
self, org, storage_ops, headers: dict | None = None
) -> PublicUserFileOut:
) -> tuple[PublicUserFileOut, datetime]:
"""Get PublicUserFileOut with new presigned url"""
path, expire_at = await self.get_absolute_presigned_url(
org, storage_ops, headers
)
return PublicUserFileOut(
name=self.filename,
path=await self.get_absolute_presigned_url(org, storage_ops, headers),
path=path,
hash=self.hash,
size=self.size,
mime=self.mime,
)
), expire_at


# ============================================================================
Expand Down Expand Up @@ -1492,12 +1506,19 @@ class SeedFile(UserFile, BaseMongoModel):
seedCount: int | None = None

async def get_file_out(
self, org, storage_ops, headers: dict | None = None
) -> SeedFileOut:
self,
org,
storage_ops,
headers: dict | None = None,
force_update_presigned: bool = False,
) -> tuple[SeedFileOut, datetime]:
"""Get SeedFileOut with new presigned url"""
path, expire_at = await self.get_absolute_presigned_url(
org, storage_ops, headers, force_update_presigned
)
return SeedFileOut(
name=self.filename,
path=await self.get_absolute_presigned_url(org, storage_ops, headers),
path=path,
hash=self.hash,
size=self.size,
originalFilename=self.originalFilename,
Expand All @@ -1510,7 +1531,7 @@ async def get_file_out(
type=self.type,
firstSeed=self.firstSeed,
seedCount=self.seedCount,
)
), expire_at


# ============================================================================
Expand Down
40 changes: 35 additions & 5 deletions backend/btrixcloud/operator/crawls.py
Original file line number Diff line number Diff line change
Expand Up @@ -230,7 +230,7 @@ async def sync_crawls(self, data: MCSyncData):
scheduled=spec.get("manual") != "1",
qa_source_crawl_id=spec.get("qaSourceCrawlId"),
is_single_page=spec.get("isSinglePage") == "1",
seed_file_url=spec.get("seedFileUrl", ""),
seed_file_id=spec.get("seedFileId", ""),
)

# if finalizing, crawl is being deleted
Expand Down Expand Up @@ -479,11 +479,25 @@ async def sync_crawls(self, data: MCSyncData):
config_update_needed = (
spec.get("lastConfigUpdate", "") != status.lastConfigUpdate
)

# Update config to refresh seed file presigned url if it has
# already expired or will within the next few minutes
seed_file_presigned_update_needed = bool(crawl.seed_file_id) and (
status.seed_file_presigned_expiry is None
or (status.seed_file_presigned_expiry <= (dt_now() + timedelta(minutes=5)))
)
if seed_file_presigned_update_needed:
logger.debug(
"seed_file_presigned_url_update_needed",
expire_at=status.seed_file_presigned_expiry,
)
config_update_needed = config_update_needed or seed_file_presigned_update_needed

status.lastConfigUpdate = spec.get("lastConfigUpdate", "")

children.extend(
await self._load_crawl_configmap(
crawl, data.children, params, config_update_needed
crawl, data.children, params, status, config_update_needed
)
)

Expand Down Expand Up @@ -571,7 +585,12 @@ def _filter_autoclick_behavior(
return behaviors

async def _load_crawl_configmap(
self, crawl: CrawlSpec, children, params, config_update_needed: bool
self,
crawl: CrawlSpec,
children,
params,
status: CrawlStatus,
config_update_needed: bool,
):
name = f"crawl-config-{crawl.id}"

Expand Down Expand Up @@ -600,8 +619,19 @@ async def _load_crawl_configmap(
raw_config["behaviors"], crawler_image
)

if crawl.seed_file_url:
raw_config["seedFile"] = crawl.seed_file_url
if crawl.seed_file_id:
seed_file_out, expire_at = await self.file_ops.get_seed_file_out(
UUID(crawl.seed_file_id), crawl.org, force_update_presigned=True
)
status.seed_file_presigned_expiry = expire_at
logger.debug(
"seed_file_presigned_url_generated",
crawl_id=crawl.id,
seed_file_url=seed_file_out.path,
seed_file_id=crawl.seed_file_id,
expire_at=expire_at,
)
raw_config["seedFile"] = seed_file_out.path
raw_config.pop("seedFileId", None)

params["config"] = json.dumps(raw_config)
Expand Down
12 changes: 3 additions & 9 deletions backend/btrixcloud/operator/cronjobs.py
Original file line number Diff line number Diff line change
Expand Up @@ -199,14 +199,6 @@ async def make_new_crawljob(
else:
profile_filename = ""

if crawlconfig.config.seedFileId:
seed_file = await self.file_ops.get_seed_file_out(
crawlconfig.config.seedFileId, org
)
seed_file_url = seed_file.path
else:
seed_file_url = ""

crawl_id, crawljob = self.k8s.new_crawl_job_yaml(
cid=str(cid),
userid=str(userid),
Expand All @@ -228,7 +220,9 @@ async def make_new_crawljob(
str(crawlconfig.dedupeCollId) if crawlconfig.dedupeCollId else ""
),
is_single_page=self.crawl_config_ops.is_single_page(crawlconfig.config),
seed_file_url=seed_file_url,
seed_file_id=str(crawlconfig.config.seedFileId)
if crawlconfig.config.seedFileId
else "",
)

return MCDecoratorSyncResponse(attachments=list(yaml.safe_load_all(crawljob)))
Expand Down
5 changes: 4 additions & 1 deletion backend/btrixcloud/operator/models.py
Original file line number Diff line number Diff line change
Expand Up @@ -101,7 +101,7 @@ class CrawlSpec(BaseModel):
profileid: str | None = None
dedupe_coll_id: str | None = None
is_single_page: bool = False
seed_file_url: str | None = ""
seed_file_id: str | None = ""

@property
def db_crawl_id(self) -> str:
Expand Down Expand Up @@ -290,3 +290,6 @@ class CrawlStatus(BaseModel):

# last state
last_state: TYPE_ALL_CRAWL_STATES = Field(default="starting", exclude=True)

# expiry time for seed file presigned url
seed_file_presigned_expiry: datetime | None = None
2 changes: 1 addition & 1 deletion chart/app-templates/crawl_job.yaml
Original file line number Diff line number Diff line change
Expand Up @@ -48,4 +48,4 @@ spec:

isSinglePage: "{{ is_single_page }}"

seedFileUrl: "{{ seed_file_url }}"
seedFileId: "{{ seed_file_id }}"
Loading