# scraper.py import os from typing import List, Dict, Any, Optional, Tuple, Set from datetime import datetime from sqlalchemy.orm import Session from sqlalchemy import select from database import InstagramAccount, InstagramPost, InstaCursor from utils.boxapi_client import get_media_page, BoxAPIError from utils.parser_fast import parse_boxapi_items_fast from utils.s3_uploader import ( upload_media_from_url, upload_image_from_url_optimized, upload_video_from_url_optimized, ) from schemas import ScrapeRequest, ScrapedPost DIAG = os.getenv("SCRAPER_DIAGNOSTICS", "0") == "1" MAX_PAGES = int(os.getenv("SCRAPER_MAX_PAGES", "400")) # safety cap PER_REQ = 12 # BoxAPI cap def dprint(msg: str) -> None: if DIAG: print(f"[SCRAPER] {msg}") # ------------------------- cursor helpers (persisted) ------------------------- def _get_cursor(db: Session, username: str) -> Optional[str]: row = db.execute( select(InstaCursor).where(InstaCursor.username == username) ).scalars().first() return row.next_max_id if row else None def _set_cursor(db: Session, username: str, next_max_id: Optional[str]) -> None: if next_max_id is None: return row = db.execute( select(InstaCursor).where(InstaCursor.username == username) ).scalars().first() if row: row.next_max_id = next_max_id else: row = InstaCursor(username=username, next_max_id=next_max_id) db.add(row) db.commit() # ------------------------------ small utilities ------------------------------ def _resolve_target(req: ScrapeRequest) -> int: # accept both "count" and "max_count" return int((getattr(req, "count", None) if getattr(req, "count", None) is not None else req.max_count)) def _normalize_page(page: List[Dict[str, Any]]) -> List[Dict[str, Any]]: if not page: return [] return parse_boxapi_items_fast(page) def _db_existing_ids(db: Session, ig_ids: List[str]) -> Set[str]: if not ig_ids: return set() rows = db.query(InstagramPost.ig_post_id).filter(InstagramPost.ig_post_id.in_(ig_ids)).all() return {r[0] for r in rows} def _get_or_create_account(db: Session, username: str, seller_id) -> InstagramAccount: acc = db.query(InstagramAccount).filter_by(username=username).first() if acc: return acc acc = InstagramAccount(username=username, seller_id=seller_id, is_active=True, last_synced=None) db.add(acc) db.commit() db.refresh(acc) return acc # --------------------------------- MAIN -------------------------------------- def scrape_and_store(db: Session, req: ScrapeRequest) -> List[ScrapedPost]: username = req.username target = _resolve_target(req) seller_id = str(req.seller_id) dprint(f"requested count={getattr(req,'count',None)} max_count={req.max_count} → target={target}") # Optional time filter since_ts: Optional[int] = None if hasattr(req, "since_taken_at") and req.since_taken_at is not None: try: since_ts = int(req.since_taken_at) except Exception: since_ts = None # Account (needed for saving) account = _get_or_create_account(db, username, req.seller_id) # Cursor: start where we left off (older pages) max_id: Optional[str] = _get_cursor(db, username) dprint(f"start_cursor={max_id!r}") # Page through BoxAPI until we collect enough NEW items (older than what we have) pages = 0 final_cursor: Optional[str] = None seen_cursors: Set[Optional[str]] = set() if max_id is not None: seen_cursors.add(max_id) # guard against cycles normalized_new: List[Dict[str, Any]] = [] try: while len(normalized_new) < target and pages < MAX_PAGES: pages += 1 page, _log, nxt = get_media_page(username=username, count=PER_REQ, max_id=max_id) dprint(f"[PAGE]#{pages}: request max_id={max_id!r} → got {len(page)} items, next_max_id={nxt!r}") if not page: dprint("No items → stopping.") break norm = _normalize_page(page) # optional time filter if since_ts is not None: try: s = int(since_ts) norm = [n for n in norm if (n.get("taken_at") or 0) > s] except Exception: pass ids = [str(n.get("ig_post_id")) for n in norm if n.get("ig_post_id")] already = _db_existing_ids(db, ids) new_here = [n for n in norm if n.get("ig_post_id") and str(n["ig_post_id"]) not in already] normalized_new.extend(new_here) dprint( f" normalized={len(norm)} new_here={len(new_here)} " f"new_total={len(normalized_new)}/{target}" ) if len(normalized_new) >= target: final_cursor = nxt # where to resume next time (older side) break if not nxt: dprint("No next_max_id → end of feed.") final_cursor = None break if nxt in seen_cursors: dprint("Repeating cursor detected → stopping.") final_cursor = nxt break seen_cursors.add(nxt) final_cursor = nxt max_id = nxt # go older except BoxAPIError as e: print(f"❌ BoxAPI error: {e}") # continue with whatever we gathered # Persist cursor only if we actually moved if pages > 0 and final_cursor: _set_cursor(db, username, final_cursor) if not normalized_new: dprint("No NEW posts found.") return [] # Save up to `target` new posts (upload media, insert rows) result: List[ScrapedPost] = [] for item in normalized_new[:target]: post_id = str(item["ig_post_id"]) # Race guard: if another worker already saved it, return from DB so we still reach N existing = db.query(InstagramPost).filter_by(ig_post_id=post_id).first() if existing: result.append(ScrapedPost( ig_post_id=existing.ig_post_id, media_type=existing.media_type, caption=existing.caption, thumbnail_url=existing.thumbnail_url, local_media=existing.local_media or [], created_at=existing.created_at, )) if len(result) >= target: break continue # Upload media media_urls = list(item.get("remote_urls", [])) if item["media_type"] == "video": thumb = (item.get("thumbnail_url") or "").strip() def is_video(u: str) -> bool: base = u.split("?", 1)[0].lower() return base.endswith((".mp4", ".mov", ".m4v", ".webm")) media_urls = [u for u in media_urls if is_video(u) and u != thumb] uploaded_urls: List[str] = [] for remote_url in media_urls: try: if item["media_type"] == "video": url = upload_video_from_url_optimized(remote_url, seller_id, post_id) elif item["media_type"] in ("image", "carousel"): url = upload_image_from_url_optimized(remote_url, seller_id, post_id) if not url: url = upload_media_from_url(remote_url, seller_id, post_id) else: url = upload_media_from_url(remote_url, seller_id, post_id) if url: uploaded_urls.append(url) except Exception as e: print(f"⚠️ Error uploading media: {e}") continue # Thumbnail final_thumbnail = item.get("thumbnail_url") or None if final_thumbnail: try: thumb_url = upload_image_from_url_optimized(final_thumbnail, seller_id, post_id) if thumb_url: final_thumbnail = thumb_url else: alt = upload_media_from_url(final_thumbnail, seller_id, post_id) if alt: final_thumbnail = alt except Exception as e: print(f"⚠️ Failed to upload thumbnail: {e}") created_at = datetime.fromtimestamp(item["taken_at"]) if item.get("taken_at") else datetime.utcnow() row = InstagramPost( ig_post_id=post_id, account_id=account.id, media_type=item["media_type"], caption=item.get("caption"), thumbnail_url=final_thumbnail, local_media=uploaded_urls, created_at=created_at, ) db.add(row) db.commit() result.append(ScrapedPost( ig_post_id=post_id, media_type=item["media_type"], caption=item.get("caption"), thumbnail_url=final_thumbnail, local_media=uploaded_urls, created_at=created_at, )) if len(result) >= target: break dprint(f"DONE username={username} returned={len(result)} requested={target} pages={pages} last_cursor={final_cursor!r}") return result