goon/app/connectors/direct_scrapers/youperv.py
goon-foss 5a0b62c3e4 feat(youperv): browse scraper + direct-mp4 extractor
youperv.com (DataLife Engine) carries paysite rips titled
"Studio - Performer - Title" (71% of a 132-title sample), ~60-70 new
scenes/day. Browse the homepage + /page/N/ (19 scenes/page, no overlap);
scene URL is /<category>/<id>-<slug>.html.

Cast is scoped to the fmeta block (up to the Related section): the whole
page carries 19-26 /xfsearch/pornstar/ links but only 1-2 are the actual
cast, so without scoping this would repeat the page-wide pollution that
got xxxfiles rejected. Studio comes from the title prefix, guarded so a
performer name is never mistaken for a studio. Duration, ISO release date,
per-scene tags and thumbnail all come from the same block.

Playback is a plain <source> mp4 (files.klubnichka-hd.com) with no token
or expiry, but the CDN hotlink-guards on Referer: bare Range gets 403,
Range + Referer + browser UA gets 206 cross-IP from the VPS. So the
extractor returns it with referer + mobile_direct_ok and the phone streams
straight from the CDN, no WebView and no proxy. Path is percent-encoded
because the filenames contain spaces.

Deep-crawl capped at 2000 pages: beyond that (<=09.2023) the catalog turns
into generic amateur uploads with no performers and dead CDN files.

Verified: 19 scenes/page with studio+cast+duration+date, max 2 performers
per scene (pollution guard holds), 0/19 missing duration, playback 206
video/mp4. Pilot ingest 3 pages: 57 seen, 39 attached to existing canonical
scenes, 18 new, 0 errors.

Co-Authored-By: Claude Opus 4.5 <noreply@anthropic.com>
2026-07-26 17:57:24 +02:00

173 lines
7 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""youperv.com — browse scraper (DataLife Engine). Dodany 2026-07-26.
Ripy paysite ze studiami (Brazzers Exxtra, Blacked, Evil Angel, Deeper, Private…),
tytuły w formacie `Studio - Performer - Title` (71% próbki 132 tytułów), świeże
(~60-70 scen/dzień). Orphan-risk LOW: nazwane studio + nazwany performer + data
co do sekundy + duration = mocny sygnał do canonical match.
Listing: homepage (newest) + `/page/N/`, 19 scen/stronę, zero overlapu między
stronami. Scene URL: `/<kategoria>/<id>-<slug>.html`.
**KRYTYCZNE — scoping obsady**: performerzy MUSZĄ być czytani tylko z bloku
`<div class="fmeta">` … `Related`. Na całej stronie jest 19-26 linków
`xfsearch/pornstar/` (blok Related), w samym fmeta 1-2 realnych. Bez scopingu
powtórzylibyśmy błąd, przez który odrzuciliśmy xxxfiles (page-wide pollution
zaśmiecająca bazę performerów).
Tytuł zostaje z prefiksem studia (jak hdporngg/porn00) — token_set_ratio i tak
złapie canonical, a prefiks niesie dodatkowy sygnał.
Playback: direct mp4 `<source>` na files.klubnichka-hd.com, BEZ tokena/expiry, ale
CDN ma hotlink-guard na Referer (bez nagłówka 403, z nagłówkiem 206 cross-IP).
Rozwiązuje extractor `youpervcom` (VPS-side, mobile gra direct, zero WebView/proxy).
Głębokość: deep-crawl capowany (`_PAGE_CAP` w deep_crawl.py) — strony ~2100+ to
stara amatorka bez performerów, z martwymi linkami (HTTP 500 na CDN).
"""
from __future__ import annotations
import html
import re
from urllib.parse import unquote
from app.connectors.base import (
RawPerformer,
RawPlaybackSource,
RawScene,
RawStudio,
RawTag,
)
from app.connectors.direct_scrapers._browse_base import BaseBrowseScraper, meta_content
from app.connectors.direct_scrapers._playtube import _parse_iso_date
from app.normalize.text import slugify
_BASE = "https://youperv.com"
_SCENE_URL_RE = re.compile(
r'href="(https://youperv\.com/[a-z0-9\-]+/\d+-[^"]+\.html)"', re.IGNORECASE
)
_H1_RE = re.compile(r'<h1[^>]*class="items-title[^"]*"[^>]*>(.*?)</h1>', re.IGNORECASE | re.DOTALL)
_PERF_RE = re.compile(r'xfsearch/pornstar/([^/"]+)', re.IGNORECASE)
_CAT_XF_RE = re.compile(r'xfsearch/cat/([^/"]+)', re.IGNORECASE)
_TAG_LINK_RE = re.compile(r'<a[^>]+href="[^"]+"[^>]*>([^<]{2,40})</a>', re.IGNORECASE)
_DUR_RE = re.compile(r"fa-clock-o[^>]*></i>\s*(\d{1,2}):(\d{2})(?::(\d{2}))?", re.IGNORECASE)
_DATE_RE = re.compile(r'"datePublished"\s*:\s*"([^"]+)"')
# Sufiks h1: `… Title 07.26.2026 <span class="xd"> HD</span>`
_H1_DATE_SUFFIX_RE = re.compile(r"\s*\d{2}\.\d{2}\.\d{4}\s*$")
def _clean_title(raw_h1: str) -> str:
text = re.sub(r"<[^>]+>", " ", raw_h1) # <span class="xd"> HD</span> itp.
text = html.unescape(re.sub(r"\s+", " ", text)).strip()
text = re.sub(r"\bHD\b\s*$", "", text).strip()
return _H1_DATE_SUFFIX_RE.sub("", text).strip()
def _perf_name(raw_slug: str) -> str:
"""`carolina%20guerrero` → `Carolina Guerrero`."""
name = unquote(raw_slug).replace("-", " ").strip()
return " ".join(w.capitalize() if w.islower() else w for w in name.split())
class YoupervScraper(BaseBrowseScraper):
sitetag = "youpervcom"
def _listing_url(self, page: int) -> str:
return f"{_BASE}/" if page <= 1 else f"{_BASE}/page/{page}/"
def _extract_scene_urls(self, listing_html: str) -> list[str]:
seen: set[str] = set()
out: list[str] = []
for m in _SCENE_URL_RE.finditer(listing_html):
url = m.group(1)
if url not in seen: # każdy link jest 2× w karcie (thumb + tytuł)
seen.add(url)
out.append(url)
return out
def _parse_detail(self, scene_url: str, detail_html: str) -> RawScene | None:
h1 = _H1_RE.search(detail_html)
title = _clean_title(h1.group(1)) if h1 else ""
if not title:
og = meta_content(detail_html, property="og:title") or ""
title = og.split(" » ")[0].strip()
if not title:
return None
# Blok metadanych TEJ sceny: od `class="fmeta` do sekcji Related (dalej idą
# linki powiązanych scen → performer pollution, patrz docstring).
i = detail_html.find('class="fmeta')
j = detail_html.find("Related", i + 1) if i >= 0 else -1
fmeta = detail_html[i:j] if i >= 0 and j > i else ""
performers: list[RawPerformer] = []
seen_p: set[str] = set()
for m in _PERF_RE.finditer(fmeta):
name = _perf_name(m.group(1))
sl = slugify(name)
if not sl or sl in seen_p:
continue
seen_p.add(sl)
performers.append(
RawPerformer(external_id=f"{self.sitetag}:performer:{sl}", name=name)
)
# Studio z prefiksu `Studio - Performer - Title` (≥3 człony). Guard: prefiks
# nie może być nazwiskiem performera (wtedy to `Performer - Title`, bez studia).
studio: RawStudio | None = None
parts = [p.strip() for p in title.split(" - ")]
if len(parts) >= 3 and 2 <= len(parts[0]) <= 40:
cand = parts[0]
if slugify(cand) not in seen_p:
studio = RawStudio(
external_id=f"{self.sitetag}:studio:{slugify(cand)}",
name=cand,
slug=slugify(cand),
)
tags: list[RawTag] = []
seen_t: set[str] = set()
tag_names = [_perf_name(m.group(1)) for m in _CAT_XF_RE.finditer(fmeta)]
ti = detail_html.find("full-tags")
if ti >= 0:
block = detail_html[ti:ti + 800]
tag_names += [html.unescape(m.group(1)).strip() for m in _TAG_LINK_RE.finditer(block)]
for name in tag_names:
sl = slugify(name)
if not sl or sl in seen_t or sl in seen_p or name.lower() in ("categories", "tags"):
continue
seen_t.add(sl)
tags.append(RawTag(external_id=f"{self.sitetag}:tag:{sl}", name=name, slug=sl))
duration_sec: int | None = None
dm = _DUR_RE.search(fmeta or detail_html)
if dm:
h_or_m, mins, secs = dm.group(1), dm.group(2), dm.group(3)
duration_sec = (
int(h_or_m) * 3600 + int(mins) * 60 + int(secs)
if secs
else int(h_or_m) * 60 + int(mins)
)
rd = _DATE_RE.search(detail_html)
release_date = _parse_iso_date(rd.group(1)) if rd else None
thumbnail_url = meta_content(detail_html, property="og:image")
return RawScene(
external_id=f"{self.sitetag}:{scene_url}",
title=title,
release_date=release_date,
duration_sec=duration_sec,
url=scene_url,
studio=studio,
performers=performers,
tags=tags,
playback_sources=[
RawPlaybackSource(
origin=f"tube:{self.sitetag}",
page_url=scene_url,
duration_sec=duration_sec,
thumbnail_url=thumbnail_url,
)
],
)