old eshop web scrape

This commit is contained in:
Priec
2026-07-29 12:19:02 +02:00
parent aac0a95b52
commit e24ad0de1b
215 changed files with 10671 additions and 0 deletions

472
data_scrape/scrape.py Normal file
View File

@@ -0,0 +1,472 @@
#!/usr/bin/env python3
"""Scrape products + category tree from http://e-shop.kompress.sk (PrestaShop 1.5/1.6).
Outputs (next to this script):
categories.json - full category tree (flat list with parent_id + path)
products.json - one entry per product, with category memberships
images/<id>/ - original (highest quality) product images
"""
import html
import json
import os
import re
import sys
import time
import urllib.request
import urllib.error
from concurrent.futures import ThreadPoolExecutor
from html.parser import HTMLParser
BASE = "http://e-shop.kompress.sk"
OUT = os.path.dirname(os.path.abspath(__file__))
IMG_DIR = os.path.join(OUT, "images")
UA = "Mozilla/5.0 (X11; Linux x86_64) data-migration-scraper"
DELAY = 0.3
# ---------------------------------------------------------------- fetching
def fetch(url, binary=False, retries=3):
for attempt in range(retries):
try:
req = urllib.request.Request(url, headers={"User-Agent": UA})
with urllib.request.urlopen(req, timeout=45) as r:
data = r.read()
return data if binary else data.decode("utf-8", "replace")
except urllib.error.HTTPError as e:
if e.code == 404:
return None
if attempt == retries - 1:
print(" !! %s -> HTTP %s" % (url, e.code), file=sys.stderr)
return None
except Exception as e: # noqa: BLE001 - network flakiness
if attempt == retries - 1:
print(" !! %s -> %s" % (url, e), file=sys.stderr)
return None
time.sleep(1 + attempt)
return None
# ---------------------------------------------------------------- helpers
TAG_RE = re.compile(r"<[^>]+>")
WS_RE = re.compile(r"[ \t\r\f\v]+")
def unesc(s):
return html.unescape(s or "").strip()
def strip_tags(h):
"""HTML -> readable plain text, keeping block/list structure as newlines."""
if not h:
return ""
t = re.sub(r"(?i)<\s*br\s*/?>", "\n", h)
t = re.sub(r"(?i)</\s*(p|div|li|tr|h[1-6]|ul|ol)\s*>", "\n", t)
t = re.sub(r"(?i)<\s*li[^>]*>", "", t)
t = TAG_RE.sub("", t)
t = html.unescape(t)
t = WS_RE.sub(" ", t)
t = re.sub(r" *\n *", "\n", t)
t = re.sub(r"\n{3,}", "\n\n", t)
return t.strip()
def find(pattern, text, group=1, flags=re.S):
m = re.search(pattern, text, flags)
return m.group(group) if m else None
def cat_id_from_url(url):
m = re.match(r"^%s/(\d+)-" % re.escape(BASE), url)
return int(m.group(1)) if m else None
def product_id_from_url(url):
m = re.search(r"/(\d+)-[^/]*\.html", url)
return int(m.group(1)) if m else None
# ---------------------------------------------------------------- categories
class TreeParser(HTMLParser):
"""Parses the nested <ul class="tree"> category block into (id, name, title, url, parent)."""
def __init__(self):
super().__init__(convert_charrefs=False)
self.in_tree = 0
self.depth = 0
self.stack = [] # category id per <ul> level
self.cur = None
self.rows = []
def handle_starttag(self, tag, attrs):
a = dict(attrs)
if tag == "ul":
if self.in_tree:
self.depth += 1
elif "tree" in (a.get("class") or ""):
self.in_tree = 1
self.depth = 1
elif tag == "a" and self.in_tree and a.get("href", "").startswith(BASE):
cid = cat_id_from_url(a["href"])
if cid is None:
return
parent = self.stack[self.depth - 2] if self.depth >= 2 else None
self.cur = {
"id": cid,
"name": "",
"title": unesc(a.get("title", "")),
"url": a["href"],
"parent_id": parent,
}
while len(self.stack) < self.depth:
self.stack.append(None)
self.stack[self.depth - 1] = cid
self.rows.append(self.cur)
def handle_endtag(self, tag):
if tag == "ul" and self.in_tree:
self.depth -= 1
if self.depth <= 0:
self.in_tree = 0
elif tag == "a":
self.cur = None
def handle_data(self, data):
if self.cur is not None:
self.cur["name"] += data
def handle_entityref(self, name):
if self.cur is not None:
self.cur["name"] += "&%s;" % name
def handle_charref(self, name):
if self.cur is not None:
self.cur["name"] += "&#%s;" % name
def scrape_categories():
home = fetch(BASE + "/")
p = TreeParser()
p.feed(home)
cats = {}
for row in p.rows:
row["name"] = unesc(row["name"])
cats[row["id"]] = row
# enrich each category from its own page: h1, description, product count
def enrich(c):
page = fetch(c["url"])
time.sleep(DELAY)
if not page:
return
h1 = find(r'<h1[^>]*>(.*?)</h1>', page)
if h1:
c["name"] = strip_tags(h1)
desc = find(r'<div[^>]*class="[^"]*cat_desc[^"]*"[^>]*>\s*(.*?)\s*</div>\s*</div>', page)
if desc is not None and desc.startswith("<div>") is False:
desc = find(r'<div[^>]*class="[^"]*cat_desc[^"]*"[^>]*>\s*(.*?)\s*</div>', page)
if desc:
desc = re.sub(r"^<div>\s*|\s*</div>$", "", desc.strip())
c["description_html"] = (desc or "").strip()
c["description"] = strip_tags(desc)
# category thumbnail (original upload, if any)
img = find(r'<img[^>]+src="[^"]*/c/%d-\w+_default/[^"]*"' % c["id"], page, 0)
c["image"] = None
if img or ('/c/%d-' % c["id"]) in page:
data = fetch("%s/img/c/%d.jpg" % (BASE, c["id"]), binary=True)
if data:
os.makedirs(os.path.join(IMG_DIR, "categories"), exist_ok=True)
rel = "images/categories/%d.jpg" % c["id"]
with open(os.path.join(OUT, rel), "wb") as f:
f.write(data)
c["image"] = {"url": "%s/img/c/%d.jpg" % (BASE, c["id"]), "file": rel}
c["product_urls"] = list_category_products(c["url"], page)
c["product_ids"] = [product_id_from_url(u) for u in c["product_urls"]]
with ThreadPoolExecutor(max_workers=4) as ex:
list(ex.map(enrich, cats.values()))
# path / depth
def path_of(cid):
parts, seen = [], set()
while cid and cid in cats and cid not in seen:
seen.add(cid)
parts.append(cats[cid]["name"])
cid = cats[cid]["parent_id"]
return list(reversed(parts))
for c in cats.values():
c["path"] = path_of(c["id"])
c["depth"] = len(c["path"]) - 1
return cats
def list_category_products(url, first_page=None):
"""All product URLs in a category, following pagination."""
urls, seen_pages, page_no = [], set(), 1
page = first_page if first_page is not None else fetch(url)
while page:
for m in re.finditer(r'<a[^>]+class="product_name"[^>]+href="([^"]+)"', page):
u = unesc(m.group(1))
if u not in urls:
urls.append(u)
if not urls: # theme variation: any product link inside the list
block = find(r'<ul id="product_list".*?</ul>', page, 0)
for m in re.finditer(r'href="(%s/[^"]+\.html)"' % re.escape(BASE), block or ""):
u = unesc(m.group(1))
if u not in urls:
urls.append(u)
page_no += 1
nxt = "%s?p=%d" % (url, page_no)
if page_no > 40 or ('p=%d' % page_no) not in page or nxt in seen_pages:
break
seen_pages.add(nxt)
time.sleep(DELAY)
page = fetch(nxt)
return urls
# ---------------------------------------------------------------- products
def original_image_url(image_id):
"""PrestaShop stores the untouched upload at /img/p/<digits split by />/<id>.jpg."""
s = str(image_id)
return "%s/img/p/%s/%s.jpg" % (BASE, "/".join(s), s)
ADD_COMBINATION_RE = re.compile(
r"addCombination\(\s*(\d+)\s*,\s*new Array\(([^)]*)\)\s*,\s*(-?\d+)\s*,"
r"\s*(-?[\d.]+)\s*,\s*(-?[\d.]+)\s*,\s*(-?\d+)\s*,\s*'([^']*)'\s*,"
r"\s*(-?[\d.]+)\s*,\s*(\d+)\s*,")
def parse_variants(page, base_price_excl, tax_rate):
"""Attribute groups + combinations (PrestaShop's addCombination() calls).
addCombination(id_product_attribute, [id_attribute...], quantity, price_impact,
ecotax, id_image, reference, unit_price, minimal_quantity, ...)
where price_impact is tax-EXCLUDED, per themes/default/js/product.js.
"""
# id_attribute -> group slug, from the attributesCombinations JS array
attr_group = {}
for blk in re.findall(r"tabInfos\['id_attribute'\].*?attributesCombinations\.push", page, re.S):
aid = find(r"tabInfos\['id_attribute'\] = '(\d+)'", blk)
if aid:
attr_group[int(aid)] = find(r"tabInfos\['group'\] = '([^']*)'", blk)
# human labels from the #attributes widgets (<select> on most products,
# radio buttons on a few)
groups, labels = [], {}
attrs_html = find(r'<div id="attributes">(.*?)\n\s*</div>\s*<p id="product_reference"', page) or \
find(r'<div id="attributes">(.*)</div>', page) or ""
for fs in re.findall(r'<fieldset class="attribute_fieldset">(.*?)</fieldset>', attrs_html, re.S):
gname = strip_tags(find(r'<label[^>]*class="attribute_label"[^>]*>(.*?)</label>', fs) or "")
gname = gname.rstrip(": ").strip()
gid = find(r'name="group_(\d+)"', fs)
opts = []
for m in re.finditer(r'<option value="(\d+)"([^>]*)>(.*?)</option>', fs, re.S):
aid, rest, txt = int(m.group(1)), m.group(2), strip_tags(m.group(3))
labels[aid] = txt
opts.append({"id": aid, "label": txt, "default": "selected" in rest})
for m in re.finditer(
r'<input type="radio"[^>]*value="(\d+)"([^>]*)>\s*<span>(.*?)</span>', fs, re.S):
aid, rest, txt = int(m.group(1)), m.group(2), strip_tags(m.group(3))
labels[aid] = txt
opts.append({"id": aid, "label": txt, "default": "checked" in rest})
groups.append({"id": int(gid) if gid else None, "name": gname,
"slug": attr_group.get(opts[0]["id"]) if opts else None,
"attributes": opts})
variants = []
for m in ADD_COMBINATION_RE.finditer(page):
ids = [int(x) for x in re.findall(r"'(\d+)'", m.group(2))]
impact = float(m.group(4))
variants.append({
"id": int(m.group(1)), # id_product_attribute
"attribute_ids": ids,
"label": " / ".join(labels.get(i, str(i)) for i in ids),
"attributes": [{"group": attr_group.get(i), "label": labels.get(i, "")} for i in ids],
"reference": m.group(7),
"quantity_available": int(m.group(3)),
"price_impact_tax_excluded": impact,
"price_tax_excluded": round(base_price_excl + impact, 6) if base_price_excl is not None else None,
"price": round((base_price_excl + impact) * (1 + tax_rate / 100.0), 2)
if base_price_excl is not None else None,
"minimal_quantity": int(m.group(9)),
"image_id": int(m.group(6)) if int(m.group(6)) > 0 else None,
})
default_ids = {a["id"] for g in groups for a in g["attributes"] if a["default"]}
for v in variants:
v["is_default"] = bool(default_ids) and set(v["attribute_ids"]) == default_ids
return groups, variants
def parse_product(url, page):
pid = product_id_from_url(url) or int(find(r"var id_product = '(\d+)'", page) or 0)
name = strip_tags(find(r'<div id="pb-left-column">\s*<h1[^>]*>(.*?)</h1>', page)
or find(r'<h1[^>]*>(.*?)</h1>', page) or "")
short_html = find(r'id="short_description_content"[^>]*>(.*?)</div>\s*(?:<p class="buttons_bottom_block"|</div>)', page)
desc_html = find(r'<div id="idTab1"[^>]*>(.*?)</div>\s*(?:<!--|<div id="idTab)', page)
# breadcrumb -> category trail (last node is the product name, not a link)
crumb = find(r'<div class="breadcrumb">(.*?)</div>', page) or ""
trail = []
for m in re.finditer(r'<a[^>]+href="(%s/(\d+)-[^"]*)"[^>]*>(.*?)</a>' % re.escape(BASE), crumb):
trail.append({"id": int(m.group(2)), "name": strip_tags(m.group(3)), "url": m.group(1)})
# images: thumbnails list, else the single main image
img_ids = [int(i) for i in re.findall(r'<li id="thumbnail_(\d+)"', page)]
if not img_ids:
main = find(r'id="bigpic"[^>]*src="[^"]*?/(\d+)-large_default/', page) or \
find(r'src="[^"]*?/(\d+)-large_default/', page)
if main:
img_ids = [int(main)]
default_img = find(r"var idDefaultImage = (\d+);", page)
qty = find(r"var quantityAvailable = (-?\d+);", page)
price_excl = find(r"var productPriceTaxExcluded = ([\d.]+)", page)
tax = float(find(r"var taxRate = ([\d.]+);", page) or 0)
groups, variants = parse_variants(
page, float(price_excl) if price_excl else None, tax)
return {
"id": pid,
"name": name,
"url": url,
"reference": unesc(find(r'<p id="product_reference"[^>]*>.*?<span[^>]*>(.*?)</span>', page) or
find(r"var productReference = '([^']*)'", page) or ""),
"price": float(find(r"var productPrice = '([\d.]+)'", page) or 0),
"price_display": strip_tags(find(r'id="our_price_display"[^>]*>(.*?)</span>', page) or ""),
"price_tax_excluded": round(float(price_excl), 6) if price_excl else None,
"price_without_reduction": float(find(r"var productPriceWithoutReduction = '([\d.]+)'", page) or 0),
"tax_rate": tax,
"currency": "EUR",
"quantity_available": int(qty) if qty is not None else None,
"available_for_order": find(r"var productAvailableForOrder = '(\d)'", page) == "1",
"has_attributes": find(r"var productHasAttributes = (\w+);", page) == "true",
"attribute_groups": groups,
"variants": variants,
"short_description_html": (short_html or "").strip(),
"short_description": strip_tags(short_html),
"description_html": (desc_html or "").strip(),
"description": strip_tags(desc_html),
"breadcrumb": trail,
"category_path": [c["name"] for c in trail],
"default_image_id": int(default_img) if default_img else None,
"images": [{"id": i,
"url": original_image_url(i),
"page_url": url,
"file": "images/%d/%d.jpg" % (pid, i)} for i in img_ids],
}
def download_images(prod):
d = os.path.join(IMG_DIR, str(prod["id"]))
os.makedirs(d, exist_ok=True)
for img in prod["images"]:
path = os.path.join(OUT, img["file"])
if os.path.exists(path) and os.path.getsize(path) > 0:
img["bytes"] = os.path.getsize(path)
continue
data = fetch(img["url"], binary=True)
if data is None: # fall back to the largest generated thumbnail
alt = "%s/%d-thickbox_default/x.jpg" % (BASE, img["id"])
data = fetch(alt, binary=True)
if data:
img["url"] = alt
if data:
with open(path, "wb") as f:
f.write(data)
img["bytes"] = len(data)
else:
img["bytes"] = 0
time.sleep(DELAY)
# ---------------------------------------------------------------- main
def main():
os.makedirs(IMG_DIR, exist_ok=True)
print("Scraping category tree ...")
cats = scrape_categories()
print(" %d categories" % len(cats))
# product URLs: category listings + sitemap (catches anything unlisted)
urls = []
for c in cats.values():
for u in c["product_urls"]:
if u not in urls:
urls.append(u)
sm = fetch(BASE + "/sitemap.xml") or ""
for m in re.finditer(r'(%s/[^\]\s]+\.html)' % re.escape(BASE), sm):
if m.group(1) not in urls:
urls.append(m.group(1))
print(" %d product URLs" % len(urls))
products, failed = [], []
def work(u):
page = fetch(u)
time.sleep(DELAY)
if not page or 'id="product_page_product_id"' not in page:
failed.append(u)
return None
p = parse_product(u, page)
download_images(p)
return p
with ThreadPoolExecutor(max_workers=4) as ex:
for i, p in enumerate(ex.map(work, urls), 1):
if p:
products.append(p)
if i % 20 == 0:
print(" %d/%d" % (i, len(urls)))
# a product can be served under an old and a new slug; keep the first
# (category-listing) URL, which is the canonical one
seen_ids, unique = set(), []
for p in products:
if p["id"] in seen_ids:
continue
seen_ids.add(p["id"])
unique.append(p)
products = sorted(unique, key=lambda p: p["id"])
# attach every category a product is listed in (a product can be in several)
by_id = {p["id"]: p for p in products}
for p in products:
p["categories"] = []
for c in sorted(cats.values(), key=lambda c: c["id"]):
for pid in c["product_ids"]:
if pid in by_id:
by_id[pid]["categories"].append(
{"id": c["id"], "name": c["name"], "path": c["path"]})
cat_list = sorted(cats.values(), key=lambda c: (c["depth"], c["id"]))
for c in cat_list:
c["product_count"] = len(c["product_ids"])
c.pop("product_urls", None)
with open(os.path.join(OUT, "categories.json"), "w", encoding="utf-8") as f:
json.dump(cat_list, f, ensure_ascii=False, indent=2)
with open(os.path.join(OUT, "products.json"), "w", encoding="utf-8") as f:
json.dump(products, f, ensure_ascii=False, indent=2)
imgs = sum(len(p["images"]) for p in products)
print("\nDone: %d products, %d categories, %d images" % (len(products), len(cat_list), imgs))
if failed:
print("Failed URLs (%d):" % len(failed))
for u in failed:
print(" " + u)
if __name__ == "__main__":
main()