Files
2026-08-09 16:51:01 -05:00

37 lines
1.6 KiB
C++

#pragma once
#include <string>
#include <vector>
#include <optional>
// Fetches an LZT Market search-result page and extracts individual offer
// "cards" as plain text, mirroring the heuristic the original Python script
// used inside the browser (`document.querySelectorAll` + innerText filtering
// for blocks that mention "followers").
//
// IMPORTANT: the python version drove a real (headless) Chrome via Selenium,
// which meant it saw the fully JS-rendered page and re-used a logged-in
// browser profile. This C++ version fetches the raw HTML over HTTP instead
// (no JS execution). LZT Market's search results are server-rendered, so
// this works for public listings; if you need an authenticated view, set
// `cookie_header` in config.json to your browser's `Cookie:` header value.
class Scraper {
public:
explicit Scraper(std::string cookie_header = "", std::string user_agent = "");
// Returns cleaned, deduplicated offer text blocks for the given URL.
std::vector<std::string> checkOffers(const std::string& url);
// Client-side post-filter equivalent to python's filter_offers().
static std::vector<std::string> filterOffers(const std::vector<std::string>& offers,
std::optional<double> min_price,
std::optional<double> max_price,
const std::vector<std::string>& keywords);
private:
std::string cookie_header_;
std::string user_agent_;
static std::string cleanText(std::string text);
static std::vector<std::string> extractOfferBlocks(const std::string& html);
};