Compare commits
2 Commits
Author | SHA1 | Date |
---|---|---|
Fijxu | 75e8d44c13 | |
Fijxu | f74c3c1a04 |
|
@ -0,0 +1,49 @@
|
|||
name: '4get CI'
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
pull_request:
|
||||
push:
|
||||
branches:
|
||||
- '*'
|
||||
paths-ignore:
|
||||
- 'README.md'
|
||||
- 'docker-compose.yaml'
|
||||
- '.gitignore'
|
||||
- 'docs/**'
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: docker
|
||||
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
name: Checkout 4get repository
|
||||
|
||||
- uses: docker/setup-buildx-action@v3
|
||||
name: Setup Docker BuildX system
|
||||
|
||||
- name: Login to Docker Container Registry
|
||||
uses: docker/login-action@v3
|
||||
with:
|
||||
registry: git.lolcat.ca
|
||||
username: ${{ secrets.USERNAME }}
|
||||
password: ${{ secrets.TOKEN }}
|
||||
|
||||
- name: Docker meta
|
||||
id: meta
|
||||
uses: docker/metadata-action@v5
|
||||
with:
|
||||
images: git.lolcat.ca/ckg/4get
|
||||
tags: |
|
||||
type=sha,format=short,prefix={{date 'YYYY.MM.DD'}}-,enable=${{ github.ref == format('refs/heads/{0}', 'master') }}
|
||||
type=raw,value=latest,enable=${{ github.ref == format('refs/heads/{0}', 'master') }}
|
||||
|
||||
- uses: docker/build-push-action@v6
|
||||
name: Build images
|
||||
with:
|
||||
context: .
|
||||
file: Dockerfile
|
||||
tags: ${{ steps.meta.outputs.tags }}
|
||||
platforms: linux/amd64
|
||||
push: true
|
20
README.md
20
README.md
|
@ -9,11 +9,9 @@ https://4get.ca/about
|
|||
## Official instance
|
||||
https://4get.ca , or visit the official instance list: https://4get.ca/instances
|
||||
|
||||
_NOT to be confused with 4get.ch, 4get.lol and friends! I **don't** host these._
|
||||
|
||||
## Totally unbiased comparison between alternatives
|
||||
|
||||
| | 4get | searx(ng) | libreY | araa | hearch.co |
|
||||
| | 4get | searx(ng) | libreY | araa | hearch |
|
||||
|----------------------------|-------------------------|-----------|-------------|-----------|-------------------|
|
||||
| RAM usage | 200-400mb~ | 2GB~ | 200-400mb~ | 2GB~ | idk |
|
||||
| Does it suck | no (debunked by snopes) | yes | yes | a little | better than searx |
|
||||
|
@ -25,9 +23,9 @@ _NOT to be confused with 4get.ch, 4get.lol and friends! I **don't** host these._
|
|||
3. Bot protection that *actually* filters out the bots (when configured)
|
||||
4. Interface doesn't require javascript
|
||||
5. Favicon fetcher with caching support & image proxy
|
||||
6. Bunch of other shits
|
||||
6. Bunch of other shit
|
||||
|
||||
tl;dr 4get is the best way to browse for shit.
|
||||
tl;dr the best way to actually browse for shit.
|
||||
|
||||
# Supported websites
|
||||
|
||||
|
@ -41,11 +39,11 @@ tl;dr 4get is the best way to browse for shit.
|
|||
| Qwant | Qwant | Startpage | Mojeek | | Kagi |
|
||||
| Ghostery | Yep | Qwant | | | Qwant |
|
||||
| Yep | Solofield | Solofield | | | Ghostery |
|
||||
| Greppr | Pinterest | | | | Yep |
|
||||
| Crowdview | 500px | | | | Marginalia |
|
||||
| Mwmbl | VSCO | | | | YouTube |
|
||||
| Mojeek | Imgur | | | | Soundcloud |
|
||||
| Solofield | FindThatMeme | | | | |
|
||||
| Greppr | Imgur | | | | Yep |
|
||||
| Crowdview | FindThatMeme | | | | Marginalia |
|
||||
| Mwmbl | | | | | YouTube |
|
||||
| Mojeek | | | | | Soundcloud |
|
||||
| Solofield | | | | | |
|
||||
| Marginalia | | | | | |
|
||||
| wiby | | | | | |
|
||||
| Curlie | | | | | |
|
||||
|
@ -54,7 +52,7 @@ tl;dr 4get is the best way to browse for shit.
|
|||
Refer to the <a href="https://git.lolcat.ca/lolcat/4get/src/branch/master/docs/">documentation index</a>. I recommend following the <a href="https://git.lolcat.ca/lolcat/4get/src/branch/master/docs/apache2.md">apache2 guide</a>.
|
||||
|
||||
## Contact
|
||||
Shit breaks all the time but I repair it all the time too. Email me here: <b>will (at) lolcat.ca</b> or create an issue.
|
||||
Shit breaks all the time but I repair it all the time too... Email me here: <b>will (at) lolcat.ca</b> or create an issue.
|
||||
|
||||
## License
|
||||
AGPL
|
||||
|
|
21
api.txt
21
api.txt
|
@ -1,17 +1,10 @@
|
|||
44
|
||||
4444444 44
|
||||
44444444 44444 444
|
||||
44444444 444444 444444444
|
||||
44444 44444444 444444444
|
||||
444444444 4444444
|
||||
4444444444 444444
|
||||
4444444444444
|
||||
444444444444444444
|
||||
444444444444444
|
||||
44444444
|
||||
4444
|
||||
44
|
||||
|
||||
__ __ __
|
||||
/ // / ____ ____ / /_
|
||||
/ // /_/ __ `/ _ \/ __/
|
||||
/__ __/ /_/ / __/ /_
|
||||
/_/ \__, /\___/\__/
|
||||
/____/
|
||||
|
||||
+ Welcome to the 4get API documentation +
|
||||
|
||||
+ Terms of use
|
||||
|
|
|
@ -3,34 +3,34 @@ class config{
|
|||
// Welcome to the 4get configuration file
|
||||
// When updating your instance, please make sure this file isn't missing
|
||||
// any parameters.
|
||||
|
||||
|
||||
// 4get version. Please keep this updated
|
||||
const VERSION = 8;
|
||||
|
||||
|
||||
// Will be shown pretty much everywhere.
|
||||
const SERVER_NAME = "4get";
|
||||
|
||||
const SERVER_NAME = "RETARD";
|
||||
|
||||
// Will be shown in <meta> tag on home page
|
||||
const SERVER_SHORT_DESCRIPTION = "4get is a proxy search engine that doesn't suck.";
|
||||
|
||||
|
||||
// Will be shown in server list ping (null for no description)
|
||||
const SERVER_LONG_DESCRIPTION = null;
|
||||
|
||||
|
||||
// Add your own themes in "static/themes". Set to "Dark" for default theme.
|
||||
// Eg. To use "static/themes/Cream.css", specify "Cream".
|
||||
const DEFAULT_THEME = "Dark";
|
||||
|
||||
|
||||
// Enable the API?
|
||||
const API_ENABLED = true;
|
||||
|
||||
|
||||
//
|
||||
// BOT PROTECTION
|
||||
//
|
||||
|
||||
|
||||
// 0 = disabled, 1 = ask for image captcha, @TODO: 2 = invite only (users needs a pass)
|
||||
// VERY useful against a targetted attack
|
||||
const BOT_PROTECTION = 0;
|
||||
|
||||
|
||||
// if BOT_PROTECTION is set to 1, specify the available datasets here
|
||||
// images should be named from 1.png to X.png, and be 100x100 in size
|
||||
// Eg. data/captcha/birds/1.png up to 2263.png
|
||||
|
@ -40,11 +40,11 @@ class config{
|
|||
//["fumo_plushies", 1006],
|
||||
//["minecraft", 848]
|
||||
];
|
||||
|
||||
|
||||
// If this regex expression matches on the user agent, it blocks the request
|
||||
// Not useful at all against a targetted attack
|
||||
const HEADER_REGEX = '/bot|wget|curl|python-requests|scrapy|go-http-client|ruby|yahoo|spider|qwant/i';
|
||||
|
||||
|
||||
// Block clients who present any of the following headers in their request (SPECIFY IN !!lowercase!!)
|
||||
// Eg: ["x-forwarded-for", "x-via", "forwarded-for", "via"];
|
||||
// Useful for blocking *some* proxies used for botting
|
||||
|
@ -62,7 +62,7 @@ class config{
|
|||
//"remote-addr",
|
||||
//"via"
|
||||
];
|
||||
|
||||
|
||||
// Block SSL ciphers used by CLI tools used for botting
|
||||
// Basically a primitive version of Cloudflare's browser integrity check
|
||||
// ** If curl can still access the site (with spoofed headers), please make sure you use the new apache2 config **
|
||||
|
@ -70,12 +70,12 @@ class config{
|
|||
const DISALLOWED_SSL = [
|
||||
// "TLS_AES_256_GCM_SHA384" // used by WGET and CURL
|
||||
];
|
||||
|
||||
|
||||
// Maximal number of searches per captcha key/pass issued. Counter gets
|
||||
// reset on every APCU cache clear (should happen once a day).
|
||||
// Only useful when BOT_PROTECTION is NOT set to 0
|
||||
const MAX_SEARCHES = 100;
|
||||
|
||||
|
||||
// List of domains that point to your servers. Include your tor/i2p
|
||||
// addresses here! Must be a valid URL. Won't affect links placed on
|
||||
// the homepage.
|
||||
|
@ -83,7 +83,7 @@ class config{
|
|||
//"https://4get.alt-tld",
|
||||
//"http://4getwebfrq5zr4sxugk6htxvawqehxtdgjrbcn2oslllcol2vepa23yd.onion"
|
||||
];
|
||||
|
||||
|
||||
// Known 4get instances. MUST use the https protocol if your instance uses
|
||||
// it. Is used to generate a distributed list of instances.
|
||||
// To appear in the list of an instance, contact the host and if everyone added
|
||||
|
@ -116,11 +116,11 @@ class config{
|
|||
"https://4get.sudovanilla.org",
|
||||
"https://search.mint.lgbt"
|
||||
];
|
||||
|
||||
|
||||
// Default user agent to use for scraper requests. Sometimes ignored to get specific webpages
|
||||
// Changing this might break things.
|
||||
const USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:134.0) Gecko/20100101 Firefox/134.0";
|
||||
|
||||
const USER_AGENT = "Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:131.0) Gecko/20100101 Firefox/131.0";
|
||||
|
||||
// Proxy pool assignments for each scraper
|
||||
// false = Use server's raw IP
|
||||
// string = will load a proxy list from data/proxies
|
||||
|
@ -143,8 +143,6 @@ class config{
|
|||
const PROXY_YT = false; // youtube
|
||||
const PROXY_YEP = false;
|
||||
const PROXY_PINTEREST = false;
|
||||
const PROXY_FIVEHPX = false;
|
||||
const PROXY_VSCO = false;
|
||||
const PROXY_SEZNAM = false;
|
||||
const PROXY_NAVER = false;
|
||||
const PROXY_GREPPR = false;
|
||||
|
@ -155,14 +153,14 @@ class config{
|
|||
const PROXY_YANDEX_W = false; // yandex web
|
||||
const PROXY_YANDEX_I = false; // yandex images
|
||||
const PROXY_YANDEX_V = false; // yandex videos
|
||||
|
||||
|
||||
//
|
||||
// Scraper-specific parameters
|
||||
//
|
||||
|
||||
|
||||
// GOOGLE CSE
|
||||
const GOOGLE_CX_ENDPOINT = "d4e68b99b876541f0";
|
||||
|
||||
|
||||
// MARGINALIA
|
||||
// Use "null" to default out to HTML scraping OR specify a string to
|
||||
// use the API (Eg: "public"). API has less filters.
|
||||
|
|
|
@ -75,7 +75,6 @@ class backend{
|
|||
break;
|
||||
|
||||
case "socks5_hostname":
|
||||
case "socks5h":
|
||||
case "socks5a":
|
||||
curl_setopt($curlproc, CURLOPT_PROXYTYPE, CURLPROXY_SOCKS5_HOSTNAME);
|
||||
curl_setopt($curlproc, CURLOPT_PROXY, $address . ":" . $port);
|
||||
|
|
|
@ -838,10 +838,10 @@ class frontend{
|
|||
}
|
||||
|
||||
$payload .=
|
||||
'<a href="https://webcache.googleusercontent.com/search?q=cache:' . $urlencode . '" class="list" target="_BLANK"><img src="/favicon?s=https://google.com" alt="go">Google cache</a>' .
|
||||
'<a href="https://web.archive.org/web/' . $urlencode . '" class="list" target="_BLANK"><img src="/favicon?s=https://archive.org" alt="ar">Archive.org</a>' .
|
||||
'<a href="https://archive.ph/newest/' . htmlspecialchars($link) . '" class="list" target="_BLANK"><img src="/favicon?s=https://archive.is" alt="ar">Archive.is</a>' .
|
||||
'<a href="https://ghostarchive.org/search?term=' . $urlencode . '" class="list" target="_BLANK"><img src="/favicon?s=https://ghostarchive.org" alt="gh">Ghostarchive</a>' .
|
||||
'<a href="https://arquivo.pt/wayback/' . htmlspecialchars($link) . '" class="list" target="_BLANK"><img src="/favicon?s=https://arquivo.pt" alt="ar">Arquivo.pt</a>' .
|
||||
'<a href="https://www.bing.com/search?q=url%3A' . $urlencode . '" class="list" target="_BLANK"><img src="/favicon?s=https://bing.com" alt="bi">Bing cache</a>' .
|
||||
'<a href="https://megalodon.jp/?url=' . $urlencode . '" class="list" target="_BLANK"><img src="/favicon?s=https://megalodon.jp" alt="me">Megalodon</a>' .
|
||||
'</div>';
|
||||
|
@ -969,9 +969,7 @@ class frontend{
|
|||
"qwant" => "Qwant",
|
||||
"yep" => "Yep",
|
||||
"solofield" => "Solofield",
|
||||
"pinterest" => "Pinterest",
|
||||
"fivehpx" => "500px",
|
||||
"vsco" => "VSCO",
|
||||
//"pinterest" => "Pinterest",
|
||||
"imgur" => "Imgur",
|
||||
"ftm" => "FindThatMeme"
|
||||
]
|
||||
|
|
|
@ -526,85 +526,4 @@ class fuckhtml{
|
|||
$string
|
||||
);
|
||||
}
|
||||
|
||||
public function extract_json($json){
|
||||
|
||||
$len = strlen($json);
|
||||
$array_level = 0;
|
||||
$object_level = 0;
|
||||
$in_quote = null;
|
||||
$start = null;
|
||||
|
||||
for($i=0; $i<$len; $i++){
|
||||
|
||||
switch($json[$i]){
|
||||
|
||||
case "[":
|
||||
if($in_quote === null){
|
||||
|
||||
$array_level++;
|
||||
if($start === null){
|
||||
|
||||
$start = $i;
|
||||
}
|
||||
}
|
||||
break;
|
||||
|
||||
case "]":
|
||||
if($in_quote === null){
|
||||
|
||||
$array_level--;
|
||||
}
|
||||
break;
|
||||
|
||||
case "{":
|
||||
if($in_quote === null){
|
||||
|
||||
$object_level++;
|
||||
if($start === null){
|
||||
|
||||
$start = $i;
|
||||
}
|
||||
}
|
||||
break;
|
||||
|
||||
case "}":
|
||||
if($in_quote === null){
|
||||
|
||||
$object_level--;
|
||||
}
|
||||
break;
|
||||
|
||||
case "\"":
|
||||
case "'":
|
||||
if(
|
||||
$i !== 0 &&
|
||||
$json[$i - 1] !== "\\"
|
||||
){
|
||||
// found a non-escaped quote
|
||||
|
||||
if($in_quote === null){
|
||||
|
||||
// open quote
|
||||
$in_quote = $json[$i];
|
||||
}elseif($in_quote === $json[$i]){
|
||||
|
||||
// close quote
|
||||
$in_quote = null;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
if(
|
||||
$start !== null &&
|
||||
$array_level === 0 &&
|
||||
$object_level === 0
|
||||
){
|
||||
|
||||
return substr($json, $start, $i - $start + 1);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
|
3682
scraper/ddg.php
3682
scraper/ddg.php
File diff suppressed because it is too large
Load Diff
|
@ -1,262 +0,0 @@
|
|||
<?php
|
||||
|
||||
class fivehpx{
|
||||
|
||||
public function __construct(){
|
||||
|
||||
include "lib/backend.php";
|
||||
$this->backend = new backend("fivehpx");
|
||||
|
||||
include "lib/fuckhtml.php";
|
||||
$this->fuckhtml = new fuckhtml();
|
||||
}
|
||||
|
||||
public function getfilters($page){
|
||||
|
||||
return [
|
||||
"sort" => [
|
||||
"display" => "Sort",
|
||||
"option" => [
|
||||
"relevance" => "Relevance",
|
||||
"pulse" => "Pulse",
|
||||
"newest" => "Newest"
|
||||
]
|
||||
]
|
||||
];
|
||||
}
|
||||
|
||||
private function get($proxy, $url, $get = [], $post_data = null){
|
||||
|
||||
$curlproc = curl_init();
|
||||
|
||||
if($get !== []){
|
||||
$get = http_build_query($get);
|
||||
$url .= "?" . $get;
|
||||
}
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_URL, $url);
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_ENCODING, ""); // default encoding
|
||||
|
||||
if($post_data === null){
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_HTTPHEADER,
|
||||
["User-Agent: " . config::USER_AGENT,
|
||||
"Accept: text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
|
||||
"Accept-Language: en-US,en;q=0.5",
|
||||
"Accept-Encoding: gzip",
|
||||
"DNT: 1",
|
||||
"Sec-GPC: 1",
|
||||
"Connection: keep-alive",
|
||||
"Upgrade-Insecure-Requests: 1",
|
||||
"Sec-Fetch-Dest: document",
|
||||
"Sec-Fetch-Mode: navigate",
|
||||
"Sec-Fetch-Site: same-origin",
|
||||
"Sec-Fetch-User: ?1",
|
||||
"Priority: u=0, i",
|
||||
"TE: trailers"]
|
||||
);
|
||||
}else{
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_HTTPHEADER,
|
||||
["User-Agent: " . config::USER_AGENT,
|
||||
"Accept: */*",
|
||||
"Accept-Language: en-US,en;q=0.5",
|
||||
"Accept-Encoding: gzip",
|
||||
"Referer: https://500px.com/",
|
||||
"content-type: application/json",
|
||||
//"x-csrf-token: undefined",
|
||||
"x-500px-source: Search",
|
||||
"Content-Length: " . strlen($post_data),
|
||||
"Origin: https://500px.com",
|
||||
"DNT: 1",
|
||||
"Sec-GPC: 1",
|
||||
"Connection: keep-alive",
|
||||
// "Cookie: _pin_unauth, _fbp, _sharedID, _sharedID_cst",
|
||||
"Sec-Fetch-Dest: empty",
|
||||
"Sec-Fetch-Mode: cors",
|
||||
"Sec-Fetch-Site: same-site",
|
||||
"Priority: u=4",
|
||||
"TE: trailers"]
|
||||
);
|
||||
|
||||
// set post data
|
||||
curl_setopt($curlproc, CURLOPT_POST, true);
|
||||
curl_setopt($curlproc, CURLOPT_POSTFIELDS, $post_data);
|
||||
}
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_RETURNTRANSFER, true);
|
||||
curl_setopt($curlproc, CURLOPT_SSL_VERIFYHOST, 2);
|
||||
curl_setopt($curlproc, CURLOPT_SSL_VERIFYPEER, true);
|
||||
curl_setopt($curlproc, CURLOPT_CONNECTTIMEOUT, 30);
|
||||
curl_setopt($curlproc, CURLOPT_TIMEOUT, 30);
|
||||
|
||||
// http2 bypass
|
||||
curl_setopt($curlproc, CURLOPT_HTTP_VERSION, CURL_HTTP_VERSION_2_0);
|
||||
|
||||
$this->backend->assign_proxy($curlproc, $proxy);
|
||||
|
||||
$data = curl_exec($curlproc);
|
||||
|
||||
if(curl_errno($curlproc)){
|
||||
|
||||
throw new Exception(curl_error($curlproc));
|
||||
}
|
||||
|
||||
curl_close($curlproc);
|
||||
return $data;
|
||||
}
|
||||
|
||||
public function image($get){
|
||||
|
||||
if($get["npt"]){
|
||||
|
||||
[$pagination, $proxy] =
|
||||
$this->backend->get(
|
||||
$get["npt"], "images"
|
||||
);
|
||||
|
||||
$pagination = json_decode($pagination, true);
|
||||
$search = $pagination["search"];
|
||||
|
||||
}else{
|
||||
|
||||
$search = $get["s"];
|
||||
if(strlen($search) === 0){
|
||||
|
||||
throw new Exception("Search term is empty!");
|
||||
}
|
||||
|
||||
$proxy = $this->backend->get_ip();
|
||||
$pagination = [
|
||||
"sort" => strtoupper($get["sort"]),
|
||||
"search" => $search,
|
||||
"filters" => [],
|
||||
"nlp" => false,
|
||||
];
|
||||
}
|
||||
|
||||
try{
|
||||
|
||||
$json =
|
||||
$this->get(
|
||||
$proxy,
|
||||
"https://api.500px.com/graphql",
|
||||
[],
|
||||
json_encode([
|
||||
"operationName" => "PhotoSearchPaginationContainerQuery",
|
||||
"variables" => $pagination,
|
||||
"query" =>
|
||||
'query PhotoSearchPaginationContainerQuery(' .
|
||||
(isset($pagination["cursor"]) ? '$cursor: String, ' : "") .
|
||||
'$sort: PhotoSort, $search: String!, $filters: [PhotoSearchFilter!], $nlp: Boolean) { ...PhotoSearchPaginationContainer_query_1vzAZD} fragment PhotoSearchPaginationContainer_query_1vzAZD on Query { photoSearch(sort: $sort, first: 100, ' .
|
||||
(isset($pagination["cursor"]) ? 'after: $cursor, ' : "") .
|
||||
'search: $search, filters: $filters, nlp: $nlp) { edges { node { id legacyId canonicalPath name description width height images(sizes: [33, 36]) { size url id } } } totalCount pageInfo { endCursor hasNextPage } }}'
|
||||
])
|
||||
);
|
||||
}catch(Exception $error){
|
||||
|
||||
throw new Exception("Failed to fetch graphQL object");
|
||||
}
|
||||
|
||||
$json = json_decode($json, true);
|
||||
|
||||
if($json === null){
|
||||
|
||||
throw new Exception("Failed to decode graphQL object");
|
||||
}
|
||||
|
||||
if(isset($json["errors"][0]["message"])){
|
||||
|
||||
throw new Exception("500px returned an API error: " . $json["errors"][0]["message"]);
|
||||
}
|
||||
|
||||
if(!isset($json["data"]["photoSearch"]["edges"])){
|
||||
|
||||
throw new Exception("No edges returned by API");
|
||||
}
|
||||
|
||||
$out = [
|
||||
"status" => "ok",
|
||||
"npt" => null,
|
||||
"image" => []
|
||||
];
|
||||
|
||||
foreach($json["data"]["photoSearch"]["edges"] as $image){
|
||||
|
||||
$image = $image["node"];
|
||||
$title =
|
||||
trim(
|
||||
$this->fuckhtml
|
||||
->getTextContent(
|
||||
$image["name"]
|
||||
) . ": " .
|
||||
$this->fuckhtml
|
||||
->getTextContent(
|
||||
$image["description"]
|
||||
)
|
||||
, " :"
|
||||
);
|
||||
|
||||
$small = $this->image_ratio(600, $image["width"], $image["height"]);
|
||||
$large = $this->image_ratio(2048, $image["width"], $image["height"]);
|
||||
|
||||
$out["image"][] = [
|
||||
"title" => $title,
|
||||
"source" => [
|
||||
[
|
||||
"url" => $image["images"][1]["url"],
|
||||
"width" => $large[0],
|
||||
"height" => $large[1]
|
||||
],
|
||||
[
|
||||
"url" => $image["images"][0]["url"],
|
||||
"width" => $small[0],
|
||||
"height" => $small[1]
|
||||
]
|
||||
],
|
||||
"url" => "https://500px.com" . $image["canonicalPath"]
|
||||
];
|
||||
}
|
||||
|
||||
// get NPT token
|
||||
if($json["data"]["photoSearch"]["pageInfo"]["hasNextPage"] === true){
|
||||
|
||||
$out["npt"] =
|
||||
$this->backend->store(
|
||||
json_encode([
|
||||
"cursor" => $json["data"]["photoSearch"]["pageInfo"]["endCursor"],
|
||||
"search" => $search,
|
||||
"sort" => $pagination["sort"],
|
||||
"filters" => [],
|
||||
"nlp" => false
|
||||
]),
|
||||
"images",
|
||||
$proxy
|
||||
);
|
||||
}
|
||||
|
||||
return $out;
|
||||
}
|
||||
|
||||
private function image_ratio($longest_edge, $width, $height){
|
||||
|
||||
$ratio = [
|
||||
$longest_edge / $width,
|
||||
$longest_edge / $height
|
||||
];
|
||||
|
||||
if($ratio[0] < $ratio[1]){
|
||||
|
||||
$ratio = $ratio[0];
|
||||
}else{
|
||||
|
||||
$ratio = $ratio[1];
|
||||
}
|
||||
|
||||
return [
|
||||
floor($width * $ratio),
|
||||
floor($height * $ratio)
|
||||
];
|
||||
}
|
||||
}
|
|
@ -136,7 +136,7 @@ class ftm{
|
|||
"source" => [
|
||||
[
|
||||
"url" =>
|
||||
"https://s3.thehackerblog.com/findthatmeme/" .
|
||||
"https://findthatmeme.us-southeast-1.linodeobjects.com/" .
|
||||
$thumb,
|
||||
"width" => null,
|
||||
"height" => null
|
||||
|
|
|
@ -2531,8 +2531,6 @@ class google{
|
|||
"div"
|
||||
);
|
||||
|
||||
$date = null;
|
||||
|
||||
if(count($date_div) !== 0){
|
||||
|
||||
foreach($date_div as $div){
|
||||
|
@ -2543,7 +2541,6 @@ class google{
|
|||
"bottom:"
|
||||
) !== false
|
||||
){
|
||||
|
||||
$date =
|
||||
strtotime(
|
||||
$this->fuckhtml
|
||||
|
@ -2551,6 +2548,7 @@ class google{
|
|||
$div
|
||||
)
|
||||
);
|
||||
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
@ -4149,7 +4147,7 @@ class google{
|
|||
throw new Exception("Failed to get HTML");
|
||||
}
|
||||
|
||||
//$html = file_get_contents("scraper/google.txt");
|
||||
//$html = file_get_contents("scraper/google.html");
|
||||
|
||||
return $this->parsepage($html, "web", $search, $proxy, $params);
|
||||
}
|
||||
|
|
|
@ -227,7 +227,7 @@ class marginalia{
|
|||
$json =
|
||||
$this->get(
|
||||
$this->backend->get_ip(), // no nextpage
|
||||
"https://api.marginalia-search.com/" . config::MARGINALIA_API_KEY . "/search/" . urlencode($search),
|
||||
"https://api.marginalia.nu/" . config::MARGINALIA_API_KEY . "/search/" . urlencode($search),
|
||||
[
|
||||
"count" => 20
|
||||
]
|
||||
|
@ -279,7 +279,7 @@ class marginalia{
|
|||
$html =
|
||||
$this->get(
|
||||
$proxy,
|
||||
"https://old-search.marginalia.nu/search?" . $params
|
||||
"https://search.marginalia.nu/search?" . $params
|
||||
);
|
||||
}catch(Exception $error){
|
||||
|
||||
|
@ -308,7 +308,7 @@ class marginalia{
|
|||
$html =
|
||||
$this->get(
|
||||
$proxy,
|
||||
"https://old-search.marginalia.nu/search",
|
||||
"https://search.marginalia.nu/search",
|
||||
$params
|
||||
);
|
||||
}catch(Exception $error){
|
||||
|
|
|
@ -13,104 +13,31 @@ class pinterest{
|
|||
return [];
|
||||
}
|
||||
|
||||
private function get($proxy, $url, $get = [], &$cookies, $header_data_post = null){
|
||||
private function get($proxy, $url, $get = []){
|
||||
|
||||
$curlproc = curl_init();
|
||||
|
||||
if($header_data_post === null){
|
||||
|
||||
// handling GET
|
||||
|
||||
// extract cookies
|
||||
$cookies_tmp = [];
|
||||
curl_setopt($curlproc, CURLOPT_HEADERFUNCTION, function($curlproc, $header) use (&$cookies_tmp){
|
||||
|
||||
$length = strlen($header);
|
||||
|
||||
$header = explode(":", $header, 2);
|
||||
|
||||
if(trim(strtolower($header[0])) == "set-cookie"){
|
||||
|
||||
$cookie_tmp = explode("=", trim($header[1]), 2);
|
||||
|
||||
$cookies_tmp[trim($cookie_tmp[0])] =
|
||||
explode(";", $cookie_tmp[1], 2)[0];
|
||||
}
|
||||
|
||||
return $length;
|
||||
});
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_HTTPHEADER,
|
||||
["User-Agent: " . config::USER_AGENT,
|
||||
"Accept: application/json, text/javascript, */*, q=0.01",
|
||||
"Accept-Language: en-US,en;q=0.5",
|
||||
"Accept-Encoding: gzip",
|
||||
"Referer: https://ca.pinterest.com/",
|
||||
"X-Requested-With: XMLHttpRequest",
|
||||
"X-APP-VERSION: 78f8764",
|
||||
"X-Pinterest-AppState: active",
|
||||
"X-Pinterest-Source-Url: /",
|
||||
"X-Pinterest-PWS-Handler: www/index.js",
|
||||
"screen-dpr: 1",
|
||||
"is-preload-enabled: 1",
|
||||
"DNT: 1",
|
||||
"Sec-GPC: 1",
|
||||
"Sec-Fetch-Dest: empty",
|
||||
"Sec-Fetch-Mode: cors",
|
||||
"Sec-Fetch-Site: same-origin",
|
||||
"Connection: keep-alive",
|
||||
"Alt-Used: ca.pinterest.com",
|
||||
"Priority: u=0",
|
||||
"TE: trailers"]
|
||||
);
|
||||
|
||||
if($get !== []){
|
||||
$get = http_build_query($get);
|
||||
$url .= "?" . $get;
|
||||
}
|
||||
}else{
|
||||
|
||||
// handling POST (pagination)
|
||||
if($get !== []){
|
||||
$get = http_build_query($get);
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_HTTPHEADER,
|
||||
["User-Agent: " . config::USER_AGENT,
|
||||
"Accept: application/json, text/javascript, */*, q=0.01",
|
||||
"Accept-Language: en-US,en;q=0.5",
|
||||
"Accept-Encoding: gzip",
|
||||
"Content-Type: application/x-www-form-urlencoded",
|
||||
"Content-Length: " . strlen($get),
|
||||
"Referer: https://ca.pinterest.com/",
|
||||
"X-Requested-With: XMLHttpRequest",
|
||||
"X-APP-VERSION: 78f8764",
|
||||
"X-CSRFToken: " . $cookies["csrf"],
|
||||
"X-Pinterest-AppState: active",
|
||||
"X-Pinterest-Source-Url: /search/pins/?rs=ac&len=2&q=" . urlencode($header_data_post) . "&eq=" . urlencode($header_data_post),
|
||||
"X-Pinterest-PWS-Handler: www/search/[scope].js",
|
||||
"screen-dpr: 1",
|
||||
"is-preload-enabled: 1",
|
||||
"Origin: https://ca.pinterest.com",
|
||||
"DNT: 1",
|
||||
"Sec-GPC: 1",
|
||||
"Sec-Fetch-Dest: empty",
|
||||
"Sec-Fetch-Mode: cors",
|
||||
"Sec-Fetch-Site: same-origin",
|
||||
"Connection: keep-alive",
|
||||
"Alt-Used: ca.pinterest.com",
|
||||
"Cookie: " . $cookies["cookie"],
|
||||
"TE: trailers"]
|
||||
);
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_POST, true);
|
||||
curl_setopt($curlproc, CURLOPT_POSTFIELDS, $get);
|
||||
$url .= "?" . $get;
|
||||
}
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_URL, $url);
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_ENCODING, ""); // default encoding
|
||||
|
||||
// http2 bypass
|
||||
curl_setopt($curlproc, CURLOPT_HTTP_VERSION, CURL_HTTP_VERSION_2_0);
|
||||
curl_setopt($curlproc, CURLOPT_HTTPHEADER,
|
||||
["User-Agent: Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:107.0) Gecko/20100101 Firefox/110.0",
|
||||
"Accept: text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
|
||||
"Accept-Language: en-US,en;q=0.5",
|
||||
"Accept-Encoding: gzip",
|
||||
"DNT: 1",
|
||||
"Connection: keep-alive",
|
||||
"Upgrade-Insecure-Requests: 1",
|
||||
"Sec-Fetch-Dest: document",
|
||||
"Sec-Fetch-Mode: navigate",
|
||||
"Sec-Fetch-Site: none",
|
||||
"Sec-Fetch-User: ?1"]
|
||||
);
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_RETURNTRANSFER, true);
|
||||
curl_setopt($curlproc, CURLOPT_SSL_VERIFYHOST, 2);
|
||||
|
@ -127,26 +54,6 @@ class pinterest{
|
|||
throw new Exception(curl_error($curlproc));
|
||||
}
|
||||
|
||||
if($header_data_post === null){
|
||||
|
||||
if(!isset($cookies_tmp["csrftoken"])){
|
||||
|
||||
throw new Exception("Failed to grep CSRF token");
|
||||
}
|
||||
|
||||
$cookies = "";
|
||||
|
||||
foreach($cookies_tmp as $cookie_name => $cookie_value){
|
||||
|
||||
$cookies .= $cookie_name . "=" . $cookie_value . "; ";
|
||||
}
|
||||
|
||||
$cookies = [
|
||||
"csrf" => $cookies_tmp["csrftoken"],
|
||||
"cookie" => rtrim($cookies, " ;")
|
||||
];
|
||||
}
|
||||
|
||||
curl_close($curlproc);
|
||||
return $data;
|
||||
}
|
||||
|
@ -155,68 +62,17 @@ class pinterest{
|
|||
|
||||
if($get["npt"]){
|
||||
|
||||
[$data, $proxy] =
|
||||
$this->backend->get(
|
||||
$get["npt"], "images"
|
||||
);
|
||||
|
||||
$data = json_decode($data, true);
|
||||
|
||||
$search = $data["q"];
|
||||
$cookies = $data["cookies"];
|
||||
|
||||
try{
|
||||
$json =
|
||||
$this->get(
|
||||
$proxy,
|
||||
"https://ca.pinterest.com/resource/BaseSearchResource/get/",
|
||||
// @TODO
|
||||
// post data for next page
|
||||
$data = [
|
||||
"source_url" => "/search/pins/?q=" . urlencode($search) . "&rs=typed",
|
||||
"data" =>
|
||||
json_encode(
|
||||
[
|
||||
"source_url" => "/search/pins/?q=" . urlencode($search) . "&rs=typed",
|
||||
"data" => json_encode(
|
||||
[
|
||||
"options" => [
|
||||
"applied_unified_filters" => null,
|
||||
"appliedProductFilters" => "---",
|
||||
"article" => null,
|
||||
"auto_correction_disabled" => false,
|
||||
"corpus" => null,
|
||||
"customized_rerank_type" => null,
|
||||
"domains" => null,
|
||||
"dynamicPageSizeExpGroup" => null,
|
||||
"filters" => null,
|
||||
"journey_depth" => null,
|
||||
"page_size" => null,
|
||||
"price_max" => null,
|
||||
"price_min" => null,
|
||||
"query_pin_sigs" => null,
|
||||
"query" => $data["q"],
|
||||
"redux_normalize_feed" => true,
|
||||
"request_params" => null,
|
||||
"rs" => "typed",
|
||||
"scope" => "pins",
|
||||
"selected_one_bar_modules" => null,
|
||||
"source_id" => null,
|
||||
"source_module_id" => null,
|
||||
"source_url" => "/search/pins/?q=" . urlencode($search) . "&rs=typed",
|
||||
"top_pin_id" => null,
|
||||
"top_pin_ids" => null,
|
||||
"bookmarks" => [
|
||||
$data["bookmark"]
|
||||
]
|
||||
],
|
||||
"context" => []
|
||||
],
|
||||
JSON_UNESCAPED_SLASHES
|
||||
)
|
||||
],
|
||||
$cookies,
|
||||
$search
|
||||
// {"options":{"applied_filters":null,"appliedProductFilters":"---","article":null,"auto_correction_disabled":false,"corpus":null,"customized_rerank_type":null,"domains":null,"filters":null,"journey_depth":null,"page_size":null,"price_max":null,"price_min":null,"query_pin_sigs":null,"query":"higurashi","redux_normalize_feed":true,"rs":"typed","scope":"pins","selected_one_bar_modules":null,"source_id":null,"source_module_id":null,"top_pin_id":null,"bookmarks":["Y2JVSG81V2sxcmNHRlpWM1J5VFVad1ZsWlVRbXhpVmtreVZsZHpOV0pIU2tkV2FscFhVbXhhVkZreU1WSmtNREZWVjIxR1RrMXNTbEJXYlhSaFVtMVdjMVZ1U2xaaWEzQnpXVlJPVTJWV1pISlhhM1JYVm10V05sVldVbE5XVjBwMVVXMUdWVll6VFhoVWJYaFhWMVp3Ums1V1RsTmlSbGt5Vm10YWFtVkdWbkpOU0dSUFZsZG9XRmxzWkc5VlZscHlWbGhrYkdKR1NubFdWelZQWVVaYWRHVkVRbFppUmtwVVZrUktWMlJIVWtWV2JHaHBVakZLU0Zkc1pEUmtNVnBZVW10b2FsSXdXbkJXYlRWRFpHeGFSMWRzVG1oaGVrWllXV3RvVTFVeFpFaFZiRUpoVm5wRk1GbHFSbXRYVjA1R1YyczFWMVpHV2pSWFZtaDNVakZrY2sxWVRsaGlhM0JXV1ZSR1MyRkdiRlZTYm1SVVVteHdXbGxWVlRGVk1VbDVWRmhrVjAxdVVuWlVhMXBTWlVaT2MxcEhSbE5TTWswMVdtdGFWMU5YU2paVmJYaFRUVmhDUjFZeU5YZFVNVkY0VjJ0b1ZXRnJOVlpVVmxwTFVURndXR042VmxOV2ExcGFXVlZWTlZVeFNYZE5WRTVYVWtWYVZGWkhNVTlXTVU1WllVWk9hR1ZyV2s1WFZ6QXhZakpPVjFWWWFHRlNWbkJRVm14U1IwMUdXWGxOVkVKVlRWWnNORll5TURWV1YwVjVWV3hDV21FeGNETmFSVnByVjFkS1IyTkhhR2xYUjJkM1ZtdGFhMlF4VVhsVGJGcE9Wa1p3YjFwWGVFdFZWbFp4VW14YWJGWnRVbHBaTUdoTFZHMUtTR1ZJYUZkV2VrWjJWMVphU21ReVJYcGpSbFpwVW10d1RGZHJVa0pPVms1SFZHNVNUbFl3V2xoVmJYUldaVVpaZUZremFGUk5hM0JYVkZaYVYyRkZNSGxWYkVKYVlrWlZlRnBGV210WFIwNUpVMnMxVTFaR1dscFdWekI0VFVaV1IxTllaR3BUUlhCb1dWUkdWbVZHVm5SbFJuQnNZbFpKTWxSVlVYaFBSVGxGV1hwR1QyVnJSVEZVVlZKT1RrVXhSVkpVUWs5bGJFVXhWRmhzZDFOR1ZsWmtNMFp0VWpGYWIxZFhjRXBsUlRGSVZWaHdUbFl4YTNoVVZWSnFUVVUxV0ZadGFFOVNSVnB6Vkd0a1drMUdiRFpUVkVaT1pXMWplRmRzVWxkaFJuQllWVlJTVDJWdFRqWlVNVkpTWlZad2NWcEhkRTlsYTFwMFZGVlNhMkpWTVZWVFZFcE9Wa1pzTmxkWE1WSk9WVEYwVlcweFVGWXdXVFJXUjNSWFYwZGFRbEJVTVRoUFJHTXhUbnBCTlUxRVRUUk5SRVV3VG5wUk5VMTVjRWhWVlhkeFprUlZlRTlFVVRKWlZHc3lUMWRSTWsxVVVUSk9iVnBvV1RKWmVrNTZXWGhPTWs1cFQwUkZNVTlFVm1sTlZGcHBUV3BTYTFsWFRtcE9SR015VG1wVk5GbHFaR2haVjFacldWUmFiVmxxWkdoYVZGWnFUa1JXT0ZSclZsaG1RVDA5fFVIbzVhRkpYZUc1WFYyUlpWVEpHYkdGNk1XWk5ha1ptVFZSR09FOUVZekZPZWtFMVRVUk5ORTFFUlRCT2VsRTFUWGx3U0ZWVmQzRm1SMWw1VFZSUk1WbDZUVEJhUjFGNVQxZFNhVnB0VlRGT1JFVXdXVlJuZVU1cVRUUk5hbU40VDBSSk1VNXFWVEZOYlZwcVdsUnJlRTFFVVhwWmVsVjNXbXBvYkU1dFJYbE9ha0Y2VDFSSk5VMTZWVEJaYWtJNFZHdFdXR1pCUFQwPXxOb25lfDg3NTcwOTAzODAxNDc0OTMqR1FMKnwzMjM3YjM3ZGNhMGU3YjYyYzYzYzAyZGJkNGU1MjdlNzMyMTExMTNlMmUyMzEyOWM2MDAzYmU1ZTlmZjkwYjAwfE5FV3w="]},"context":{}}
|
||||
]
|
||||
);
|
||||
|
||||
}catch(Exception $error){
|
||||
|
||||
throw new Exception("Failed to fetch JSON");
|
||||
}
|
||||
];
|
||||
|
||||
}else{
|
||||
|
||||
|
@ -225,45 +81,27 @@ class pinterest{
|
|||
|
||||
throw new Exception("Search term is empty!");
|
||||
}
|
||||
|
||||
// https://ca.pinterest.com/resource/BaseSearchResource/get/?source_url=%2Fsearch%2Fpins%2F%3Feq%3Dhigurashi%26etslf%3D5966%26len%3D2%26q%3Dhigurashi%2520when%2520they%2520cry%26rs%3Dac&data=%7B%22options%22%3A%7B%22applied_unified_filters%22%3Anull%2C%22appliedProductFilters%22%3A%22---%22%2C%22article%22%3Anull%2C%22auto_correction_disabled%22%3Afalse%2C%22corpus%22%3Anull%2C%22customized_rerank_type%22%3Anull%2C%22domains%22%3Anull%2C%22dynamicPageSizeExpGroup%22%3Anull%2C%22filters%22%3Anull%2C%22journey_depth%22%3Anull%2C%22page_size%22%3Anull%2C%22price_max%22%3Anull%2C%22price_min%22%3Anull%2C%22query_pin_sigs%22%3Anull%2C%22query%22%3A%22higurashi%20when%20they%20cry%22%2C%22redux_normalize_feed%22%3Atrue%2C%22request_params%22%3Anull%2C%22rs%22%3A%22ac%22%2C%22scope%22%3A%22pins%22%2C%22selected_one_bar_modules%22%3Anull%2C%22source_id%22%3Anull%2C%22source_module_id%22%3Anull%2C%22source_url%22%3A%22%2Fsearch%2Fpins%2F%3Feq%3Dhigurashi%26etslf%3D5966%26len%3D2%26q%3Dhigurashi%2520when%2520they%2520cry%26rs%3Dac%22%2C%22top_pin_id%22%3Anull%2C%22top_pin_ids%22%3Anull%7D%2C%22context%22%3A%7B%7D%7D&_=1736116313987
|
||||
// source_url=%2Fsearch%2Fpins%2F%3Feq%3Dhigurashi%26etslf%3D5966%26len%3D2%26q%3Dhigurashi%2520when%2520they%2520cry%26rs%3Dac
|
||||
// &data=%7B%22options%22%3A%7B%22applied_unified_filters%22%3Anull%2C%22appliedProductFilters%22%3A%22---%22%2C%22article%22%3Anull%2C%22auto_correction_disabled%22%3Afalse%2C%22corpus%22%3Anull%2C%22customized_rerank_type%22%3Anull%2C%22domains%22%3Anull%2C%22dynamicPageSizeExpGroup%22%3Anull%2C%22filters%22%3Anull%2C%22journey_depth%22%3Anull%2C%22page_size%22%3Anull%2C%22price_max%22%3Anull%2C%22price_min%22%3Anull%2C%22query_pin_sigs%22%3Anull%2C%22query%22%3A%22higurashi%20when%20they%20cry%22%2C%22redux_normalize_feed%22%3Atrue%2C%22request_params%22%3Anull%2C%22rs%22%3A%22ac%22%2C%22scope%22%3A%22pins%22%2C%22selected_one_bar_modules%22%3Anull%2C%22source_id%22%3Anull%2C%22source_module_id%22%3Anull%2C%22source_url%22%3A%22%2Fsearch%2Fpins%2F%3Feq%3Dhigurashi%26etslf%3D5966%26len%3D2%26q%3Dhigurashi%2520when%2520they%2520cry%26rs%3Dac%22%2C%22top_pin_id%22%3Anull%2C%22top_pin_ids%22%3Anull%7D%2C%22context%22%3A%7B%7D%7D
|
||||
// &_=1736116313987
|
||||
|
||||
$source_url = "/search/pins/?q=" . urlencode($search) . "&rs=" . urlencode($search);
|
||||
|
||||
$filter = [
|
||||
"source_url" => $source_url,
|
||||
"source_url" => "/search/pins/?q=" . urlencode($search),
|
||||
"rs" => "typed",
|
||||
"data" =>
|
||||
json_encode(
|
||||
[
|
||||
"options" => [
|
||||
"applied_unified_filters" => null,
|
||||
"appliedProductFilters" => "---",
|
||||
"article" => null,
|
||||
"applied_filters" => null,
|
||||
"appliedProductFilters" => "---",
|
||||
"auto_correction_disabled" => false,
|
||||
"corpus" => null,
|
||||
"customized_rerank_type" => null,
|
||||
"domains" => null,
|
||||
"dynamicPageSizeExpGroup" => null,
|
||||
"filters" => null,
|
||||
"journey_depth" => null,
|
||||
"page_size" => null,
|
||||
"price_max" => null,
|
||||
"price_min" => null,
|
||||
"query_pin_sigs" => null,
|
||||
"query" => $search,
|
||||
"query_pin_sigs" => null,
|
||||
"redux_normalize_feed" => true,
|
||||
"request_params" => null,
|
||||
"rs" => "ac",
|
||||
"rs" => "typed",
|
||||
"scope" => "pins", // pins, boards, videos,
|
||||
"selected_one_bar_modules" => null,
|
||||
"source_id" => null,
|
||||
"source_module_id" => null,
|
||||
"source_url" => $source_url,
|
||||
"top_pin_id" => null,
|
||||
"top_pin_ids" => null
|
||||
"source_id" => null
|
||||
],
|
||||
"context" => []
|
||||
]
|
||||
|
@ -272,25 +110,23 @@ class pinterest{
|
|||
];
|
||||
|
||||
$proxy = $this->backend->get_ip();
|
||||
$cookies = [];
|
||||
|
||||
try{
|
||||
$json =
|
||||
$this->get(
|
||||
$proxy,
|
||||
"https://ca.pinterest.com/resource/BaseSearchResource/get/",
|
||||
$filter,
|
||||
$cookies,
|
||||
null
|
||||
);
|
||||
|
||||
}catch(Exception $error){
|
||||
|
||||
throw new Exception("Failed to fetch JSON");
|
||||
}
|
||||
}
|
||||
|
||||
$json = json_decode($json, true);
|
||||
try{
|
||||
$json =
|
||||
json_decode(
|
||||
$this->get(
|
||||
$proxy,
|
||||
"https://www.pinterest.ca/resource/BaseSearchResource/get/",
|
||||
$filter
|
||||
),
|
||||
true
|
||||
);
|
||||
|
||||
}catch(Exception $error){
|
||||
|
||||
throw new Exception("Failed to fetch JSON");
|
||||
}
|
||||
|
||||
if($json === null){
|
||||
|
||||
|
@ -303,60 +139,6 @@ class pinterest{
|
|||
"image" => []
|
||||
];
|
||||
|
||||
if(
|
||||
!isset(
|
||||
$json["resource_response"]
|
||||
["status"]
|
||||
)
|
||||
){
|
||||
|
||||
throw new Exception("Unknown API failure");
|
||||
}
|
||||
|
||||
if($json["resource_response"]["status"] != "success"){
|
||||
|
||||
$status = "Got non-OK response: " . $json["resource_response"]["status"];
|
||||
|
||||
if(
|
||||
isset(
|
||||
$json["resource_response"]["message"]
|
||||
)
|
||||
){
|
||||
|
||||
$status .= " - " . $json["resource_response"]["message"];
|
||||
}
|
||||
|
||||
throw new Exception($status);
|
||||
}
|
||||
|
||||
if(
|
||||
isset(
|
||||
$json["resource_response"]["sensitivity"]
|
||||
["notices"][0]["description"]["text"]
|
||||
)
|
||||
){
|
||||
|
||||
throw new Exception(
|
||||
"Pinterest returned a notice: " .
|
||||
$json["resource_response"]["sensitivity"]["notices"][0]["description"]["text"]
|
||||
);
|
||||
}
|
||||
|
||||
// get NPT
|
||||
if(isset($json["resource_response"]["bookmark"])){
|
||||
|
||||
$out["npt"] =
|
||||
$this->backend->store(
|
||||
json_encode([
|
||||
"q" => $search,
|
||||
"bookmark" => $json["resource_response"]["bookmark"],
|
||||
"cookies" => $cookies
|
||||
]),
|
||||
"images",
|
||||
$proxy
|
||||
);
|
||||
}
|
||||
|
||||
foreach(
|
||||
$json
|
||||
["resource_response"]
|
||||
|
@ -368,7 +150,6 @@ class pinterest{
|
|||
switch($item["type"]){
|
||||
|
||||
case "pin":
|
||||
case "board":
|
||||
|
||||
/*
|
||||
Handle image object
|
||||
|
@ -425,15 +206,42 @@ class pinterest{
|
|||
"height" => (int)$thumb["height"]
|
||||
]
|
||||
],
|
||||
"url" =>
|
||||
$item["link"] === null ?
|
||||
"https://ca.pinterest.com/pin/" . $item["id"] :
|
||||
$item["link"]
|
||||
"url" => "https://www.pinterest.com/pin/" . $item["id"]
|
||||
];
|
||||
break;
|
||||
|
||||
case "board":
|
||||
if(isset($item["cover_pin"]["image_url"])){
|
||||
|
||||
$image = [
|
||||
"url" => $item["cover_pin"]["image_url"],
|
||||
"width" => (int)$item["cover_pin"]["size"][0],
|
||||
"height" => (int)$item["cover_pin"]["size"][1]
|
||||
];
|
||||
}elseif(isset($item["image_cover_url_hd"])){
|
||||
/*
|
||||
$image = [
|
||||
"url" =>
|
||||
"width" => null,
|
||||
"height" => null
|
||||
];*/
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
return $out;
|
||||
}
|
||||
|
||||
private function getfullresimage($image, $has_og){
|
||||
|
||||
$has_og = $has_og ? "1200x" : "originals";
|
||||
|
||||
return
|
||||
preg_replace(
|
||||
'/https:\/\/i\.pinimg\.com\/[^\/]+\//',
|
||||
"https://i.pinimg.com/" . $has_og . "/",
|
||||
$image
|
||||
);
|
||||
}
|
||||
}
|
||||
|
|
257
scraper/vsco.php
257
scraper/vsco.php
|
@ -1,257 +0,0 @@
|
|||
<?php
|
||||
|
||||
class vsco{
|
||||
|
||||
public function __construct(){
|
||||
|
||||
include "lib/backend.php";
|
||||
$this->backend = new backend("vsco");
|
||||
}
|
||||
|
||||
public function getfilters($page){
|
||||
|
||||
return [];
|
||||
}
|
||||
|
||||
private function get($proxy, $url, $get = [], $bearer = null){
|
||||
|
||||
$curlproc = curl_init();
|
||||
|
||||
if($get !== []){
|
||||
$get_tmp = http_build_query($get);
|
||||
$url .= "?" . $get_tmp;
|
||||
}
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_URL, $url);
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_ENCODING, ""); // default encoding
|
||||
|
||||
if($bearer === null){
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_HTTPHEADER,
|
||||
["User-Agent: " . config::USER_AGENT,
|
||||
"Accept: text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8",
|
||||
"Accept-Language: en-US,en;q=0.5",
|
||||
"Accept-Encoding: gzip",
|
||||
"DNT: 1",
|
||||
"Sec-GPC: 1",
|
||||
"Connection: keep-alive",
|
||||
"Upgrade-Insecure-Requests: 1",
|
||||
"Sec-Fetch-Dest: document",
|
||||
"Sec-Fetch-Mode: navigate",
|
||||
"Sec-Fetch-Site: same-origin",
|
||||
"Sec-Fetch-User: ?1",
|
||||
"Priority: u=0, i",
|
||||
"TE: trailers"]
|
||||
);
|
||||
}else{
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_HTTPHEADER,
|
||||
["User-Agent: " . config::USER_AGENT,
|
||||
"Accept: */*",
|
||||
"Accept-Language: en-US",
|
||||
"Accept-Encoding: gzip",
|
||||
"Referer: https://vsco.co/search/images/" . urlencode($get["query"]),
|
||||
"authorization: Bearer " . $bearer,
|
||||
"content-type: application/json",
|
||||
"x-client-build: 1",
|
||||
"x-client-platform: web",
|
||||
"DNT: 1",
|
||||
"Sec-GPC: 1",
|
||||
"Connection: keep-alive",
|
||||
"Sec-Fetch-Dest: empty",
|
||||
"Sec-Fetch-Mode: cors",
|
||||
"Sec-Fetch-Site: same-origin",
|
||||
"Priority: u=0",
|
||||
"TE: trailers"]
|
||||
);
|
||||
}
|
||||
|
||||
curl_setopt($curlproc, CURLOPT_RETURNTRANSFER, true);
|
||||
curl_setopt($curlproc, CURLOPT_SSL_VERIFYHOST, 2);
|
||||
curl_setopt($curlproc, CURLOPT_SSL_VERIFYPEER, true);
|
||||
curl_setopt($curlproc, CURLOPT_CONNECTTIMEOUT, 30);
|
||||
curl_setopt($curlproc, CURLOPT_TIMEOUT, 30);
|
||||
|
||||
// http2 bypass
|
||||
curl_setopt($curlproc, CURLOPT_HTTP_VERSION, CURL_HTTP_VERSION_2_0);
|
||||
|
||||
$this->backend->assign_proxy($curlproc, $proxy);
|
||||
|
||||
$data = curl_exec($curlproc);
|
||||
|
||||
if(curl_errno($curlproc)){
|
||||
|
||||
throw new Exception(curl_error($curlproc));
|
||||
}
|
||||
|
||||
curl_close($curlproc);
|
||||
return $data;
|
||||
}
|
||||
|
||||
public function image($get){
|
||||
|
||||
if($get["npt"]){
|
||||
|
||||
[$data, $proxy] =
|
||||
$this->backend->get(
|
||||
$get["npt"], "images"
|
||||
);
|
||||
|
||||
$data = json_decode($data, true);
|
||||
|
||||
}else{
|
||||
|
||||
$search = $get["s"];
|
||||
if(strlen($search) === 0){
|
||||
|
||||
throw new Exception("Search term is empty!");
|
||||
}
|
||||
|
||||
$proxy = $this->backend->get_ip();
|
||||
|
||||
// get bearer token
|
||||
try{
|
||||
|
||||
$html =
|
||||
$this->get(
|
||||
$proxy,
|
||||
"https://vsco.co/feed"
|
||||
);
|
||||
|
||||
}catch(Exception $error){
|
||||
|
||||
throw new Exception("Failed to fetch feed page");
|
||||
}
|
||||
|
||||
preg_match(
|
||||
'/"tkn":"([A-z0-9]+)"/',
|
||||
$html,
|
||||
$bearer
|
||||
);
|
||||
|
||||
if(!isset($bearer[1])){
|
||||
|
||||
throw new Exception("Failed to grep bearer token");
|
||||
}
|
||||
|
||||
$data = [
|
||||
"pagination" => [
|
||||
"query" => $search,
|
||||
"page" => 0,
|
||||
"size" => 100
|
||||
],
|
||||
"bearer" => $bearer[1]
|
||||
];
|
||||
}
|
||||
|
||||
try{
|
||||
|
||||
$json =
|
||||
$this->get(
|
||||
$proxy,
|
||||
"https://vsco.co/api/2.0/search/images",
|
||||
$data["pagination"],
|
||||
$data["bearer"]
|
||||
);
|
||||
}catch(Exception $error){
|
||||
|
||||
throw new Exception("Failed to fetch JSON");
|
||||
}
|
||||
|
||||
$json = json_decode($json, true);
|
||||
|
||||
if($json === null){
|
||||
|
||||
throw new Exception("Failed to decode JSON");
|
||||
}
|
||||
|
||||
$out = [
|
||||
"status" => "ok",
|
||||
"npt" => null,
|
||||
"image" => []
|
||||
];
|
||||
|
||||
if(!isset($json["results"])){
|
||||
|
||||
throw new Exception("Failed to access results object");
|
||||
}
|
||||
|
||||
foreach($json["results"] as $image){
|
||||
|
||||
$image_domain = parse_url("https://" . $image["responsive_url"], PHP_URL_HOST);
|
||||
$thumbnail = explode($image_domain, $image["responsive_url"], 2)[1];
|
||||
|
||||
if(substr($thumbnail, 0, 3) != "/1/"){
|
||||
|
||||
$thumbnail =
|
||||
preg_replace(
|
||||
'/^\/[^\/]+/',
|
||||
"",
|
||||
$thumbnail
|
||||
);
|
||||
}
|
||||
|
||||
$thumbnail = "https://img.vsco.co/cdn-cgi/image/width=480,height=360" . $thumbnail;
|
||||
$size =
|
||||
$this->image_ratio(
|
||||
(int)$image["dimensions"]["width"],
|
||||
(int)$image["dimensions"]["height"]
|
||||
);
|
||||
|
||||
$out["image"][] = [
|
||||
"title" => $image["description"],
|
||||
"source" => [
|
||||
[
|
||||
"url" => "https://" . $image["responsive_url"],
|
||||
"width" => (int)$image["dimensions"]["width"],
|
||||
"height" => (int)$image["dimensions"]["height"]
|
||||
],
|
||||
[
|
||||
"url" => $thumbnail,
|
||||
"width" => $size[0],
|
||||
"height" => $size[1]
|
||||
]
|
||||
],
|
||||
"url" => "https://" . $image["grid"]["domain"] . "/media/" . $image["imageId"]
|
||||
];
|
||||
}
|
||||
|
||||
// get NPT
|
||||
$max_page = ceil($json["total"] / 100);
|
||||
$data["pagination"]["page"]++;
|
||||
|
||||
if($max_page > $data["pagination"]["page"]){
|
||||
|
||||
$out["npt"] =
|
||||
$this->backend->store(
|
||||
json_encode($data),
|
||||
"images",
|
||||
$proxy
|
||||
);
|
||||
}
|
||||
|
||||
return $out;
|
||||
}
|
||||
|
||||
private function image_ratio($width, $height){
|
||||
|
||||
$ratio = [
|
||||
480 / $width,
|
||||
360 / $height
|
||||
];
|
||||
|
||||
if($ratio[0] < $ratio[1]){
|
||||
|
||||
$ratio = $ratio[0];
|
||||
}else{
|
||||
|
||||
$ratio = $ratio[1];
|
||||
}
|
||||
|
||||
return [
|
||||
floor($width * $ratio),
|
||||
floor($height * $ratio)
|
||||
];
|
||||
}
|
||||
}
|
|
@ -1209,16 +1209,15 @@ class yt{
|
|||
|
||||
$reel =
|
||||
$reel
|
||||
->shortsLockupViewModel;
|
||||
->reelItemRenderer;
|
||||
|
||||
array_push(
|
||||
$this->out["reel"],
|
||||
[
|
||||
"title" =>
|
||||
$reel
|
||||
->overlayMetadata
|
||||
->primaryText
|
||||
->content,
|
||||
->headline
|
||||
->simpleText,
|
||||
"description" => null,
|
||||
"author" => [
|
||||
"name" => null,
|
||||
|
@ -1226,22 +1225,30 @@ class yt{
|
|||
"avatar" => null
|
||||
],
|
||||
"date" => null,
|
||||
"duration" => null,
|
||||
"views" => null,
|
||||
"duration" =>
|
||||
$this->textualtime2int(
|
||||
$reel
|
||||
->accessibility
|
||||
->accessibilityData
|
||||
->label
|
||||
),
|
||||
"views" =>
|
||||
$this->truncatedcount2int(
|
||||
$reel
|
||||
->viewCountText
|
||||
->simpleText
|
||||
),
|
||||
"thumb" => [
|
||||
"url" =>
|
||||
$reel
|
||||
->thumbnail
|
||||
->sources[0]
|
||||
->thumbnails[0]
|
||||
->url,
|
||||
"ratio" => "9:16"
|
||||
],
|
||||
"url" =>
|
||||
"https://www.youtube.com/watch?v=" .
|
||||
$reel
|
||||
->onTap
|
||||
->innertubeCommand
|
||||
->reelWatchEndpoint
|
||||
->videoId
|
||||
]
|
||||
);
|
||||
|
|
12
settings.php
12
settings.php
|
@ -227,18 +227,10 @@ $settings = [
|
|||
"value" => "solofield",
|
||||
"text" => "Solofield"
|
||||
],
|
||||
[
|
||||
/*[
|
||||
"value" => "pinterest",
|
||||
"text" => "Pinterest"
|
||||
],
|
||||
[
|
||||
"value" => "fivehpx",
|
||||
"text" => "500px"
|
||||
],
|
||||
[
|
||||
"value" => "vsco",
|
||||
"text" => "VSCO"
|
||||
],
|
||||
],*/
|
||||
[
|
||||
"value" => "imgur",
|
||||
"text" => "Imgur"
|
||||
|
|
|
@ -16,7 +16,6 @@
|
|||
|
||||
body{
|
||||
padding:15px 4% 40px;
|
||||
margin:unset;
|
||||
}
|
||||
|
||||
h1,h2,h3,h4,h5,h6{
|
||||
|
|
Loading…
Reference in New Issue