Fixed beforeFetchScenes not running for update releases. Fixed Hookup Hotshot deep scrape broken due missing NATS cookie.

This commit is contained in:
DebaucheryLibrarian
2026-10-03 06:32:03 +02:00
parent b700e05c5e
commit 84ac366e36
5 changed files with 84 additions and 11 deletions
@@ -0,0 +1,24 @@
exports.up = async function(knex) {
// we want to be able to assign unidentified actors
await knex.raw(`
ALTER TABLE releases_actors
ALTER COLUMN actor_id DROP NOT NULL
`);
await knex.schema.alterTable('releases_actors', (table) => {
table.increments('id');
table.string('gender');
});
};
exports.down = async function(knex) {
await knex.raw(`
ALTER TABLE releases_actors
ALTER COLUMN actor_id SET NOT NULL
`);
await knex.schema.alterTable('releases_actors', (table) => {
table.dropColumn('id');
table.dropColumn('gender');
});
};
+5
View File
@@ -348,6 +348,11 @@ const tags = [
slug: 'brunette',
group: 'hair',
},
{
name: 'body writing',
slug: 'body-writing',
description: 'Grab a marker or a lipstick and use your body as a canvas. Label all your holes and make it clear what they should call you.',
},
{
name: 'boots',
slug: 'boots',
+1
View File
@@ -706,6 +706,7 @@ const tagMedia = [
['blowjob', 'azul_hermosa_realitykings', 'Azul Hermosa and Scott Nails in "Diva For A Day"', 'brazzers'],
['blowjob', 3, 'Rose Valie', 'handsonhardcore'],
['blowjob', 2, 'Luna Kitsuen in "Gag Reflex"', 'evilangel'],
['body-writing', 'emma_rosie_julesjordan', 'Emma Rosie', 'julesjordan'],
['bondage', 0, 'Veronica Leal', 'herlimit'],
['brunette', 0, 'Darcie Dolce', 'playboy'],
['bts', 'charlie_red_private', 'Charlie Red', 'private'],
+31 -10
View File
@@ -123,15 +123,21 @@ function fetchMovie(scraper, url, entity, baseRelease, options) {
return fetchScene(scraper, url, entity, baseRelease, options, 'movie');
}
async function scrapeRelease(baseRelease, entitiesByHostname, type = 'scene') {
const entity = baseRelease.entity || entitiesByHostname[urlToHostname(baseRelease.url)];
function shouldDeepScrape(baseRelease) {
return argv.deep && !!(baseRelease.url || baseRelease.path || baseRelease.forceDeep);
}
async function scrapeRelease(baseRelease, entities, type = 'scene') {
const entity = entities.byId[baseRelease.entity?.id]
|| entities.byHostname[urlToHostname(baseRelease.url)]
|| baseRelease.entity;
if (!entity) {
logger.warn(`No entity available for ${baseRelease.url}`);
return baseRelease;
}
if ((!baseRelease.url && !baseRelease.path && !baseRelease.forceDeep) || !argv.deep) {
if (!shouldDeepScrape(baseRelease)) {
return {
...baseRelease,
entity,
@@ -247,22 +253,37 @@ async function scrapeRelease(baseRelease, entitiesByHostname, type = 'scene') {
}
async function scrapeReleases(baseReleases, entitiesByHostname, type) {
const entitiesWithBeforeDataEntries = await Promise.all(Object.entries(entitiesByHostname).map(async ([slug, entity]) => {
if (entity.scraper?.beforeFetchScenes) {
// entities attached to base releases (e.g. from updates) are not included in entitiesByHostname
const entitiesById = Object.fromEntries([
...Object.values(entitiesByHostname),
...baseReleases.map((baseRelease) => baseRelease.entity),
].filter(Boolean).map((entity) => [entity.id, entity]));
// only run before hook for entities that will actually be deep scraped
const deepEntityIds = new Set(baseReleases
.filter((baseRelease) => shouldDeepScrape(baseRelease))
.map((baseRelease) => baseRelease.entity?.id || entitiesByHostname[urlToHostname(baseRelease.url)]?.id)
.filter(Boolean));
const entitiesWithBeforeDataById = Object.fromEntries(await Promise.all(Object.values(entitiesById).map(async (entity) => {
if (deepEntityIds.has(entity.id) && entity.scraper?.beforeFetchScenes) {
const parameters = getRecursiveParameters(entity);
const preData = await entity.scraper.beforeFetchScenes(entity, parameters);
return [slug, { ...entity, preData }];
return [entity.id, { ...entity, preData }];
}
return [slug, entity];
}));
return [entity.id, entity];
})));
const entitiesWithBeforeDataBySlug = Object.fromEntries(entitiesWithBeforeDataEntries.filter(Boolean));
const entitiesWithBeforeData = {
byId: entitiesWithBeforeDataById,
byHostname: Object.fromEntries(Object.entries(entitiesByHostname).map(([hostname, entity]) => [hostname, entitiesWithBeforeDataById[entity.id]])),
};
return Promise.map(
baseReleases,
async (baseRelease) => scrapeRelease(baseRelease, entitiesWithBeforeDataBySlug, type),
async (baseRelease) => scrapeRelease(baseRelease, entitiesWithBeforeData, type),
{ concurrency: 1 },
);
}
+23 -1
View File
@@ -1,6 +1,7 @@
'use strict';
const unprint = require('unprint');
const { nanoid } = require('nanoid');
function scrapeAll(scenes, _channel) {
return scenes.map(({ query }) => {
@@ -40,6 +41,12 @@ async function fetchLatest(channel, page = 1) {
return res.status;
}
async function beforeFetchScenes(channel) {
const res = await unprint.get(channel.url);
return { nats: res.cookies.nats || nanoid() };
}
function scrapeScene({ query, html }, { url, entity, baseRelease }) {
const release = {};
@@ -81,6 +88,20 @@ function scrapeScene({ query, html }, { url, entity, baseRelease }) {
return release;
}
async function fetchScene(url, entity, baseRelease, context) {
const res = await unprint.get(url, {
cookies: {
nats: context.beforeFetchScenes?.nats,
},
});
if (res.ok) {
return scrapeScene(res.context, { url, entity, baseRelease });
}
return res.status;
}
function scrapeProfile({ query }) {
const profile = {};
@@ -110,5 +131,6 @@ async function fetchProfile(actor, entity) {
module.exports = {
fetchLatest,
fetchProfile,
scrapeScene,
fetchScene,
beforeFetchScenes,
};