Files
dredgegram/app/Console/Commands/ScrapeStories.php
T
dredgy b2f2d8b104
linter / quality (push) Canceled after 0s
tests / ci (8.3) (push) Canceled after 0s
tests / ci (8.4) (push) Canceled after 0s
tests / ci (8.5) (push) Canceled after 0s
Initial Commit
2026-08-29 00:17:41 +10:00

63 lines
2.2 KiB
PHP

<?php
namespace App\Console\Commands;
use App\Models\Story;
use App\Services\StoryMediaCacher;
use Illuminate\Console\Attributes\Description;
use Illuminate\Console\Attributes\Signature;
#[Signature('direct:scrape-stories')]
#[Description('Scrape Instagram story content via Playwright')]
class ScrapeStories extends RunsPlaywrightScript
{
protected string $script = 'scrape-story-content.spec.js';
protected function afterRun(): void
{
$path = base_path('playwright/output/scrape-stories.json');
if (! file_exists($path)) {
$this->warn('No output file found — nothing to upsert.');
return;
}
$mediaCacher = app(StoryMediaCacher::class);
$items = json_decode(file_get_contents($path), true);
foreach ($items as $item) {
$attributes = [
'user_id' => $item['user_id'],
'username' => $item['username'],
'profile_pic_url' => $item['profile_pic_url'],
'code' => $item['code'],
'is_video' => $item['is_video'],
'taken_at' => $item['taken_at'],
'expiring_at' => $item['expiring_at'],
'image_url' => $item['image_url'],
'video_url' => $item['video_url'],
'accessibility_caption' => $item['accessibility_caption'],
'muted' => $item['muted'],
];
$existing = Story::find($item['id']);
// Once a story's media has been cached locally, its url columns
// no longer point at Instagram — don't let a re-scrape clobber
// them with a freshly (and just as temporarily) signed CDN url.
if ($existing) {
if (! $mediaCacher->isInstagramUrl($existing->image_url)) {
unset($attributes['image_url']);
}
if (! $mediaCacher->isInstagramUrl($existing->video_url)) {
unset($attributes['video_url']);
}
}
Story::updateOrCreate(['id' => $item['id']], $attributes);
}
$this->info('Upserted ' . count($items) . ' story items.');
}
}