simedkom/app/Services/NewsCrawlerService.php

248 lines
8.2 KiB
PHP

<?php
namespace App\Services;
use App\Enums\Channel;
use App\Models\MediaMonitoring;
use Carbon\Carbon;
use Illuminate\Http\Client\Pool;
use Illuminate\Http\Client\Response;
use Illuminate\Support\Facades\DB;
use Illuminate\Support\Facades\Http;
use Illuminate\Support\Facades\Log;
use Illuminate\Support\Str;
use Symfony\Component\DomCrawler\Crawler;
class NewsCrawlerService
{
/**
* Crawl news with high performance using HTTP Pooling.
*/
public function crawlFromGoogleNews(string $keyword, int $limit = 10, array $themeIds = []): int
{
$searchQuery = $keyword.' Purwakarta';
$url = 'https://news.google.com/rss/search?q='.urlencode($searchQuery).'&hl=id&gl=ID&ceid=ID:id';
try {
$response = Http::get($url);
if ($response->failed()) {
return 0;
}
$xml = simplexml_load_string($response->body());
if (! $xml) {
return 0;
}
$items = [];
foreach ($xml->channel->item as $item) {
if (count($items) >= $limit) {
break;
}
$items[] = [
'google_link' => (string) $item->link,
'title' => (string) $item->title,
'pubDate' => (string) $item->pubDate,
'source' => (string) $item->source,
'description' => (string) $item->description,
];
}
// Phase 1: Parallel Resolving Links
$resolvedLinks = Http::pool(
fn (Pool $pool) => array_map(fn ($item) => $pool->withHeaders([
'User-Agent' => 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/91.0.4472.124 Safari/537.36',
])->get($item['google_link']), $items)
);
$parsedData = [];
$parsedData = [];
foreach ($resolvedLinks as $key => $res) {
// Check if $res is a response or an exception (ConnectionException)
$originalLink = ($res instanceof Response && $res->successful())
? (string) $res->effectiveUri()
: $items[$key]['google_link'];
// Anti-Duplicate Check
if (MediaMonitoring::where('link', $originalLink)->exists()) {
continue;
}
$parsedData[] = array_merge($items[$key], ['original_link' => $originalLink]);
}
// Phase 2: Parallel Fetching Extra Content (Metadata)
$contentResponses = Http::pool(
fn (Pool $pool) => array_map(fn ($data) => $pool->timeout(5)->get($data['original_link']), $parsedData)
);
$count = 0;
foreach ($contentResponses as $key => $res) {
try {
if (! isset($parsedData[$key])) {
continue;
}
$itemData = $parsedData[$key];
$extra = ['writer' => 'Redaksi', 'content' => null];
if ($res instanceof Response && $res->successful()) {
$crawler = new Crawler($res->body());
$extra['writer'] = $this->extractWriter($crawler);
$extra['content'] = $this->extractContent($crawler);
}
// Clean generic Google News text
$finalContent = $this->cleanContent($extra['content'] ?: $itemData['description']);
$cleanTitle = $this->cleanTitle($itemData['title']);
$monitoring = MediaMonitoring::create([
'code' => $this->generateCode(),
'media_name' => $itemData['source'] ?: 'Online Media',
'title' => Str::limit($cleanTitle, 255),
'channel' => Channel::WEBSITE->value,
'writter' => Str::limit($extra['writer'] ?: 'Redaksi', 50),
'link' => $itemData['original_link'],
'news_page' => null, // Asked to be null if not specific pagination
'quote' => Str::limit($finalContent, 500),
'content' => $finalContent,
'keyword' => Str::limit($searchQuery, 100),
'release_date' => Carbon::parse($itemData['pubDate'])->format('Y-m-d'),
]);
if (! empty($themeIds)) {
$monitoring->themes()->syncWithoutDetaching($themeIds);
}
$count++;
} catch (\Throwable $e) {
Log::error('Failed to insert item', [
'error' => $e->getMessage(),
'trace' => $e->getTraceAsString(),
'link' => $itemData['original_link'] ?? 'unknown',
]);
}
}
return $count;
} catch (\Throwable $e) {
Log::error('Crawler error', [
'error' => $e->getMessage(),
'trace' => $e->getTraceAsString(),
]);
return 0;
}
}
private function cleanTitle(string $title): string
{
if (Str::contains($title, ' - ')) {
$parts = explode(' - ', $title);
array_pop($parts);
return implode(' - ', $parts);
}
return $title;
}
private function extractWriter(Crawler $crawler): ?string
{
$selectors = ['meta[name="author"]', 'meta[property="article:author"]', '.author', '.writer'];
foreach ($selectors as $selector) {
$el = $crawler->filter($selector);
if ($el->count()) {
return $el->attr('content') ?: $el->text();
}
}
return null;
}
private function extractContent(Crawler $crawler): ?string
{
// Priority 1: Open Graph Description
$ogDesc = $crawler->filter('meta[property="og:description"]');
if ($ogDesc->count()) {
$content = $ogDesc->attr('content');
if (! $this->isGarbage($content)) {
return $content;
}
}
// Priority 2: Meta Description
$metaDesc = $crawler->filter('meta[name="description"]');
if ($metaDesc->count()) {
$content = $metaDesc->attr('content');
if (! $this->isGarbage($content)) {
return $content;
}
}
// Priority 3: First substantial paragraph
try {
$paragraphs = $crawler->filter('p');
foreach ($paragraphs as $p) {
$text = trim($p->textContent);
if (strlen($text) > 100) {
return $text;
}
}
} catch (\Throwable $e) {
Log::warning('Failed to check paragraphs content', [
'error' => $e->getMessage(),
]);
}
return null;
}
private function isGarbage(string $content): bool
{
$garbage = [
'Comprehensive, up-to-date news coverage',
'Google News',
'Baca berita tanpa iklan',
];
foreach ($garbage as $g) {
if (Str::contains($content, $g)) {
return true;
}
}
return false;
}
private function cleanContent(?string $content): string
{
if (! $content) {
return 'Tidak ada kutipan berita.';
}
$cleaned = strip_tags($content);
$cleaned = preg_replace('/\s+/', ' ', $cleaned);
return trim($cleaned);
}
private function generateCode(): string
{
// Use a lock-safe approach for sequential code
return DB::transaction(function () {
$latestCode = MediaMonitoring::where('code', 'LIKE', 'ADN%')
->lockForUpdate()
->orderBy('code', 'desc')
->first();
$lastNumber = $latestCode
? intval(substr($latestCode->code, 3))
: 0;
$newNumber = str_pad($lastNumber + 1, 6, '0', STR_PAD_LEFT);
return 'ADN'.$newNumber;
});
}
}