ben biraz daha takintili davrandim, daha detayli bir class yazdim laravel icin.
<?php
namespace App\Services;
use Illuminate\Support\Facades\Log;
class ContentSanitizer
{
/**
* ana temizleme metodu - agresif yaklasim
*/
public static function sanitize(string $content): string
{
// control karakterleri temizle (newline/tab/space haric)
$content = preg_replace('/[\x00-\x08\x0B\x0C\x0E-\x1F\x7F]/u', '', $content);
// kapsamli zero-width ve gorunmez karakter temizligi
$content = self::removeInvisibleCharacters($content);
// format karakterleri (direction marks, vb.)
$content = self::removeFormatCharacters($content);
// mathematical alphanumeric variants (A vs 𝐀 vs 𝐴 vs 𝑨)
$content = self::normalizeMathematicalAlphanumerics($content);
// fullwidth/halfwidth karakter normalizasyonu (A → A)
$content = self::normalizeFullwidthCharacters($content);
// exotic bosluklari temizle
$content = self::normalizeSpaces($content);
// em-dash ve en-dash temizligi
$content = str_replace(['', ''], '-', $content);
// fancy quoteslari normal quotesa cevir
$content = str_replace(['\'', '\'', '"', '"'], ["'", "'", '"', '"'], $content);
// ardisik bosluklari temizle
$content = preg_replace('/[ \t]+/', ' ', $content);
$content = preg_replace('/\n{3,}/', "\n\n", $content);
// basta ve sonda bosluk kalmasin
$content = trim($content);
return $content;
}
/**
* xero-width ve gorunmez karakterleri temizle. unicode ranges: U+200B - U+200F, U+202A - U+202E, U+FEFF, vs.
*/
private static function removeInvisibleCharacters(string $content): string
{
// zero-width characters
$patterns = [
'/\x{200B}/u', // zero-width space
'/\x{200C}/u', // zero-width non-joiner
'/\x{200D}/u', // zero-width joiner
'/\x{200E}/u', // left-to-right mark
'/\x{200F}/u', // right-to-left mark
'/\x{202A}/u', // left-to-right embedding
'/\x{202B}/u', // right-to-left embedding
'/\x{202C}/u', // Pop directional formatting
'/\x{202D}/u', // left-to-right override
'/\x{202E}/u', // right-to-left override
'/\x{2060}/u', // word joiner
'/\x{2061}/u', // function application
'/\x{2062}/u', // invisible times
'/\x{2063}/u', // invisible separator
'/\x{2064}/u', // invisible plus
'/\x{206A}/u', // inhibit symmetric swapping
'/\x{206B}/u', // activate symmetric swapping
'/\x{206C}/u', // inhibit arabic form shaping
'/\x{206D}/u', // activate arabic form shaping
'/\x{206E}/u', // national digit shapes
'/\x{206F}/u', // nominal digit shapes
'/\x{FEFF}/u', // zero-width no-break space (BOM)
'/\x{FFF9}/u', // interlinear annotation anchor
'/\x{FFFA}/u', // interlinear annotation separator
'/\x{FFFB}/u', // interlinear annotation terminator
'/\x{180E}/u', // mongolian vowel separator
'/\x{061C}/u', // arabic letter mark
'/\x{17B4}/u', // khmer vowel inherent Aq
'/\x{17B5}/u', // khmer vowel inherent Aa
];
foreach ($patterns as $pattern) {
$content = preg_replace($pattern, '', $content);
}
return $content;
}
/**
* format karakterlerinide temizle
*/
private static function removeFormatCharacters(string $content): string
{
// unicode category: Cf (format characters)
$content = preg_replace('/\p{Cf}/u', '', $content);
return $content;
}
/**
* mathematical alphanumeric symbols normalize et
*/
private static function normalizeMathematicalAlphanumerics(string $content): string
{
// mathematical bold (𝐀-𝐙, 𝐚-𝐳, 𝟎-𝟗)
$content = preg_replace('/[\x{1D400}-\x{1D7FF}]/u', '', $content);
// Circled, squared, parenthesized variants
$content = preg_replace('/[\x{2460}-\x{24FF}]/u', '', $content); // enclosed alphanumerics
$content = preg_replace('/[\x{1F100}-\x{1F1FF}]/u', '', $content); // enclosed alphanumeric supplement
return $content;
}
/**
* fullwidth ve halfwidth karakterleri normalize et
*/
private static function normalizeFullwidthCharacters(string $content): string
{
// fullwidth ascii variants (U+FF00 - U+FFEF). bunlari normal ascii'ye cevirmek yerine kaldiralim (potansiyel watermark)
$content = preg_replace('/[\x{FF00}-\x{FFEF}]/u', '', $content);
return $content;
}
/**
* exotic space karakteleri normal space cevir
*/
private static function normalizeSpaces(string $content): string
{
$exoticSpaces = [
'\x{00A0}', // non-breaking space
'\x{1680}', // ogham space mark
'\x{2000}', // nn quad
'\x{2001}', // mm quad
'\x{2002}', // nn space
'\x{2003}', // mm space
'\x{2004}', // three-per-em space
'\x{2005}', // four-per-em space
'\x{2006}', // six-per-em space
'\x{2007}', // figure space
'\x{2008}', // punctuation space
'\x{2009}', // thin space
'\x{200A}', // hair space
'\x{202F}', // narrow no-break space
'\x{205F}', // medium mathematical space
'\x{3000}', // ideographic space
];
foreach ($exoticSpaces as $space) {
$content = preg_replace('/' . $space . '/u', ' ', $content);
}
return $content;
}
/**
* icerik kalite kontrolu
*/
public static function validate(string $content): array
{
$issues = [];
// kelime sayisi kontrolu
$wordCount = str_word_count($content);
if ($wordCount < 100) {
$issues[] = "Content too short: {$wordCount} words (minimum 150 expected)";
} elseif ($wordCount > 250) {
$issues[] = "Content too long: {$wordCount} words (maximum 200 expected)";
}
// em-dash kontrolu
$emDashCount = substr_count($content, '') + substr_count($content, '');
if ($emDashCount > 2) {
$issues[] = "Too many em-dashes: {$emDashCount} (maximum 2 allowed)";
}
// cumle uzunlugu kontrolu
$sentences = preg_split('/[.!?]+/', $content, -1, PREG_SPLIT_NO_EMPTY);
foreach ($sentences as $index => $sentence) {
$words = str_word_count(trim($sentence));
if ($words > 25) {
$preview = substr(trim($sentence), 0, 50) . '...';
$issues[] = "Sentence #{$index} too long: {$words} words - \"{$preview}\"";
}
}
// generic llm pattern kontrolu
$genericPatterns = [
'it is important to note',
'it\'s important to note',
'furthermore',
'moreover',
'in conclusion',
'to summarize',
'in summary',
'as previously mentioned',
'it should be noted',
'one should consider',
'it\'s worth noting',
];
$contentLower = strtolower($content);
foreach ($genericPatterns as $pattern) {
if (stripos($contentLower, $pattern) !== false) {
$issues[] = "Generic AI phrase detected: \"{$pattern}\"";
}
}
// ellipsis kontrolu
$ellipsisCount = substr_count($content, '
') + substr_count($content, '...');
if ($ellipsisCount > 2) {
$issues[] = "Too many ellipsis: {$ellipsisCount} (use sparingly)";
}
return $issues;
}
/**
* sanitize ve validate'i birlikte calistir
*/
public static function sanitizeAndValidate(string $content, ?string $entityId = null): array
{
$original = $content;
$sanitized = self::sanitize($content);
$issues = self::validate($sanitized);
// karakter degisikliklerini logla. bak bakalim ne kadar degisiklik yapmisiz.
$removedChars = strlen($original) - strlen($sanitized);
if ($removedChars > 0) {
Log::info('Hidden characters removed from content', [
'entity_id' => $entityId,
'removed_char_count' => $removedChars,
'original_length' => strlen($original),
'sanitized_length' => strlen($sanitized),
]);
}
if (!empty($issues)) {
Log::warning('Content quality issues detected', [
'entity_id' => $entityId,
'issues' => $issues,
'word_count' => str_word_count($sanitized),
]);
}
return [
'content' => $sanitized,
'issues' => $issues,
'word_count' => str_word_count($sanitized),
'removed_chars' => $removedChars,
];
}
}
// kullanmak icin
$sanitized = ContentSanitizer::sanitizeAndValidate($trim_content, $entity->id);
$content = $sanitized['content'];artik tertemiz bir content var elimizde.