ben biraz daha takintili davrandim, daha detayli bir class yazdim laravel icin.
<?php
namespace App\Services;

use Illuminate\Support\Facades\Log;

class ContentSanitizer
{
    /**
     * ana temizleme metodu - agresif yaklasim
     */
    public static function sanitize(string $content): string
    {
        // control karakterleri temizle (newline/tab/space haric)
        $content = preg_replace('/[\x00-\x08\x0B\x0C\x0E-\x1F\x7F]/u', '', $content);

        // kapsamli zero-width ve gorunmez karakter temizligi
        $content = self::removeInvisibleCharacters($content);

        // format karakterleri (direction marks, vb.)
        $content = self::removeFormatCharacters($content);

        // mathematical alphanumeric variants (A vs 𝐀 vs 𝐴 vs 𝑨)
        $content = self::normalizeMathematicalAlphanumerics($content);

        // fullwidth/halfwidth karakter normalizasyonu (A → A)
        $content = self::normalizeFullwidthCharacters($content);

        // exotic bosluklari temizle
        $content = self::normalizeSpaces($content);

        // em-dash ve en-dash temizligi
        $content = str_replace(['—', '–'], '-', $content);

        // fancy quoteslari normal quotesa cevir
        $content = str_replace(['\'', '\'', '"', '"'], ["'", "'", '"', '"'], $content);

        // ardisik bosluklari temizle
        $content = preg_replace('/[ \t]+/', ' ', $content);
        $content = preg_replace('/\n{3,}/', "\n\n", $content);

        // basta ve sonda bosluk kalmasin
        $content = trim($content);

        return $content;
    }

    /**
     * xero-width ve gorunmez karakterleri temizle. unicode ranges: U+200B - U+200F, U+202A - U+202E, U+FEFF, vs.
     */
    private static function removeInvisibleCharacters(string $content): string
    {
        // zero-width characters
        $patterns = [
            '/\x{200B}/u',  // zero-width space
            '/\x{200C}/u',  // zero-width non-joiner
            '/\x{200D}/u',  // zero-width joiner
            '/\x{200E}/u',  // left-to-right mark
            '/\x{200F}/u',  // right-to-left mark
            '/\x{202A}/u',  // left-to-right embedding
            '/\x{202B}/u',  // right-to-left embedding
            '/\x{202C}/u',  // Pop directional formatting
            '/\x{202D}/u',  // left-to-right override
            '/\x{202E}/u',  // right-to-left override
            '/\x{2060}/u',  // word joiner
            '/\x{2061}/u',  // function application
            '/\x{2062}/u',  // invisible times
            '/\x{2063}/u',  // invisible separator
            '/\x{2064}/u',  // invisible plus
            '/\x{206A}/u',  // inhibit symmetric swapping
            '/\x{206B}/u',  // activate symmetric swapping
            '/\x{206C}/u',  // inhibit arabic form shaping
            '/\x{206D}/u',  // activate arabic form shaping
            '/\x{206E}/u',  // national digit shapes
            '/\x{206F}/u',  // nominal digit shapes
            '/\x{FEFF}/u',  // zero-width no-break space (BOM)
            '/\x{FFF9}/u',  // interlinear annotation anchor
            '/\x{FFFA}/u',  // interlinear annotation separator
            '/\x{FFFB}/u',  // interlinear annotation terminator
            '/\x{180E}/u',  // mongolian vowel separator
            '/\x{061C}/u',  // arabic letter mark
            '/\x{17B4}/u',  // khmer vowel inherent Aq
            '/\x{17B5}/u',  // khmer vowel inherent Aa
        ];

        foreach ($patterns as $pattern) {
            $content = preg_replace($pattern, '', $content);
        }

        return $content;
    }

    /**
     * format karakterlerinide temizle
     */
    private static function removeFormatCharacters(string $content): string
    {
        // unicode category: Cf (format characters)
        $content = preg_replace('/\p{Cf}/u', '', $content);

        return $content;
    }

    /**
     * mathematical alphanumeric symbols normalize et
     */
    private static function normalizeMathematicalAlphanumerics(string $content): string
    {
        // mathematical bold (𝐀-𝐙, 𝐚-𝐳, 𝟎-𝟗)
        $content = preg_replace('/[\x{1D400}-\x{1D7FF}]/u', '', $content);

        // Circled, squared, parenthesized variants
        $content = preg_replace('/[\x{2460}-\x{24FF}]/u', '', $content); // enclosed alphanumerics
        $content = preg_replace('/[\x{1F100}-\x{1F1FF}]/u', '', $content); // enclosed alphanumeric supplement

        return $content;
    }

    /**
     * fullwidth ve halfwidth karakterleri normalize et
     */
    private static function normalizeFullwidthCharacters(string $content): string
    {
        // fullwidth ascii variants (U+FF00 - U+FFEF). bunlari normal ascii'ye cevirmek yerine kaldiralim (potansiyel watermark)
        $content = preg_replace('/[\x{FF00}-\x{FFEF}]/u', '', $content);

        return $content;
    }

    /**
     * exotic space karakteleri normal space cevir
     */
    private static function normalizeSpaces(string $content): string
    {
        $exoticSpaces = [
            '\x{00A0}',  // non-breaking space
            '\x{1680}',  // ogham space mark
            '\x{2000}',  // nn quad
            '\x{2001}',  // mm quad
            '\x{2002}',  // nn space
            '\x{2003}',  // mm space
            '\x{2004}',  // three-per-em space
            '\x{2005}',  // four-per-em space
            '\x{2006}',  // six-per-em space
            '\x{2007}',  // figure space
            '\x{2008}',  // punctuation space
            '\x{2009}',  // thin space
            '\x{200A}',  // hair space
            '\x{202F}',  // narrow no-break space
            '\x{205F}',  // medium mathematical space
            '\x{3000}',  // ideographic space
        ];

        foreach ($exoticSpaces as $space) {
            $content = preg_replace('/' . $space . '/u', ' ', $content);
        }

        return $content;
    }

    /**
     * icerik kalite kontrolu
     */
    public static function validate(string $content): array
    {
        $issues = [];

        // kelime sayisi kontrolu
        $wordCount = str_word_count($content);
        if ($wordCount < 100) {
            $issues[] = "Content too short: {$wordCount} words (minimum 150 expected)";
        } elseif ($wordCount > 250) {
            $issues[] = "Content too long: {$wordCount} words (maximum 200 expected)";
        }

        // em-dash kontrolu
        $emDashCount = substr_count($content, '—') + substr_count($content, '–');
        if ($emDashCount > 2) {
            $issues[] = "Too many em-dashes: {$emDashCount} (maximum 2 allowed)";
        }

        // cumle uzunlugu kontrolu
        $sentences = preg_split('/[.!?]+/', $content, -1, PREG_SPLIT_NO_EMPTY);
        foreach ($sentences as $index => $sentence) {
            $words = str_word_count(trim($sentence));
            if ($words > 25) {
                $preview = substr(trim($sentence), 0, 50) . '...';
                $issues[] = "Sentence #{$index} too long: {$words} words - \"{$preview}\"";
            }
        }

        // generic llm pattern kontrolu
        $genericPatterns = [
            'it is important to note',
            'it\'s important to note',
            'furthermore',
            'moreover',
            'in conclusion',
            'to summarize',
            'in summary',
            'as previously mentioned',
            'it should be noted',
            'one should consider',
            'it\'s worth noting',
        ];

        $contentLower = strtolower($content);
        foreach ($genericPatterns as $pattern) {
            if (stripos($contentLower, $pattern) !== false) {
                $issues[] = "Generic AI phrase detected: \"{$pattern}\"";
            }
        }

        // ellipsis kontrolu
        $ellipsisCount = substr_count($content, '…') + substr_count($content, '...');
        if ($ellipsisCount > 2) {
            $issues[] = "Too many ellipsis: {$ellipsisCount} (use sparingly)";
        }

        return $issues;
    }

    /**
     * sanitize ve validate'i birlikte calistir
     */
    public static function sanitizeAndValidate(string $content, ?string $entityId = null): array
    {
        $original = $content;
        $sanitized = self::sanitize($content);
        $issues = self::validate($sanitized);

        // karakter degisikliklerini logla. bak bakalim ne kadar degisiklik yapmisiz.
        $removedChars = strlen($original) - strlen($sanitized);
        if ($removedChars > 0) {
            Log::info('Hidden characters removed from content', [
                'entity_id' => $entityId,
                'removed_char_count' => $removedChars,
                'original_length' => strlen($original),
                'sanitized_length' => strlen($sanitized),
            ]);
        }

        if (!empty($issues)) {
            Log::warning('Content quality issues detected', [
                'entity_id' => $entityId,
                'issues' => $issues,
                'word_count' => str_word_count($sanitized),
            ]);
        }

        return [
            'content' => $sanitized,
            'issues' => $issues,
            'word_count' => str_word_count($sanitized),
            'removed_chars' => $removedChars,
        ];
    }
}

// kullanmak icin
$sanitized = ContentSanitizer::sanitizeAndValidate($trim_content, $entity->id);
$content = $sanitized['content'];
artik tertemiz bir content var elimizde.