From de78653447f62392cffc51f39222635e29fbdb7f Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Thu, 14 Aug 2025 16:32:17 +0000 Subject: [PATCH 1/5] Initial plan From 7d7f6431e8690b2773d97bbfda969571089fbb34 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Thu, 14 Aug 2025 16:47:33 +0000 Subject: [PATCH 2/5] Implement cleaning pipeline with universal repetition suppression Co-authored-by: mssoftjp <115339791+mssoftjp@users.noreply.github.com> --- src/config/CleaningConfig.ts | 178 ++++++++ .../transcription/TranscriptionService.ts | 40 +- .../cleaning/PromptContaminationCleaner.ts | 232 ++++++++++ .../cleaning/StandardCleaningPipeline.ts | 246 ++++++++++ .../cleaning/UniversalRepetitionCleaner.ts | 420 ++++++++++++++++++ src/core/transcription/cleaning/interfaces.ts | 117 +++++ 6 files changed, 1230 insertions(+), 3 deletions(-) create mode 100644 src/config/CleaningConfig.ts create mode 100644 src/core/transcription/cleaning/PromptContaminationCleaner.ts create mode 100644 src/core/transcription/cleaning/StandardCleaningPipeline.ts create mode 100644 src/core/transcription/cleaning/UniversalRepetitionCleaner.ts create mode 100644 src/core/transcription/cleaning/interfaces.ts diff --git a/src/config/CleaningConfig.ts b/src/config/CleaningConfig.ts new file mode 100644 index 0000000..6488861 --- /dev/null +++ b/src/config/CleaningConfig.ts @@ -0,0 +1,178 @@ +/** + * Configuration for the text cleaning pipeline + * Centralized management of thresholds, patterns, and safety limits + * Language-independent approach with universal patterns + */ + +export interface SafetyThresholds { + /** Maximum reduction allowed by a single cleaner (0.0-1.0) */ + singleCleanerMaxReduction: number; + /** Maximum reduction allowed for a single pattern match (0.0-1.0) */ + singlePatternMaxReduction: number; + /** Maximum reduction allowed for repetition patterns (0.0-1.0) */ + repetitionPatternMaxReduction: number; + /** Maximum reduction allowed per iteration (0.0-1.0) */ + iterationReductionLimit: number; + /** Emergency fallback threshold - rollback if exceeded (0.0-1.0) */ + emergencyFallbackThreshold: number; + /** Warning threshold for logging (0.0-1.0) */ + warningThreshold: number; +} + +export interface RepetitionThresholds { + /** Base threshold for repetition detection */ + baseThreshold: number; + /** Length factor for dynamic threshold calculation */ + lengthFactor: number; + /** Divisor for dynamic threshold based on text length */ + dynamicThresholdDivisor: number; + /** Ratio of repeated items to keep (0.0-1.0) */ + shortCharKeepRatio: number; + /** Threshold for sentence repetition detection */ + sentenceRepetition: number; + /** Similarity threshold for sentence comparison (0.0-1.0) */ + similarityThreshold: number; + /** Minimum sentence length for similarity comparison */ + minimumSentenceLengthForSimilarity: number; + /** Maximum consecutive newlines to allow */ + consecutiveNewlineLimit: number; + /** N-gram configuration for phrase repetition detection */ + ngram: { + min: number; + max: number; + thresholds: Array<{ n: number; repeat: number }>; + }; + /** Enumeration detection configuration */ + enumerationDetection: { + enabled: boolean; + minRepeatCount: number; + }; + /** Paragraph repetition detection configuration */ + paragraphRepeat: { + enabled: boolean; + headChars: number; + }; +} + +export interface ContaminationPatterns { + /** Instruction patterns to remove from text beginning */ + instructionPatterns: string[]; + /** XML pattern groups for different contexts */ + xmlPatternGroups: { + completeXmlTags: string[]; + sentenceBoundedTags: string[]; + lineBoundedTags: string[]; + standaloneTags: string[]; + }; + /** Context patterns for general cleanup */ + contextPatterns: string[]; + /** Prompt snippet lengths for partial matching */ + promptSnippetLengths: number[]; +} + +export interface CleaningConfig { + safety: SafetyThresholds; + repetition: RepetitionThresholds; + contamination: ContaminationPatterns; +} + +/** + * Default cleaning configuration with conservative settings + * Optimized for safety and minimal false positives + */ +export const CLEANING_CONFIG: CleaningConfig = { + safety: { + singleCleanerMaxReduction: 0.3, + singlePatternMaxReduction: 0.15, + repetitionPatternMaxReduction: 0.25, + iterationReductionLimit: 0.2, + emergencyFallbackThreshold: 0.5, + warningThreshold: 0.15 + }, + + repetition: { + baseThreshold: 3, + lengthFactor: 2, + dynamicThresholdDivisor: 100, + shortCharKeepRatio: 0.3, + sentenceRepetition: 3, + similarityThreshold: 0.85, + minimumSentenceLengthForSimilarity: 10, + consecutiveNewlineLimit: 3, + ngram: { + min: 3, + max: 10, + thresholds: [ + { n: 3, repeat: 4 }, + { n: 4, repeat: 3 }, + { n: 5, repeat: 3 }, + { n: 6, repeat: 2 }, + { n: 7, repeat: 2 }, + { n: 8, repeat: 2 }, + { n: 9, repeat: 2 }, + { n: 10, repeat: 2 } + ] + }, + enumerationDetection: { + enabled: true, + minRepeatCount: 3 + }, + paragraphRepeat: { + enabled: true, + headChars: 50 + } + }, + + contamination: { + instructionPatterns: [ + // Universal instruction patterns (language-independent structure) + 'Please transcribe only the following audio content', + 'Do not include this instruction in your output', + 'Record only the speaker\'s statements accurately', + 'PLEASE TRANSCRIBE ONLY THE FOLLOWING AUDIO CONTENT', + 'RECORD ONLY THE SPEAKER\'S STATEMENTS ACCURATELY', + '以下の音声内容のみを文字に起こしてください', + 'この指示文は出力に含めないでください', + '話者の発言内容だけを正確に記録してください', + '请仅转录以下音频内容', + '不要包含此指令在输出中', + '请准确记录说话者的发言内容', + '다음 음성 내용만 전사해주세요', + '이 지시사항을 출력에 포함하지 마세요', + '화자의 발언 내용만 정확히 기록해주세요' + ], + + xmlPatternGroups: { + completeXmlTags: [ + '/]*>([\\s\\S]*?)<\\/TRANSCRIPT>/g', + '/]*>([\\s\\S]*?)<\\/transcript>/g', + '/]*>([\\s\\S]*?)<\\/TRANSCRIPTION>/g' + ], + sentenceBoundedTags: [ + '/<\\/?TRANSCRIPT[^>]*>/g', + '/<\\/?transcript[^>]*>/g', + '/<\\/?TRANSCRIPTION[^>]*>/g' + ], + lineBoundedTags: [ + '/^\\s*<[^>]*>\\s*$/gm', + '/^\\s*<\\/[^>]*>\\s*$/gm' + ], + standaloneTags: [ + '/<[^>]*\\/>/g', + '/<\\w+[^>]*>\\s*<\\/\\w+>/g' + ] + }, + + contextPatterns: [ + // Universal context patterns (structural) + '/^\\s*\\([^)]*Speaker[^)]*only[^)]*\\)\\s*$/gmi', + '/^\\s*([^)]*話者[^)]*のみ[^)]*)\\s*$/gm', + '/^\\s*([^)]*说话者[^)]*内容[^)]*)\\s*$/gm', + '/^\\s*([^)]*화자[^)]*발언만[^)]*)\\s*$/gm', + '/\\b(output|format|instruction)\\s*:?\\s*$/gmi', + '/^\\s*(transcribe|record|speak)\\s+only\\b.*$/gmi' + ], + + promptSnippetLengths: [20, 30, 40, 50] + } +}; diff --git a/src/core/transcription/TranscriptionService.ts b/src/core/transcription/TranscriptionService.ts index 0fcb533..332a538 100644 --- a/src/core/transcription/TranscriptionService.ts +++ b/src/core/transcription/TranscriptionService.ts @@ -5,6 +5,9 @@ import { API_CONSTANTS, DEFAULT_TRANSCRIPTION_SETTINGS, TRANSCRIPTION_MODEL_COST import { ObsidianHttpClient } from '../../utils/ObsidianHttpClient'; import { createServiceLogger } from '../../services'; import { Logger } from '../../utils'; +import { StandardCleaningPipeline } from './cleaning/StandardCleaningPipeline'; +import { PromptContaminationCleaner } from './cleaning/PromptContaminationCleaner'; +import { UniversalRepetitionCleaner } from './cleaning/UniversalRepetitionCleaner'; // Prompt constants for string drift prevention (文字列ドリフト対策) const PROMPT_CONSTANTS = { @@ -51,6 +54,7 @@ export class TranscriptionService implements ITranscriptionProvider { private apiKey: string; private corrector: DictionaryCorrector; private logger: Logger; + private cleaningPipeline: StandardCleaningPipeline; // 注: gpt-4o-mini-transcribe と gpt-4o-transcribe の両方が日本語音声認識で高精度 // コスト重視なら mini、わずかな精度向上が必要なら通常版を選択 private model: 'gpt-4o-transcribe' | 'gpt-4o-mini-transcribe' = DEFAULT_TRANSCRIPTION_SETTINGS.model; @@ -62,6 +66,12 @@ export class TranscriptionService implements ITranscriptionProvider { this.corrector = new DictionaryCorrector({ correctionDictionary: dictionary }); + + // Initialize the new cleaning pipeline + this.cleaningPipeline = new StandardCleaningPipeline([ + new PromptContaminationCleaner(), + new UniversalRepetitionCleaner() + ]); } async transcribe(audioBlob: Blob, language: string): Promise { @@ -152,7 +162,7 @@ export class TranscriptionService implements ITranscriptionProvider { } // Clean up GPT-4o specific artifacts - originalText = this.cleanGPT4oResponse(originalText, language); + originalText = await this.cleanGPT4oResponse(originalText, language); // プロンプトエラーの検出(音声が無音の場合にプロンプトが返される問題) @@ -270,9 +280,33 @@ ${PROMPT_CONSTANTS.KOREAN.SPEAKER_ONLY} } /** - * Clean GPT-4o specific response artifacts with language-specific processing + * Clean GPT-4o specific response artifacts using the new cleaning pipeline + * Falls back to legacy cleaning if the pipeline fails */ - private cleanGPT4oResponse(text: string, language: string): string { + private async cleanGPT4oResponse(text: string, language: string): Promise { + const normalizedLang = this.normalizeLanguage(language); + + try { + // Use the new cleaning pipeline + const result = await this.cleaningPipeline.execute(text, normalizedLang, { + language: normalizedLang, + originalLength: text.length, + enableDetailedLogging: false + }); + + return result.finalText.trim().replace(/\n{3,}/g, '\n\n'); + + } catch (error) { + // Fall back to legacy cleaning if pipeline fails + this.logger.warn('Cleaning pipeline failed, falling back to legacy cleaning', error); + return this.legacyCleanGPT4oResponse(text, language); + } + } + + /** + * Legacy cleaning method (preserved for fallback) + */ + private legacyCleanGPT4oResponse(text: string, language: string): string { // Normalize language for processing const normalizedLang = this.normalizeLanguage(language); // First attempt: Extract content from complete TRANSCRIPT tags diff --git a/src/core/transcription/cleaning/PromptContaminationCleaner.ts b/src/core/transcription/cleaning/PromptContaminationCleaner.ts new file mode 100644 index 0000000..42956e1 --- /dev/null +++ b/src/core/transcription/cleaning/PromptContaminationCleaner.ts @@ -0,0 +1,232 @@ +/** + * Universal prompt contamination cleaner + * Removes instruction prompts, XML tags, and context patterns + * Language-independent approach using structural patterns + */ + +import { CleaningResult, TextCleaner, CleaningContext } from './interfaces'; +import { CLEANING_CONFIG } from '../../../config/CleaningConfig'; + +// Simple logger interface for tests +interface SimpleLogger { + debug: (message: string, metadata?: any) => void; + warn: (message: string, metadata?: any) => void; +} + +// Fallback logger for test environments +const createFallbackLogger = (): SimpleLogger => ({ + debug: () => {/* no-op */}, + warn: () => {/* no-op */} +}); + +export class PromptContaminationCleaner implements TextCleaner { + readonly name = 'PromptContaminationCleaner'; + readonly enabled = true; + private logger: SimpleLogger; + + constructor() { + try { + // Try to use the real logger + const { createServiceLogger } = require('../../../services'); + this.logger = createServiceLogger('PromptContaminationCleaner'); + } catch { + // Fall back to no-op logger for tests + this.logger = createFallbackLogger(); + } + } + + clean(text: string, language: string, context?: CleaningContext): CleaningResult { + const original = text; + const startTime = performance.now(); + let cleaned = text; + const issues: string[] = []; + let patternsMatched = 0; + + const { instructionPatterns, xmlPatternGroups, contextPatterns, promptSnippetLengths } = + CLEANING_CONFIG.contamination; + + // Step 1: Handle XML tags (highest priority) + cleaned = this.removeXmlTags(cleaned, xmlPatternGroups); + if (cleaned !== text) { + patternsMatched++; + this.logger.debug('XML tags removed'); + } + + // Step 2: Remove exact instruction matches at text beginning + const beforeInstructions = cleaned; + cleaned = this.removeInstructionPatterns(cleaned, instructionPatterns); + if (cleaned !== beforeInstructions) { + patternsMatched++; + this.logger.debug('Instruction patterns removed'); + } + + // Step 3: Remove snippet matches (conservative) + const beforeSnippets = cleaned; + cleaned = this.removeSnippetPatterns(cleaned, instructionPatterns, promptSnippetLengths); + if (cleaned !== beforeSnippets) { + patternsMatched++; + this.logger.debug('Snippet patterns removed'); + } + + // Step 4: Apply context patterns + const beforeContext = cleaned; + cleaned = this.removeContextPatterns(cleaned, contextPatterns); + if (cleaned !== beforeContext) { + patternsMatched++; + this.logger.debug('Context patterns removed'); + } + + // Step 5: Clean up excessive whitespace + cleaned = this.normalizeWhitespace(cleaned); + + const processingTime = performance.now() - startTime; + const reductionRatio = original.length > 0 ? (original.length - cleaned.length) / original.length : 0; + + // Add warnings for significant changes + if (reductionRatio > CLEANING_CONFIG.safety.warningThreshold) { + issues.push(`High reduction ratio: ${Math.round(reductionRatio * 100)}%`); + } + + if (context?.enableDetailedLogging) { + this.logger.debug('Prompt contamination cleaning completed', { + originalLength: original.length, + cleanedLength: cleaned.length, + reductionRatio: reductionRatio.toFixed(3), + patternsMatched, + processingTime: `${processingTime.toFixed(2)}ms` + }); + } + + return { + cleanedText: cleaned, + issues, + hasSignificantChanges: reductionRatio > 0.05, + metadata: { + reductionRatio, + patternsMatched, + processingTime + } + }; + } + + /** + * Remove XML tags in priority order + */ + private removeXmlTags(text: string, xmlPatternGroups: typeof CLEANING_CONFIG.contamination.xmlPatternGroups): string { + let cleaned = text; + + // Process in priority order: complete tags first, then partial cleanup + const groups = ['completeXmlTags', 'sentenceBoundedTags', 'lineBoundedTags', 'standaloneTags'] as const; + + for (const group of groups) { + const patterns = xmlPatternGroups[group]; + for (const pattern of patterns) { + try { + // Extract regex pattern and flags from string representation + const lastSlash = pattern.lastIndexOf('/'); + if (lastSlash > 0) { + const regexPattern = pattern.slice(1, lastSlash); + const flags = pattern.slice(lastSlash + 1); + const regex = new RegExp(regexPattern, flags); + + if (group === 'completeXmlTags') { + // For complete XML tags, extract content instead of removing + const match = cleaned.match(regex); + if (match && match[1]) { + cleaned = match[1].trim(); + } + } else { + // For other patterns, remove entirely + cleaned = cleaned.replace(regex, ''); + } + } + } catch (error) { + this.logger.warn(`Invalid regex pattern in ${group}`, { pattern, error }); + } + } + } + + return cleaned; + } + + /** + * Remove instruction patterns from the beginning of text + */ + private removeInstructionPatterns(text: string, instructionPatterns: string[]): string { + let cleaned = text; + + for (const pattern of instructionPatterns) { + // Check if text starts with this pattern + if (cleaned.trimStart().startsWith(pattern)) { + const index = cleaned.indexOf(pattern); + if (index >= 0) { + cleaned = cleaned.slice(index + pattern.length).trim(); + } + } + } + + return cleaned; + } + + /** + * Remove snippet patterns (partial matches with context) + */ + private removeSnippetPatterns(text: string, instructionPatterns: string[], snippetLengths: number[]): string { + let cleaned = text; + + for (const pattern of instructionPatterns) { + for (const length of snippetLengths) { + if (pattern.length < length) continue; + + const snippet = pattern.slice(0, length); + // Escape special regex characters + const escapedSnippet = snippet.replace(/[-/\\^$*+?.()|[\]{}]/g, '\\$&'); + + // Create contextual regex for snippet matching + // Look for snippet followed by instruction-like endings + const contextRegex = new RegExp( + `${escapedSnippet}[^。.!?!?\\n]{0,50}(?:ください|してください|です|ます|please|only|content)\\b`, + 'gi' + ); + + cleaned = cleaned.replace(contextRegex, ''); + } + } + + return cleaned; + } + + /** + * Apply context patterns for general cleanup + */ + private removeContextPatterns(text: string, contextPatterns: string[]): string { + let cleaned = text; + + for (const pattern of contextPatterns) { + try { + // Extract regex pattern and flags from string representation + const lastSlash = pattern.lastIndexOf('/'); + if (lastSlash > 0) { + const regexPattern = pattern.slice(1, lastSlash); + const flags = pattern.slice(lastSlash + 1); + const regex = new RegExp(regexPattern, flags); + cleaned = cleaned.replace(regex, ''); + } + } catch (error) { + this.logger.warn('Invalid context pattern regex', { pattern, error }); + } + } + + return cleaned; + } + + /** + * Normalize whitespace while preserving structure + */ + private normalizeWhitespace(text: string): string { + return text + .replace(/\n{3,}/g, '\n\n') // Limit consecutive newlines + .replace(/^\s*\n/gm, '') // Remove lines with only whitespace + .trim(); + } +} \ No newline at end of file diff --git a/src/core/transcription/cleaning/StandardCleaningPipeline.ts b/src/core/transcription/cleaning/StandardCleaningPipeline.ts new file mode 100644 index 0000000..94d997a --- /dev/null +++ b/src/core/transcription/cleaning/StandardCleaningPipeline.ts @@ -0,0 +1,246 @@ +/** + * Standard implementation of the cleaning pipeline + * Orchestrates multiple cleaners with safety monitoring and metrics collection + */ + +import { CleaningPipeline, CleaningContext, TextCleaner, SafetyCheckResult } from './interfaces'; +import { CLEANING_CONFIG } from '../../../config/CleaningConfig'; + +// Simple logger interface for tests +interface SimpleLogger { + info: (message: string, metadata?: any) => void; + debug: (message: string, metadata?: any) => void; + warn: (message: string, metadata?: any) => void; + error: (message: string, error?: any) => void; +} + +// Fallback logger for test environments +const createFallbackLogger = (): SimpleLogger => ({ + info: () => {/* no-op */}, + debug: () => {/* no-op */}, + warn: () => {/* no-op */}, + error: () => {/* no-op */} +}); + +export class StandardCleaningPipeline implements CleaningPipeline { + readonly name = 'StandardCleaningPipeline'; + private logger: SimpleLogger; + + constructor(private cleaners: TextCleaner[] = []) { + try { + // Try to use the real logger + const { createServiceLogger } = require('../../../services'); + this.logger = createServiceLogger('StandardCleaningPipeline'); + } catch { + // Fall back to no-op logger for tests + this.logger = createFallbackLogger(); + } + } + + async execute(text: string, language: string, context?: CleaningContext) { + const startTime = performance.now(); + const originalLength = text.length; + + const fullContext: CleaningContext = { + language, + originalLength, + startTime, + ...context + }; + + this.logger.info('Starting cleaning pipeline', { + originalLength, + language, + cleanerCount: this.cleaners.length, + enabledCleaners: this.cleaners.filter(c => c.enabled).length + }); + + let currentText = text; + const cleanerResults: Array<{ + cleanerName: string; + reductionRatio: number; + processingTime: number; + issues: string[]; + }> = []; + + // Execute each cleaner in sequence + for (const cleaner of this.cleaners) { + if (!cleaner.enabled) { + this.logger.debug(`Skipping disabled cleaner: ${cleaner.name}`); + continue; + } + + const cleanerStartTime = performance.now(); + const textBeforeCleaner = currentText; + + try { + this.logger.debug(`Executing cleaner: ${cleaner.name}`, { + inputLength: textBeforeCleaner.length + }); + + const result = await Promise.resolve( + cleaner.clean(textBeforeCleaner, language, fullContext) + ); + + // Safety check + const safetyCheck = this.performSafetyCheck( + textBeforeCleaner, + result.cleanedText, + cleaner.name + ); + + if (safetyCheck.action === 'rollback') { + this.logger.warn(`Rolling back cleaner ${cleaner.name}`, { + reason: safetyCheck.reason, + reductionRatio: safetyCheck.reductionRatio + }); + // Keep original text, don't apply this cleaner + } else if (safetyCheck.action === 'skip') { + this.logger.warn(`Skipping cleaner ${cleaner.name}`, { + reason: safetyCheck.reason + }); + // Keep original text, don't apply this cleaner + } else { + // Apply the cleaned text + currentText = result.cleanedText; + } + + const processingTime = performance.now() - cleanerStartTime; + const reductionRatio = textBeforeCleaner.length > 0 + ? (textBeforeCleaner.length - currentText.length) / textBeforeCleaner.length + : 0; + + cleanerResults.push({ + cleanerName: cleaner.name, + reductionRatio, + processingTime, + issues: result.issues + }); + + this.logger.debug(`Completed cleaner: ${cleaner.name}`, { + outputLength: currentText.length, + reductionRatio: reductionRatio.toFixed(3), + processingTime: `${processingTime.toFixed(2)}ms`, + issueCount: result.issues.length + }); + + // Log warnings for significant changes + if (reductionRatio > CLEANING_CONFIG.safety.warningThreshold) { + this.logger.warn(`High reduction ratio in ${cleaner.name}`, { + reductionRatio: reductionRatio.toFixed(3), + threshold: CLEANING_CONFIG.safety.warningThreshold + }); + } + + } catch (error) { + this.logger.error(`Error in cleaner ${cleaner.name}`, error); + cleanerResults.push({ + cleanerName: cleaner.name, + reductionRatio: 0, + processingTime: performance.now() - cleanerStartTime, + issues: [`Error: ${error instanceof Error ? error.message : String(error)}`] + }); + // Continue with the original text if a cleaner fails + } + } + + const totalProcessingTime = performance.now() - startTime; + const totalReductionRatio = originalLength > 0 + ? (originalLength - currentText.length) / originalLength + : 0; + + // Final safety check for the entire pipeline + const finalSafetyCheck = this.performSafetyCheck(text, currentText, 'Pipeline'); + if (finalSafetyCheck.action === 'rollback') { + this.logger.error('Pipeline safety failure - reverting to original text', { + reason: finalSafetyCheck.reason, + totalReductionRatio: finalSafetyCheck.reductionRatio + }); + currentText = text; + } + + this.logger.info('Cleaning pipeline completed', { + originalLength, + finalLength: currentText.length, + totalReductionRatio: totalReductionRatio.toFixed(3), + totalProcessingTime: `${totalProcessingTime.toFixed(2)}ms`, + cleanersExecuted: cleanerResults.length + }); + + return { + finalText: currentText, + metadata: { + totalOriginalLength: originalLength, + totalFinalLength: currentText.length, + totalReductionRatio, + cleanerResults + } + }; + } + + /** + * Perform safety checks on cleaning results + */ + private performSafetyCheck( + originalText: string, + cleanedText: string, + cleanerName: string + ): SafetyCheckResult { + const originalLength = originalText.length; + const cleanedLength = cleanedText.length; + + if (originalLength === 0) { + return { isSafe: true, action: 'proceed', reductionRatio: 0 }; + } + + const reductionRatio = (originalLength - cleanedLength) / originalLength; + + // Check against emergency fallback threshold + if (reductionRatio > CLEANING_CONFIG.safety.emergencyFallbackThreshold) { + return { + isSafe: false, + reason: `Reduction ratio ${reductionRatio.toFixed(3)} exceeds emergency threshold ${CLEANING_CONFIG.safety.emergencyFallbackThreshold}`, + action: 'rollback', + reductionRatio + }; + } + + // Check against single cleaner threshold + if (reductionRatio > CLEANING_CONFIG.safety.singleCleanerMaxReduction) { + return { + isSafe: false, + reason: `Reduction ratio ${reductionRatio.toFixed(3)} exceeds single cleaner threshold ${CLEANING_CONFIG.safety.singleCleanerMaxReduction}`, + action: 'skip', + reductionRatio + }; + } + + return { isSafe: true, action: 'proceed', reductionRatio }; + } + + /** + * Add a cleaner to the pipeline + */ + addCleaner(cleaner: TextCleaner): void { + this.cleaners.push(cleaner); + } + + /** + * Remove a cleaner from the pipeline + */ + removeCleaner(cleanerName: string): boolean { + const index = this.cleaners.findIndex(c => c.name === cleanerName); + if (index >= 0) { + this.cleaners.splice(index, 1); + return true; + } + return false; + } + + /** + * Get all cleaners in the pipeline + */ + getCleaners(): readonly TextCleaner[] { + return [...this.cleaners]; + } +} \ No newline at end of file diff --git a/src/core/transcription/cleaning/UniversalRepetitionCleaner.ts b/src/core/transcription/cleaning/UniversalRepetitionCleaner.ts new file mode 100644 index 0000000..c1b7b29 --- /dev/null +++ b/src/core/transcription/cleaning/UniversalRepetitionCleaner.ts @@ -0,0 +1,420 @@ +/** + * Universal repetition cleaner - language-independent approach + * Handles character/token/phrase/sentence/paragraph/enumeration/tail repetitions + * Uses structural patterns rather than language-specific rules + */ + +import { CleaningResult, TextCleaner, CleaningContext } from './interfaces'; +import { CLEANING_CONFIG } from '../../../config/CleaningConfig'; + +// Simple logger interface for tests +interface SimpleLogger { + debug: (message: string, metadata?: any) => void; +} + +// Fallback logger for test environments +const createFallbackLogger = (): SimpleLogger => ({ + debug: () => {/* no-op */} +}); + +export class UniversalRepetitionCleaner implements TextCleaner { + readonly name = 'UniversalRepetitionCleaner'; + readonly enabled = true; + private logger: SimpleLogger; + + constructor() { + try { + // Try to use the real logger + const { createServiceLogger } = require('../../../services'); + this.logger = createServiceLogger('UniversalRepetitionCleaner'); + } catch { + // Fall back to no-op logger for tests + this.logger = createFallbackLogger(); + } + } + + clean(text: string, language: string, context?: CleaningContext): CleaningResult { + const original = text; + const startTime = performance.now(); + let cleaned = text; + const issues: string[] = []; + let totalReductionSteps = 0; + + // Safety check: don't process very short texts + if (text.length < 10) { + return { + cleanedText: text, + issues: [], + hasSignificantChanges: false, + metadata: { processingTime: performance.now() - startTime } + }; + } + + // Step 1: Character and symbol repetition suppression + const step1Result = this.suppressCharacterRepetitions(cleaned); + cleaned = step1Result.text; + totalReductionSteps += step1Result.changes; + + // Step 2: Token repetition suppression + const step2Result = this.suppressTokenRepetitions(cleaned, original.length); + cleaned = step2Result.text; + totalReductionSteps += step2Result.changes; + + // Step 3: Sentence repetition suppression + const step3Result = this.suppressSentenceRepetitions(cleaned); + cleaned = step3Result.text; + totalReductionSteps += step3Result.changes; + + // Step 4: Enumeration compression + if (CLEANING_CONFIG.repetition.enumerationDetection.enabled) { + const step4Result = this.compressEnumerations(cleaned); + cleaned = step4Result.text; + totalReductionSteps += step4Result.changes; + } + + // Step 5: Paragraph repetition suppression + if (CLEANING_CONFIG.repetition.paragraphRepeat.enabled) { + const step5Result = this.suppressParagraphRepetitions(cleaned); + cleaned = step5Result.text; + totalReductionSteps += step5Result.changes; + } + + // Step 6: Tail repetition suppression + const step6Result = this.suppressTailRepetitions(cleaned); + cleaned = step6Result.text; + totalReductionSteps += step6Result.changes; + + // Step 7: Final cleanup + cleaned = this.finalCleanup(cleaned); + + const processingTime = performance.now() - startTime; + const reductionRatio = original.length > 0 ? (original.length - cleaned.length) / original.length : 0; + + // Safety check: if reduction is too high, fall back to original + if (reductionRatio > CLEANING_CONFIG.safety.emergencyFallbackThreshold) { + issues.push(`Emergency fallback: excessive reduction (${Math.round(reductionRatio * 100)}%)`); + cleaned = original; + } + + if (context?.enableDetailedLogging) { + this.logger.debug('Universal repetition cleaning completed', { + originalLength: original.length, + cleanedLength: cleaned.length, + reductionRatio: reductionRatio.toFixed(3), + totalReductionSteps, + processingTime: `${processingTime.toFixed(2)}ms` + }); + } + + return { + cleanedText: cleaned, + issues, + hasSignificantChanges: reductionRatio > 0.05, + metadata: { + reductionRatio, + totalReductionSteps, + processingTime + } + }; + } + + /** + * Suppress character and symbol repetitions (language-independent) + */ + private suppressCharacterRepetitions(text: string): { text: string; changes: number } { + let result = text; + let changes = 0; + + // Define repetition patterns with their limits + const patterns = [ + { pattern: /([.!?])\1{5,}/g, replacement: '$1$1$1', description: 'punctuation' }, + { pattern: /[…]{3,}/g, replacement: '…', description: 'ellipsis' }, + { pattern: /[-—–]{6,}/g, replacement: '—', description: 'dashes' }, + { pattern: /[•·・]{6,}/g, replacement: '・', description: 'bullets' }, + { pattern: /[,,]{4,}/g, replacement: ',,', description: 'commas' }, + { pattern: /[\s]{4,}/g, replacement: ' ', description: 'spaces' } + ]; + + for (const { pattern, replacement } of patterns) { + const before = result; + result = result.replace(pattern, replacement); + if (result !== before) changes++; + } + + return { text: result, changes }; + } + + /** + * Suppress token repetitions using language-independent tokenization + */ + private suppressTokenRepetitions(text: string, originalLength: number): { text: string; changes: number } { + const config = CLEANING_CONFIG.repetition; + let changes = 0; + + // Language-independent tokenization + const tokens = this.tokenizeText(text); + const tokenCounts = new Map(); + + // Count normalized tokens (case-insensitive, normalized) + for (const token of tokens) { + const normalizedToken = this.normalizeToken(token); + if (this.isSignificantToken(normalizedToken)) { + tokenCounts.set(normalizedToken, (tokenCounts.get(normalizedToken) || 0) + 1); + } + } + + // Calculate dynamic threshold based on text length + const dynamicThreshold = config.baseThreshold + + Math.floor(originalLength / config.dynamicThresholdDivisor) * config.lengthFactor; + + let result = text; + + // Reduce excessive repetitions + for (const [normalizedToken, count] of tokenCounts) { + if (count >= dynamicThreshold) { + const keepCount = Math.max(1, Math.floor(count * config.shortCharKeepRatio)); + const reductionNeeded = count - keepCount; + + // Create escape pattern for regex + const escapedToken = this.escapeRegex(normalizedToken); + const tokenRegex = new RegExp(`\\b${escapedToken}\\b`, 'gi'); + + // Remove excess occurrences + let removed = 0; + result = result.replace(tokenRegex, (match) => { + removed++; + return removed <= reductionNeeded ? '' : match; + }); + + if (removed > 0) changes++; + } + } + + return { text: result, changes }; + } + + /** + * Suppress sentence repetitions using structural similarity + */ + private suppressSentenceRepetitions(text: string): { text: string; changes: number } { + const config = CLEANING_CONFIG.repetition; + let changes = 0; + + // Split into sentences using universal punctuation + const sentences = this.splitIntoSentences(text); + const result: string[] = []; + const seenSentences = new Map(); + + for (const sentence of sentences) { + const normalizedSentence = this.normalizeSentence(sentence); + + if (normalizedSentence.length >= config.minimumSentenceLengthForSimilarity) { + const count = seenSentences.get(normalizedSentence) || 0; + seenSentences.set(normalizedSentence, count + 1); + + // Only keep if under repetition threshold + if (count + 1 <= config.sentenceRepetition) { + result.push(sentence); + } else { + changes++; + } + } else { + // Always keep short sentences + result.push(sentence); + } + } + + return { text: result.join(''), changes }; + } + + /** + * Compress enumeration patterns (A,B,A,B... -> A,B) + */ + private compressEnumerations(text: string): { text: string; changes: number } { + const config = CLEANING_CONFIG.repetition.enumerationDetection; + let changes = 0; + + // Split by sentences and process each + const sentences = this.splitIntoSentences(text); + const result = sentences.map(sentence => { + // Detect common separators + const separators = [',', ';', '、', '·', '\t', /\s{2,}/]; + + for (const sep of separators) { + const sepRegex = sep instanceof RegExp ? sep : new RegExp(`\\s*${this.escapeRegex(sep)}\\s*`); + + // Check if sentence contains this separator + if (sepRegex.test(sentence)) { + const parts = sentence.split(sepRegex); + + if (parts.length >= config.minRepeatCount * 2) { + const compressed = this.detectAndCompressPattern(parts, config.minRepeatCount); + if (compressed.changed) { + changes++; + const sepStr = sep instanceof RegExp ? ' ' : (sep === '、' ? '、' : `${sep} `); + return compressed.parts.join(sepStr) + + (sentence.match(/[。.!?!?]+$/) ? sentence.match(/[。.!?!?]+$/)?.[0] || '' : ''); + } + } + } + } + + return sentence; + }); + + return { text: result.join(''), changes }; + } + + /** + * Suppress paragraph repetitions using fingerprinting + */ + private suppressParagraphRepetitions(text: string): { text: string; changes: number } { + const config = CLEANING_CONFIG.repetition.paragraphRepeat; + let changes = 0; + + const paragraphs = text.split(/\n\s*\n/); + const seenFingerprints = new Set(); + const result: string[] = []; + + for (const paragraph of paragraphs) { + const trimmed = paragraph.trim(); + if (trimmed.length === 0) continue; + + // Create fingerprint from first N characters + const fingerprint = this.normalizeSentence(trimmed.slice(0, config.headChars)); + + if (!seenFingerprints.has(fingerprint)) { + seenFingerprints.add(fingerprint); + result.push(paragraph); + } else { + changes++; + } + } + + return { text: result.join('\n\n'), changes }; + } + + /** + * Suppress tail repetitions (high self-repetition density at end) + */ + private suppressTailRepetitions(text: string): { text: string; changes: number } { + let changes = 0; + + // Analyze last 400 characters for repetition density + const tailLength = Math.min(400, text.length); + if (tailLength < 80) return { text, changes }; + + const tail = text.slice(-tailLength); + const diversity = this.calculateLexicalDiversity(tail); + const repetitionDensity = this.calculateRepetitionDensity(tail); + + // If tail has low diversity or high repetition, cut back to last sentence + if (diversity < 0.3 || repetitionDensity >= 2) { + const cutPoint = text.length - tailLength; + const beforeTail = text.slice(0, cutPoint); + + // Find last sentence boundary + const sentenceEnds = ['.', '。', '!', '!', '?', '?']; + let lastSentenceEnd = -1; + + for (let i = beforeTail.length - 1; i >= 0; i--) { + if (sentenceEnds.includes(beforeTail[i])) { + lastSentenceEnd = i; + break; + } + } + + if (lastSentenceEnd > 0) { + changes++; + return { text: beforeTail.slice(0, lastSentenceEnd + 1), changes }; + } + } + + return { text, changes }; + } + + /** + * Final cleanup of the text + */ + private finalCleanup(text: string): string { + return text + .replace(/\uFFFD+/g, '') // Remove replacement characters + .replace(/\n{3,}/g, '\n\n') // Normalize line breaks + .trim(); + } + + // Helper methods + + private tokenizeText(text: string): string[] { + // Universal tokenization: split on whitespace and punctuation + // Handle CJK characters as individual tokens when needed + return Array.from(text.matchAll(/\p{L}+|\p{N}+|\p{P}+|\s+/gu), m => m[0]); + } + + private normalizeToken(token: string): string { + return token.normalize('NFKC').toLowerCase().trim(); + } + + private isSignificantToken(token: string): boolean { + // Skip pure punctuation, whitespace, and very short tokens + return token.length > 0 && + !/^\p{P}+$|^\s+$|^[、。.!?!?…]+$/u.test(token) && + token.length >= 2; + } + + private splitIntoSentences(text: string): string[] { + // Universal sentence splitting using common punctuation + return text.split(/(?<=[。.!?!?])\s*/); + } + + private normalizeSentence(sentence: string): string { + return sentence + .replace(/[、。,.;:!!??\s]/g, '') + .normalize('NFKC') + .toLowerCase(); + } + + private escapeRegex(str: string): string { + return str.replace(/[-/\\^$*+?.()|[\]{}]/g, '\\$&'); + } + + private detectAndCompressPattern(parts: string[], minRepeat: number): { parts: string[]; changed: boolean } { + // Try different pattern lengths + for (let patternLength = 2; patternLength <= Math.floor(parts.length / minRepeat); patternLength++) { + const pattern = parts.slice(0, patternLength); + let matches = 1; + + // Check how many times this pattern repeats + for (let i = patternLength; i < parts.length; i += patternLength) { + const candidate = parts.slice(i, i + patternLength); + if (this.arraysEqual(pattern, candidate)) { + matches++; + } else { + break; + } + } + + if (matches >= minRepeat) { + return { parts: pattern, changed: true }; + } + } + + return { parts, changed: false }; + } + + private arraysEqual(a: string[], b: string[]): boolean { + return a.length === b.length && a.every((val, i) => val === b[i]); + } + + private calculateLexicalDiversity(text: string): number { + const tokens = this.tokenizeText(text).filter(t => this.isSignificantToken(this.normalizeToken(t))); + const uniqueTokens = new Set(tokens.map(t => this.normalizeToken(t))); + return tokens.length > 0 ? uniqueTokens.size / tokens.length : 1; + } + + private calculateRepetitionDensity(text: string): number { + // Count overlapping repetitive patterns + const repetitions = (text.match(/(.{2,20})\1{2,}/gs) || []).length; + return repetitions; + } +} \ No newline at end of file diff --git a/src/core/transcription/cleaning/interfaces.ts b/src/core/transcription/cleaning/interfaces.ts new file mode 100644 index 0000000..d140fe8 --- /dev/null +++ b/src/core/transcription/cleaning/interfaces.ts @@ -0,0 +1,117 @@ +/** + * Core interfaces for the text cleaning pipeline + * Provides contracts for cleaners, results, and pipeline execution + */ + +/** + * Result of a text cleaning operation + */ +export interface CleaningResult { + /** The cleaned text */ + cleanedText: string; + /** List of issues or warnings encountered during cleaning */ + issues: string[]; + /** Whether significant changes were made */ + hasSignificantChanges: boolean; + /** Additional metadata about the cleaning operation */ + metadata?: Record; +} + +/** + * Context information for cleaning operations + */ +export interface CleaningContext { + /** Language code ('auto' for auto-detection) */ + language: string; + /** Original text length before any cleaning */ + originalLength: number; + /** Enable detailed logging for debugging */ + enableDetailedLogging?: boolean; + /** Original prompt used for transcription (if any) */ + originalPrompt?: string; + /** Processing start time for performance tracking */ + startTime?: number; +} + +/** + * Interface for text cleaning implementations + */ +export interface TextCleaner { + /** Unique name for this cleaner */ + readonly name: string; + /** Whether this cleaner is enabled */ + readonly enabled: boolean; + + /** + * Clean the given text + * @param text Text to clean + * @param language Language code + * @param context Additional context for cleaning + * @returns Cleaning result (can be async) + */ + clean(text: string, language: string, context?: CleaningContext): Promise | CleaningResult; +} + +/** + * Interface for cleaning pipeline implementations + */ +export interface CleaningPipeline { + /** Unique name for this pipeline */ + readonly name: string; + + /** + * Execute the complete cleaning pipeline + * @param text Text to clean + * @param language Language code + * @param context Additional context for cleaning + * @returns Final cleaned text with metadata + */ + execute(text: string, language: string, context?: CleaningContext): Promise<{ + finalText: string; + metadata: { + totalOriginalLength: number; + totalFinalLength: number; + totalReductionRatio: number; + cleanerResults?: Array<{ + cleanerName: string; + reductionRatio: number; + processingTime: number; + issues: string[]; + }>; + }; + }>; +} + +/** + * Safety check result for cleaning operations + */ +export interface SafetyCheckResult { + /** Whether the operation is safe to proceed */ + isSafe: boolean; + /** Reason for safety failure (if any) */ + reason?: string; + /** Suggested action */ + action: 'proceed' | 'skip' | 'rollback'; + /** Calculated reduction ratio */ + reductionRatio: number; +} + +/** + * Performance metrics for cleaning operations + */ +export interface CleaningMetrics { + /** Cleaner name */ + cleanerName: string; + /** Processing time in milliseconds */ + processingTime: number; + /** Original text length */ + originalLength: number; + /** Final text length */ + finalLength: number; + /** Reduction ratio (0.0-1.0) */ + reductionRatio: number; + /** Number of patterns matched */ + patternsMatched: number; + /** Issues encountered */ + issues: string[]; +} \ No newline at end of file From 641054f034d4cefe740081fd90897c6231f9abe5 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Thu, 14 Aug 2025 16:51:04 +0000 Subject: [PATCH 3/5] Fix XML tag extraction and improve cleaning pipeline robustness Co-authored-by: mssoftjp <115339791+mssoftjp@users.noreply.github.com> --- .../cleaning/PromptContaminationCleaner.ts | 83 ++++++++++--------- .../cleaning/StandardCleaningPipeline.ts | 2 +- .../cleaning/UniversalRepetitionCleaner.ts | 3 +- 3 files changed, 45 insertions(+), 43 deletions(-) diff --git a/src/core/transcription/cleaning/PromptContaminationCleaner.ts b/src/core/transcription/cleaning/PromptContaminationCleaner.ts index 42956e1..76b1db6 100644 --- a/src/core/transcription/cleaning/PromptContaminationCleaner.ts +++ b/src/core/transcription/cleaning/PromptContaminationCleaner.ts @@ -115,37 +115,27 @@ export class PromptContaminationCleaner implements TextCleaner { private removeXmlTags(text: string, xmlPatternGroups: typeof CLEANING_CONFIG.contamination.xmlPatternGroups): string { let cleaned = text; - // Process in priority order: complete tags first, then partial cleanup - const groups = ['completeXmlTags', 'sentenceBoundedTags', 'lineBoundedTags', 'standaloneTags'] as const; - - for (const group of groups) { - const patterns = xmlPatternGroups[group]; - for (const pattern of patterns) { - try { - // Extract regex pattern and flags from string representation - const lastSlash = pattern.lastIndexOf('/'); - if (lastSlash > 0) { - const regexPattern = pattern.slice(1, lastSlash); - const flags = pattern.slice(lastSlash + 1); - const regex = new RegExp(regexPattern, flags); - - if (group === 'completeXmlTags') { - // For complete XML tags, extract content instead of removing - const match = cleaned.match(regex); - if (match && match[1]) { - cleaned = match[1].trim(); - } - } else { - // For other patterns, remove entirely - cleaned = cleaned.replace(regex, ''); - } - } - } catch (error) { - this.logger.warn(`Invalid regex pattern in ${group}`, { pattern, error }); - } + // First: Extract content from complete TRANSCRIPT tags (highest priority) + const completeTagMatch = cleaned.match(/]*>\s*([\s\S]*?)\s*<\/TRANSCRIPT>/); + if (completeTagMatch) { + cleaned = completeTagMatch[1].trim(); + } else { + // Second: Handle incomplete TRANSCRIPT tags (missing closing tag) + const openingMatch = cleaned.match(/]*>\s*([\s\S]*)/); + if (openingMatch) { + cleaned = openingMatch[1].trim(); } } + // Third: Remove any remaining XML-like tags + cleaned = cleaned.replace(/<\/?TRANSCRIPT[^>]*>/g, ''); + cleaned = cleaned.replace(/<\/?transcript[^>]*>/gi, ''); + cleaned = cleaned.replace(/<\/?TRANSCRIPTION[^>]*>/gi, ''); + + // Remove standalone tags and empty XML tags + cleaned = cleaned.replace(/<[^>]*\/>/g, ''); + cleaned = cleaned.replace(/<\w+[^>]*>\s*<\/\w+>/g, ''); + return cleaned; } @@ -202,19 +192,30 @@ export class PromptContaminationCleaner implements TextCleaner { private removeContextPatterns(text: string, contextPatterns: string[]): string { let cleaned = text; - for (const pattern of contextPatterns) { - try { - // Extract regex pattern and flags from string representation - const lastSlash = pattern.lastIndexOf('/'); - if (lastSlash > 0) { - const regexPattern = pattern.slice(1, lastSlash); - const flags = pattern.slice(lastSlash + 1); - const regex = new RegExp(regexPattern, flags); - cleaned = cleaned.replace(regex, ''); - } - } catch (error) { - this.logger.warn('Invalid context pattern regex', { pattern, error }); - } + // Direct pattern matching for speaker-only phrases + const speakerPatterns = [ + /\(Speaker content only\)/gi, + /\(SPEAKER CONTENT ONLY\)/gi, + /(話者の発言のみ)/g, + /(仅说话者内容)/g, + /(화자 발언만)/g + ]; + + for (const pattern of speakerPatterns) { + cleaned = cleaned.replace(pattern, ''); + } + + // Remove format-only lines + const formatPatterns = [ + /^Output format:\s*$/gm, + /^Format:\s*$/gm, + /^出力形式:\s*$/gm, + /^输出格式:\s*$/gm, + /^출력 형식:\s*$/gm + ]; + + for (const pattern of formatPatterns) { + cleaned = cleaned.replace(pattern, ''); } return cleaned; diff --git a/src/core/transcription/cleaning/StandardCleaningPipeline.ts b/src/core/transcription/cleaning/StandardCleaningPipeline.ts index 94d997a..a0c5985 100644 --- a/src/core/transcription/cleaning/StandardCleaningPipeline.ts +++ b/src/core/transcription/cleaning/StandardCleaningPipeline.ts @@ -241,6 +241,6 @@ export class StandardCleaningPipeline implements CleaningPipeline { * Get all cleaners in the pipeline */ getCleaners(): readonly TextCleaner[] { - return [...this.cleaners]; + return Object.freeze([...this.cleaners]); } } \ No newline at end of file diff --git a/src/core/transcription/cleaning/UniversalRepetitionCleaner.ts b/src/core/transcription/cleaning/UniversalRepetitionCleaner.ts index c1b7b29..1cb8ca1 100644 --- a/src/core/transcription/cleaning/UniversalRepetitionCleaner.ts +++ b/src/core/transcription/cleaning/UniversalRepetitionCleaner.ts @@ -128,6 +128,7 @@ export class UniversalRepetitionCleaner implements TextCleaner { // Define repetition patterns with their limits const patterns = [ { pattern: /([.!?])\1{5,}/g, replacement: '$1$1$1', description: 'punctuation' }, + { pattern: /([。!?])\1{3,}/g, replacement: '$1', description: 'cjk-punctuation' }, { pattern: /[…]{3,}/g, replacement: '…', description: 'ellipsis' }, { pattern: /[-—–]{6,}/g, replacement: '—', description: 'dashes' }, { pattern: /[•·・]{6,}/g, replacement: '・', description: 'bullets' }, @@ -363,7 +364,7 @@ export class UniversalRepetitionCleaner implements TextCleaner { } private splitIntoSentences(text: string): string[] { - // Universal sentence splitting using common punctuation + // Universal sentence splitting using common punctuation, preserving the punctuation return text.split(/(?<=[。.!?!?])\s*/); } From 74255f12631cc01382ca2613ec3695861262892a Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Fri, 15 Aug 2025 03:18:12 +0000 Subject: [PATCH 4/5] Clean up unused configuration items and fix xmlPatternGroups usage Co-authored-by: mssoftjp <115339791+mssoftjp@users.noreply.github.com> --- src/config/CleaningConfig.ts | 29 ------------------- .../cleaning/PromptContaminationCleaner.ts | 8 ++--- 2 files changed, 4 insertions(+), 33 deletions(-) diff --git a/src/config/CleaningConfig.ts b/src/config/CleaningConfig.ts index 6488861..3d0913d 100644 --- a/src/config/CleaningConfig.ts +++ b/src/config/CleaningConfig.ts @@ -7,12 +7,6 @@ export interface SafetyThresholds { /** Maximum reduction allowed by a single cleaner (0.0-1.0) */ singleCleanerMaxReduction: number; - /** Maximum reduction allowed for a single pattern match (0.0-1.0) */ - singlePatternMaxReduction: number; - /** Maximum reduction allowed for repetition patterns (0.0-1.0) */ - repetitionPatternMaxReduction: number; - /** Maximum reduction allowed per iteration (0.0-1.0) */ - iterationReductionLimit: number; /** Emergency fallback threshold - rollback if exceeded (0.0-1.0) */ emergencyFallbackThreshold: number; /** Warning threshold for logging (0.0-1.0) */ @@ -36,12 +30,6 @@ export interface RepetitionThresholds { minimumSentenceLengthForSimilarity: number; /** Maximum consecutive newlines to allow */ consecutiveNewlineLimit: number; - /** N-gram configuration for phrase repetition detection */ - ngram: { - min: number; - max: number; - thresholds: Array<{ n: number; repeat: number }>; - }; /** Enumeration detection configuration */ enumerationDetection: { enabled: boolean; @@ -83,9 +71,6 @@ export interface CleaningConfig { export const CLEANING_CONFIG: CleaningConfig = { safety: { singleCleanerMaxReduction: 0.3, - singlePatternMaxReduction: 0.15, - repetitionPatternMaxReduction: 0.25, - iterationReductionLimit: 0.2, emergencyFallbackThreshold: 0.5, warningThreshold: 0.15 }, @@ -99,20 +84,6 @@ export const CLEANING_CONFIG: CleaningConfig = { similarityThreshold: 0.85, minimumSentenceLengthForSimilarity: 10, consecutiveNewlineLimit: 3, - ngram: { - min: 3, - max: 10, - thresholds: [ - { n: 3, repeat: 4 }, - { n: 4, repeat: 3 }, - { n: 5, repeat: 3 }, - { n: 6, repeat: 2 }, - { n: 7, repeat: 2 }, - { n: 8, repeat: 2 }, - { n: 9, repeat: 2 }, - { n: 10, repeat: 2 } - ] - }, enumerationDetection: { enabled: true, minRepeatCount: 3 diff --git a/src/core/transcription/cleaning/PromptContaminationCleaner.ts b/src/core/transcription/cleaning/PromptContaminationCleaner.ts index 76b1db6..da5c883 100644 --- a/src/core/transcription/cleaning/PromptContaminationCleaner.ts +++ b/src/core/transcription/cleaning/PromptContaminationCleaner.ts @@ -42,11 +42,11 @@ export class PromptContaminationCleaner implements TextCleaner { const issues: string[] = []; let patternsMatched = 0; - const { instructionPatterns, xmlPatternGroups, contextPatterns, promptSnippetLengths } = + const { instructionPatterns, contextPatterns, promptSnippetLengths } = CLEANING_CONFIG.contamination; // Step 1: Handle XML tags (highest priority) - cleaned = this.removeXmlTags(cleaned, xmlPatternGroups); + cleaned = this.removeXmlTags(cleaned); if (cleaned !== text) { patternsMatched++; this.logger.debug('XML tags removed'); @@ -110,9 +110,9 @@ export class PromptContaminationCleaner implements TextCleaner { } /** - * Remove XML tags in priority order + * Remove XML tags using fixed patterns for TRANSCRIPT tags */ - private removeXmlTags(text: string, xmlPatternGroups: typeof CLEANING_CONFIG.contamination.xmlPatternGroups): string { + private removeXmlTags(text: string): string { let cleaned = text; // First: Extract content from complete TRANSCRIPT tags (highest priority) From 4fb0ba7ca73d17c9ecd7ca02948ee4db09246309 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Fri, 15 Aug 2025 06:55:43 +0000 Subject: [PATCH 5/5] Remove unused xmlPatternGroups configuration for consistency Co-authored-by: mssoftjp <115339791+mssoftjp@users.noreply.github.com> --- src/config/CleaningConfig.ts | 28 ---------------------------- 1 file changed, 28 deletions(-) diff --git a/src/config/CleaningConfig.ts b/src/config/CleaningConfig.ts index 3d0913d..4cbc51e 100644 --- a/src/config/CleaningConfig.ts +++ b/src/config/CleaningConfig.ts @@ -45,13 +45,6 @@ export interface RepetitionThresholds { export interface ContaminationPatterns { /** Instruction patterns to remove from text beginning */ instructionPatterns: string[]; - /** XML pattern groups for different contexts */ - xmlPatternGroups: { - completeXmlTags: string[]; - sentenceBoundedTags: string[]; - lineBoundedTags: string[]; - standaloneTags: string[]; - }; /** Context patterns for general cleanup */ contextPatterns: string[]; /** Prompt snippet lengths for partial matching */ @@ -113,27 +106,6 @@ export const CLEANING_CONFIG: CleaningConfig = { '화자의 발언 내용만 정확히 기록해주세요' ], - xmlPatternGroups: { - completeXmlTags: [ - '/]*>([\\s\\S]*?)<\\/TRANSCRIPT>/g', - '/]*>([\\s\\S]*?)<\\/transcript>/g', - '/]*>([\\s\\S]*?)<\\/TRANSCRIPTION>/g' - ], - sentenceBoundedTags: [ - '/<\\/?TRANSCRIPT[^>]*>/g', - '/<\\/?transcript[^>]*>/g', - '/<\\/?TRANSCRIPTION[^>]*>/g' - ], - lineBoundedTags: [ - '/^\\s*<[^>]*>\\s*$/gm', - '/^\\s*<\\/[^>]*>\\s*$/gm' - ], - standaloneTags: [ - '/<[^>]*\\/>/g', - '/<\\w+[^>]*>\\s*<\\/\\w+>/g' - ] - }, - contextPatterns: [ // Universal context patterns (structural) '/^\\s*\\([^)]*Speaker[^)]*only[^)]*\\)\\s*$/gmi',