refactor(nlp): enhance EPUB text splitting for oversized sentences and remove unused functions
- Add logic to split oversized sentences in `splitTextToTtsBlocksEPUB` using `splitOversizedText` to ensure blocks stay within `MAX_BLOCK_LENGTH` - Remove `extractRawSentences` and `processTextWithMapping` functions as they are no longer needed - Update tests to reflect the changes, including a new test for oversized sentence splitting and removal of related test suites
This commit is contained in:
parent
efe6ffec86
commit
54b9691524
2 changed files with 21 additions and 108 deletions
|
|
@ -203,14 +203,20 @@ export const splitTextToTtsBlocksEPUB = (text: string): string[] => {
|
||||||
|
|
||||||
for (const sentence of mergedSentences) {
|
for (const sentence of mergedSentences) {
|
||||||
const trimmedSentence = sentence.trim();
|
const trimmedSentence = sentence.trim();
|
||||||
|
const sentenceParts =
|
||||||
|
trimmedSentence.length > MAX_BLOCK_LENGTH
|
||||||
|
? splitOversizedText(trimmedSentence, MAX_BLOCK_LENGTH)
|
||||||
|
: [trimmedSentence];
|
||||||
|
|
||||||
if (currentBlock && (currentBlock.length + trimmedSentence.length + 1) > MAX_BLOCK_LENGTH) {
|
for (const sentencePart of sentenceParts) {
|
||||||
blocks.push(currentBlock.trim());
|
if (currentBlock && (currentBlock.length + sentencePart.length + 1) > MAX_BLOCK_LENGTH) {
|
||||||
currentBlock = trimmedSentence;
|
blocks.push(currentBlock.trim());
|
||||||
} else {
|
currentBlock = sentencePart;
|
||||||
currentBlock = currentBlock
|
} else {
|
||||||
? `${currentBlock} ${trimmedSentence}`
|
currentBlock = currentBlock
|
||||||
: trimmedSentence;
|
? `${currentBlock} ${sentencePart}`
|
||||||
|
: sentencePart;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
@ -232,72 +238,6 @@ export const splitTextToTtsBlocksEPUB = (text: string): string[] => {
|
||||||
export const normalizeTextForTts = (text: string): string =>
|
export const normalizeTextForTts = (text: string): string =>
|
||||||
splitTextToTtsBlocks(text).join(' ');
|
splitTextToTtsBlocks(text).join(' ');
|
||||||
|
|
||||||
/**
|
|
||||||
* Extracts raw sentence strings from text without preprocessing or block grouping.
|
|
||||||
* Useful for text matching and highlighting.
|
|
||||||
*
|
|
||||||
* @param {string} text - The text to extract sentences from
|
|
||||||
* @returns {string[]} Array of raw sentences
|
|
||||||
*/
|
|
||||||
export const extractRawSentences = (text: string): string[] => {
|
|
||||||
if (!text || text.length < 1) {
|
|
||||||
return [];
|
|
||||||
}
|
|
||||||
|
|
||||||
return nlp(text).sentences().out('array') as string[];
|
|
||||||
};
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Enhanced sentence processing that returns both processed sentences and raw sentences
|
|
||||||
* This allows for better mapping between the two for click-to-highlight functionality
|
|
||||||
*
|
|
||||||
* @param {string} text - The text to process
|
|
||||||
* @returns {Object} Object containing processed sentences and raw sentences with mapping
|
|
||||||
*/
|
|
||||||
export const processTextWithMapping = (text: string): {
|
|
||||||
processedSentences: string[];
|
|
||||||
rawSentences: string[];
|
|
||||||
sentenceMapping: Array<{ processedIndex: number; rawIndices: number[] }>;
|
|
||||||
} => {
|
|
||||||
const rawSentences = extractRawSentences(text);
|
|
||||||
const processedSentences = splitTextToTtsBlocks(text);
|
|
||||||
|
|
||||||
// Create a mapping between processed sentences and raw sentences
|
|
||||||
const sentenceMapping: Array<{ processedIndex: number; rawIndices: number[] }> = [];
|
|
||||||
|
|
||||||
// For simple mapping, we'll track which raw sentences contributed to each processed sentence
|
|
||||||
let rawIndex = 0;
|
|
||||||
|
|
||||||
for (let processedIndex = 0; processedIndex < processedSentences.length; processedIndex++) {
|
|
||||||
const processedSentence = processedSentences[processedIndex];
|
|
||||||
const rawIndices: number[] = [];
|
|
||||||
|
|
||||||
// Find which raw sentences are contained in this processed sentence
|
|
||||||
const remainingText = processedSentence;
|
|
||||||
|
|
||||||
while (rawIndex < rawSentences.length && remainingText.length > 0) {
|
|
||||||
const rawSentence = rawSentences[rawIndex];
|
|
||||||
const cleanedRawSentence = preprocessSentenceForAudio(rawSentence);
|
|
||||||
|
|
||||||
if (remainingText.includes(cleanedRawSentence) || cleanedRawSentence.includes(remainingText)) {
|
|
||||||
rawIndices.push(rawIndex);
|
|
||||||
rawIndex++;
|
|
||||||
break;
|
|
||||||
} else {
|
|
||||||
rawIndex++;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
|
|
||||||
sentenceMapping.push({ processedIndex, rawIndices });
|
|
||||||
}
|
|
||||||
|
|
||||||
return {
|
|
||||||
processedSentences,
|
|
||||||
rawSentences,
|
|
||||||
sentenceMapping
|
|
||||||
};
|
|
||||||
};
|
|
||||||
|
|
||||||
// Helper functions to merge quoted dialogue across sentences
|
// Helper functions to merge quoted dialogue across sentences
|
||||||
const countDoubleQuotes = (s: string): number => {
|
const countDoubleQuotes = (s: string): number => {
|
||||||
const matches = s.match(/["“”]/g);
|
const matches = s.match(/["“”]/g);
|
||||||
|
|
|
||||||
|
|
@ -4,8 +4,6 @@ import {
|
||||||
splitTextToTtsBlocks,
|
splitTextToTtsBlocks,
|
||||||
splitTextToTtsBlocksEPUB,
|
splitTextToTtsBlocksEPUB,
|
||||||
normalizeTextForTts,
|
normalizeTextForTts,
|
||||||
extractRawSentences,
|
|
||||||
processTextWithMapping,
|
|
||||||
MAX_BLOCK_LENGTH
|
MAX_BLOCK_LENGTH
|
||||||
} from '../../src/lib/nlp';
|
} from '../../src/lib/nlp';
|
||||||
|
|
||||||
|
|
@ -169,6 +167,14 @@ test.describe('splitTextToTtsBlocksEPUB (highlight-friendly)', () => {
|
||||||
expect(result[0]).toBe('One.');
|
expect(result[0]).toBe('One.');
|
||||||
expect(result[1]).toBe('Two.');
|
expect(result[1]).toBe('Two.');
|
||||||
});
|
});
|
||||||
|
|
||||||
|
test('splits oversized sentences to keep blocks bounded', () => {
|
||||||
|
const input = Array(1200).fill('word').join(' '); // no punctuation; guaranteed to exceed MAX_BLOCK_LENGTH
|
||||||
|
const result = splitTextToTtsBlocksEPUB(input);
|
||||||
|
|
||||||
|
expect(result.length).toBeGreaterThan(1);
|
||||||
|
expectNormalizedBlocks(result, MAX_BLOCK_LENGTH);
|
||||||
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
test.describe('normalizeTextForTts', () => {
|
test.describe('normalizeTextForTts', () => {
|
||||||
|
|
@ -180,36 +186,3 @@ test.describe('normalizeTextForTts', () => {
|
||||||
expect(normalized.length).toBeGreaterThan(0);
|
expect(normalized.length).toBeGreaterThan(0);
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
test.describe('extractRawSentences', () => {
|
|
||||||
test('returns [] for empty input', () => {
|
|
||||||
expect(extractRawSentences('')).toEqual([]);
|
|
||||||
});
|
|
||||||
|
|
||||||
test('returns sentence-like strings without preprocessing', () => {
|
|
||||||
const input = 'First sentence. Second sentence.';
|
|
||||||
const result = extractRawSentences(input);
|
|
||||||
expect(result.length).toBeGreaterThanOrEqual(2);
|
|
||||||
expect(result[0]).toContain('First');
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
test.describe('processTextWithMapping', () => {
|
|
||||||
test('returns mapping entries with valid indices', () => {
|
|
||||||
const text = 'First (1). Second (2).';
|
|
||||||
const { processedSentences, rawSentences, sentenceMapping } = processTextWithMapping(text);
|
|
||||||
|
|
||||||
expect(processedSentences.length).toBeGreaterThan(0);
|
|
||||||
expect(rawSentences.length).toBeGreaterThan(0);
|
|
||||||
expect(sentenceMapping).toHaveLength(processedSentences.length);
|
|
||||||
|
|
||||||
for (let i = 0; i < sentenceMapping.length; i++) {
|
|
||||||
const entry = sentenceMapping[i];
|
|
||||||
expect(entry.processedIndex).toBe(i);
|
|
||||||
for (const rawIndex of entry.rawIndices) {
|
|
||||||
expect(rawIndex).toBeGreaterThanOrEqual(0);
|
|
||||||
expect(rawIndex).toBeLessThan(rawSentences.length);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
|
||||||
Loading…
Reference in a new issue