fix: preserve a codepoint spanning the head|tail split (codex round 1)
When headTail budgets cover the whole stream, omitted is 0 and the two retained halves are contiguous — the split is artificial and a multibyte codepoint can straddle it. finish() now decodes the contiguous buffer as one in that case; the per-side UTF-8 boundary trims and separate decoding apply only when a real middle gap exists. Without this, a headTail retainer could drop a character while reporting truncated:false.
This commit is contained in:
@@ -391,21 +391,22 @@ export class TextRetainer {
|
||||
const omitted = this.omittedAt(this.total)
|
||||
const truncated = omitted > 0
|
||||
|
||||
// A cut exists at the prefix end only if content followed it (moved to the
|
||||
// suffix or omitted); likewise the suffix start is a cut only if content
|
||||
// preceded it. When the whole stream fits in one side, pass bytes through
|
||||
// untrimmed so valid output is never altered.
|
||||
let prefix = concat(this.prefixChunks)
|
||||
if (suffixLen > 0 || omitted > 0) prefix = trimTrailingPartialUtf8(prefix)
|
||||
const prefix = concat(this.prefixChunks) // exactly prefixLen bytes (prefixHeld === prefixLen)
|
||||
const suffix = concat(this.suffixChunks).subarray(this.suffixHeld - suffixLen)
|
||||
|
||||
const suffixBuf = concat(this.suffixChunks)
|
||||
let suffix = suffixBuf.subarray(this.suffixHeld - suffixLen)
|
||||
if (prefixLen > 0 || omitted > 0) suffix = trimLeadingContinuationUtf8(suffix)
|
||||
// With nothing omitted, prefix and suffix are ADJACENT slices of one stream
|
||||
// (prefixLen + suffixLen === total), so the head|tail split is artificial: a
|
||||
// codepoint may span it. Decode the contiguous whole as one buffer — trimming
|
||||
// or decoding the halves separately here would corrupt a boundary-spanning
|
||||
// codepoint though no content was actually dropped. Only a real omitted gap
|
||||
// makes each side a true cut: trim each to a UTF-8 boundary and decode
|
||||
// separately so a codepoint is never reconstructed across the gap.
|
||||
const text = truncated
|
||||
? decoder.decode(trimTrailingPartialUtf8(prefix)) + decoder.decode(trimLeadingContinuationUtf8(suffix))
|
||||
: decoder.decode(concat([prefix, suffix]))
|
||||
|
||||
return {
|
||||
// Decode the two sides separately so a codepoint is never reconstructed
|
||||
// across the omitted middle.
|
||||
text: decoder.decode(prefix) + decoder.decode(suffix),
|
||||
text,
|
||||
truncated,
|
||||
omittedBytes: truncated
|
||||
? { kind: this.allowStop ? 'atLeast' : 'exact', count: omitted }
|
||||
|
||||
@@ -156,6 +156,32 @@ describe('TextRetainer — headTail (prefix + suffix, omit the middle)', () => {
|
||||
expect(result.truncated).toBe(false)
|
||||
expect(result.omittedBytes).toEqual<Omitted>({ kind: 'none' })
|
||||
})
|
||||
|
||||
it('does not drop a codepoint that spans the head|tail split when nothing is omitted', () => {
|
||||
// Regression: with head+tail covering the whole stream, the split is
|
||||
// artificial — a multibyte codepoint may straddle it. 'éab' is C3 A9 61 62
|
||||
// (4 bytes); headBytes 1 + tailBytes 3 covers all 4 with omitted === 0, but
|
||||
// the split falls INSIDE 'é'. The bytes are contiguous, so the full 'éab'
|
||||
// must survive — not be trimmed to 'ab'.
|
||||
const r = new TextRetainer({ kind: 'headTail', headBytes: 1, tailBytes: 3 })
|
||||
r.push('éab')
|
||||
const result = r.finish()
|
||||
expect(result.text).toBe('éab')
|
||||
expect(result.truncated).toBe(false)
|
||||
expect(result.omittedBytes).toEqual<Omitted>({ kind: 'none' })
|
||||
})
|
||||
|
||||
it('still trims boundary partials once a real middle is omitted', () => {
|
||||
// With a genuine gap the two sides ARE true cuts: '€' (3 bytes) split across
|
||||
// the omitted middle must not resurface as a replacement char on either side.
|
||||
const r = new TextRetainer({ kind: 'headTail', headBytes: 2, tailBytes: 2 })
|
||||
r.push('a€€b') // 8 bytes; head 'a'+partial, tail partial+'b', middle omitted
|
||||
const result = r.finish()
|
||||
expect(result.truncated).toBe(true)
|
||||
expect(result.text).not.toContain('�')
|
||||
expect(result.text.startsWith('a')).toBe(true)
|
||||
expect(result.text.endsWith('b')).toBe(true)
|
||||
})
|
||||
})
|
||||
|
||||
describe('TextRetainer — zero budgets', () => {
|
||||
|
||||
Reference in New Issue
Block a user