mirror of
https://github.com/MoonshotAI/kimi-code.git
synced 2026-08-19 05:35:34 +00:00
fix(agent-core-v2): unify text/binary classification for UTF-8 multibyte files (#2972)
This commit is contained in:
parent
44a6c70e66
commit
04d23e2dab
8 changed files with 207 additions and 42 deletions
5
.changeset/fix-utf8-text-binary-detection.md
Normal file
5
.changeset/fix-utf8-text-binary-detection.md
Normal file
|
|
@ -0,0 +1,5 @@
|
|||
---
|
||||
"@moonshot-ai/kimi-code": patch
|
||||
---
|
||||
|
||||
Fix UTF-8 text files containing Chinese or emoji being misdetected as binary, so log files preview correctly in the web UI.
|
||||
|
|
@ -23,6 +23,13 @@
|
|||
|
||||
export type UtfTextEncoding = 'utf-8' | 'utf-16le' | 'utf-16be';
|
||||
|
||||
export interface TextClassification {
|
||||
readonly isBinary: boolean;
|
||||
readonly encoding: UtfTextEncoding;
|
||||
}
|
||||
|
||||
export const FS_BINARY_NONPRINTABLE_FRACTION = 0.3;
|
||||
|
||||
export interface TextEncodingDetection {
|
||||
/**
|
||||
* Detected encoding. `'utf-8'` when no signal points elsewhere (also the
|
||||
|
|
@ -51,16 +58,7 @@ const UTF16BE_BOM = [0xfe, 0xff] as const;
|
|||
const UTF16LE_BOM = [0xff, 0xfe] as const;
|
||||
const UTF8_BOM = [0xef, 0xbb, 0xbf] as const;
|
||||
|
||||
/**
|
||||
* Detect the encoding of a text file from its leading bytes.
|
||||
*
|
||||
* Known limitation inherited from the reference implementation: a BOM-less
|
||||
* UTF-16 file whose content carries no zero bytes at all (e.g. purely CJK
|
||||
* text) is reported as `'utf-8'`; strict UTF-8 decoding of it will then fail
|
||||
* or produce garbage. Notepad and most editors write a BOM, so this is rare
|
||||
* in practice.
|
||||
*/
|
||||
export function detectTextEncoding(sample: Uint8Array): TextEncodingDetection {
|
||||
function sniffTextEncoding(sample: Uint8Array): TextEncodingDetection {
|
||||
// Always trust a BOM first.
|
||||
if (sample.length >= 2) {
|
||||
const b0 = sample[0]!;
|
||||
|
|
@ -101,6 +99,67 @@ export function detectTextEncoding(sample: Uint8Array): TextEncodingDetection {
|
|||
return { encoding: 'utf-8', seemsBinary: true };
|
||||
}
|
||||
|
||||
export function classifyTextSample(sample: Uint8Array): TextClassification {
|
||||
const sniffed = sniffTextEncoding(sample);
|
||||
if (sniffed.seemsBinary || sniffed.encoding !== 'utf-8') {
|
||||
return { isBinary: sniffed.seemsBinary, encoding: sniffed.encoding };
|
||||
}
|
||||
if (sample.includes(0)) {
|
||||
return { isBinary: true, encoding: 'utf-8' };
|
||||
}
|
||||
let end = sample.length;
|
||||
for (let i = Math.max(0, sample.length - 3); i < sample.length; i++) {
|
||||
const b = sample[i]!;
|
||||
const expected =
|
||||
b >= 0xc2 && b <= 0xdf ? 2 : b >= 0xe0 && b <= 0xef ? 3 : b >= 0xf0 && b <= 0xf4 ? 4 : 0;
|
||||
if (expected === 0 || i + expected <= sample.length) continue;
|
||||
let validPrefix = true;
|
||||
for (let j = i + 1; j < sample.length; j++) {
|
||||
const cb = sample[j]!;
|
||||
if (cb < 0x80 || cb > 0xbf) {
|
||||
validPrefix = false;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (validPrefix) {
|
||||
end = i;
|
||||
break;
|
||||
}
|
||||
}
|
||||
let text: string;
|
||||
try {
|
||||
text = new TextDecoder('utf-8', { fatal: true }).decode(sample.subarray(0, end));
|
||||
} catch {
|
||||
return { isBinary: true, encoding: 'utf-8' };
|
||||
}
|
||||
let nonPrintable = 0;
|
||||
let total = 0;
|
||||
for (const ch of text) {
|
||||
const cp = ch.codePointAt(0)!;
|
||||
total++;
|
||||
if (cp === 9 || cp === 10 || cp === 13) continue;
|
||||
if (cp < 32 || (cp >= 0x7f && cp <= 0x9f)) nonPrintable++;
|
||||
}
|
||||
if (total > 0 && nonPrintable / total > FS_BINARY_NONPRINTABLE_FRACTION) {
|
||||
return { isBinary: true, encoding: 'utf-8' };
|
||||
}
|
||||
return { isBinary: false, encoding: 'utf-8' };
|
||||
}
|
||||
|
||||
/**
|
||||
* Detect the encoding of a text file from its leading bytes.
|
||||
*
|
||||
* Known limitation inherited from the reference implementation: a BOM-less
|
||||
* UTF-16 file whose content carries no zero bytes at all (e.g. purely CJK
|
||||
* text) is reported as `'utf-8'`; strict UTF-8 decoding of it will then fail
|
||||
* or produce garbage. Notepad and most editors write a BOM, so this is rare
|
||||
* in practice.
|
||||
*/
|
||||
export function detectTextEncoding(sample: Uint8Array): TextEncodingDetection {
|
||||
const classification = classifyTextSample(sample);
|
||||
return { encoding: classification.encoding, seemsBinary: classification.isBinary };
|
||||
}
|
||||
|
||||
/**
|
||||
* Decode bytes in a detected UTF encoding to a JS string. Malformed
|
||||
* sequences are replaced (non-fatal) and a leading BOM is stripped.
|
||||
|
|
|
|||
|
|
@ -11,8 +11,11 @@
|
|||
|
||||
import { extname } from 'node:path';
|
||||
|
||||
import { classifyTextSample } from '#/_base/text/encoding';
|
||||
|
||||
export { FS_BINARY_NONPRINTABLE_FRACTION } from '#/_base/text/encoding';
|
||||
|
||||
export const FS_BINARY_SAMPLE_BYTES = 4096;
|
||||
export const FS_BINARY_NONPRINTABLE_FRACTION = 0.3;
|
||||
|
||||
export interface FileMetaStat {
|
||||
readonly size: number;
|
||||
|
|
@ -21,16 +24,7 @@ export interface FileMetaStat {
|
|||
}
|
||||
|
||||
export function detectBinary(buf: Uint8Array): boolean {
|
||||
if (buf.length === 0) return false;
|
||||
let nonPrintable = 0;
|
||||
for (let i = 0; i < buf.length; i++) {
|
||||
const b = buf[i]!;
|
||||
if (b === 0) return true;
|
||||
if (b === 9 || b === 10 || b === 13) continue;
|
||||
if (b >= 32 && b <= 126) continue;
|
||||
nonPrintable++;
|
||||
}
|
||||
return nonPrintable / buf.length > FS_BINARY_NONPRINTABLE_FRACTION;
|
||||
return classifyTextSample(buf).isBinary;
|
||||
}
|
||||
|
||||
export function countLines(text: string): number {
|
||||
|
|
|
|||
|
|
@ -56,11 +56,10 @@ const FsWireErrorCode = {
|
|||
} as const;
|
||||
import ignore, { type Ignore } from 'ignore';
|
||||
|
||||
import { decodeUtfText, detectTextEncoding, type UtfTextEncoding } from '#/_base/text/encoding';
|
||||
import { classifyTextSample, decodeUtfText } from '#/_base/text/encoding';
|
||||
import {
|
||||
buildEtag,
|
||||
countLines,
|
||||
detectBinary,
|
||||
FS_BINARY_SAMPLE_BYTES,
|
||||
guessLanguageId,
|
||||
guessMime,
|
||||
|
|
@ -259,20 +258,14 @@ export class WorkspaceFsService implements IWorkspaceFsService {
|
|||
const sampleSize = Math.min(FS_BINARY_SAMPLE_BYTES, st.size);
|
||||
const sample =
|
||||
sampleSize === 0 ? new Uint8Array() : await this.hostFs.readBytes(abs, sampleSize);
|
||||
let isBinary = detectBinary(sample);
|
||||
|
||||
// Trust encoding detection over the binary heuristic: a binary-looking
|
||||
// sample can still be UTF-16 LE/BE text, and a BOM-marked UTF-16 file
|
||||
// may not look binary at all (CJK-only content carries no zero bytes).
|
||||
// Both are transcoded to UTF-8 so text clients can display them.
|
||||
let transcodeEncoding: UtfTextEncoding | undefined;
|
||||
if (req.encoding !== 'base64') {
|
||||
const detection = detectTextEncoding(sample);
|
||||
if (!detection.seemsBinary && detection.encoding !== 'utf-8') {
|
||||
transcodeEncoding = detection.encoding;
|
||||
isBinary = false;
|
||||
}
|
||||
}
|
||||
const classification = classifyTextSample(sample);
|
||||
const transcodeEncoding =
|
||||
!classification.isBinary && classification.encoding !== 'utf-8' && req.encoding !== 'base64'
|
||||
? classification.encoding
|
||||
: undefined;
|
||||
const isBinary =
|
||||
classification.isBinary ||
|
||||
(classification.encoding !== 'utf-8' && transcodeEncoding === undefined);
|
||||
|
||||
if (isBinary && req.encoding === 'utf-8') {
|
||||
throw new Error2(ErrorCodes.FS_IS_BINARY, `file is binary: ${req.path}`, {
|
||||
|
|
@ -448,7 +441,8 @@ export class WorkspaceFsService implements IWorkspaceFsService {
|
|||
const sampleSize = Math.min(FS_BINARY_SAMPLE_BYTES, st.size);
|
||||
const sample =
|
||||
sampleSize === 0 ? new Uint8Array() : await this.hostFs.readBytes(abs, sampleSize);
|
||||
const isBinary = detectBinary(sample);
|
||||
const classification = classifyTextSample(sample);
|
||||
const isBinary = classification.isBinary || classification.encoding !== 'utf-8';
|
||||
return {
|
||||
absolute: abs,
|
||||
relative: rel,
|
||||
|
|
|
|||
|
|
@ -1,6 +1,7 @@
|
|||
import { describe, expect, it } from 'vitest';
|
||||
|
||||
import {
|
||||
classifyTextSample,
|
||||
decodeUtfText,
|
||||
detectTextEncoding,
|
||||
ENCODING_DETECTION_SAMPLE_BYTES,
|
||||
|
|
@ -59,7 +60,7 @@ describe('detectTextEncoding', () => {
|
|||
it('limits the zero-byte heuristic to the leading sample window', () => {
|
||||
const sample = Buffer.alloc(ENCODING_DETECTION_SAMPLE_BYTES + 2, 0x61);
|
||||
sample[ENCODING_DETECTION_SAMPLE_BYTES + 1] = 0x00;
|
||||
expect(detectTextEncoding(sample)).toEqual({ encoding: 'utf-8', seemsBinary: false });
|
||||
expect(detectTextEncoding(sample)).toEqual({ encoding: 'utf-8', seemsBinary: true });
|
||||
});
|
||||
|
||||
it('flags zero bytes at both parities as binary', () => {
|
||||
|
|
@ -78,6 +79,75 @@ describe('detectTextEncoding', () => {
|
|||
});
|
||||
});
|
||||
|
||||
describe('classifyTextSample', () => {
|
||||
it('classifies UTF-8 multibyte text (CJK, emoji) as utf-8 text', () => {
|
||||
const sample = Buffer.from('2026-08-16 INFO 启动完成 ✅\n处理请求 🚀 成功\n'.repeat(20), 'utf8');
|
||||
expect(classifyTextSample(sample)).toEqual({ isBinary: false, encoding: 'utf-8' });
|
||||
});
|
||||
|
||||
it('classifies an empty sample as utf-8 text', () => {
|
||||
expect(classifyTextSample(new Uint8Array())).toEqual({ isBinary: false, encoding: 'utf-8' });
|
||||
});
|
||||
|
||||
it('classifies samples carrying NUL bytes as binary', () => {
|
||||
expect(
|
||||
classifyTextSample(Buffer.from([0x61, 0x62, 0x63, 0x00, 0x64, 0x65, 0x66])).isBinary,
|
||||
).toBe(true);
|
||||
expect(classifyTextSample(Buffer.from([0x00, 0x00, 0x61, 0x62])).isBinary).toBe(true);
|
||||
});
|
||||
|
||||
it('classifies control-char-heavy samples over the threshold as binary', () => {
|
||||
const sample = Buffer.concat([Buffer.alloc(40, 0x1b), Buffer.alloc(60, 0x61)]);
|
||||
expect(classifyTextSample(sample).isBinary).toBe(true);
|
||||
});
|
||||
|
||||
it('keeps ANSI-colored log lines under the control-char threshold as text', () => {
|
||||
const esc = String.fromCodePoint(0x1b);
|
||||
const sample = Buffer.from(`${esc}[32mINFO${esc}[0m 启动完成 ✅\n`.repeat(10), 'utf8');
|
||||
expect(classifyTextSample(sample)).toEqual({ isBinary: false, encoding: 'utf-8' });
|
||||
});
|
||||
|
||||
it('classifies invalid UTF-8 without UTF-16 features as binary', () => {
|
||||
expect(classifyTextSample(Buffer.from([0xd6, 0xd0, 0xc4, 0xe3, 0x31, 0x32]))).toEqual({
|
||||
isBinary: true,
|
||||
encoding: 'utf-8',
|
||||
});
|
||||
});
|
||||
|
||||
it('tolerates a multi-byte sequence truncated at the sample tail', () => {
|
||||
const sample = Buffer.concat([Buffer.from('日志记录\n', 'utf8'), Buffer.from([0xe4, 0xb8])]);
|
||||
expect(classifyTextSample(sample)).toEqual({ isBinary: false, encoding: 'utf-8' });
|
||||
});
|
||||
|
||||
it('treats a NUL byte beyond the UTF-16 parity window as binary', () => {
|
||||
const sample = Buffer.concat([
|
||||
Buffer.alloc(600, 0x61),
|
||||
Buffer.from([0x00]),
|
||||
Buffer.alloc(100, 0x62),
|
||||
]);
|
||||
expect(classifyTextSample(sample).isBinary).toBe(true);
|
||||
});
|
||||
|
||||
it('rejects an impossible UTF-8 lead byte at the sample tail', () => {
|
||||
const sample = Buffer.concat([Buffer.from('plain ascii log line\n'), Buffer.from([0xff])]);
|
||||
expect(classifyTextSample(sample).isBinary).toBe(true);
|
||||
});
|
||||
|
||||
it('rejects a tail lead byte not followed by continuation bytes', () => {
|
||||
const sample = Buffer.concat([Buffer.from('plain ascii log line\n'), Buffer.from([0xe4, 0x41])]);
|
||||
expect(classifyTextSample(sample).isBinary).toBe(true);
|
||||
});
|
||||
|
||||
it('classifies UTF-16 BOM and zero-byte parity samples as text with the right encoding', () => {
|
||||
const le = Buffer.concat([Buffer.from([0xff, 0xfe]), utf16Le('hello 你好')]);
|
||||
expect(classifyTextSample(le)).toEqual({ isBinary: false, encoding: 'utf-16le' });
|
||||
expect(classifyTextSample(utf16Be('hello world, plain ascii'))).toEqual({
|
||||
isBinary: false,
|
||||
encoding: 'utf-16be',
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe('decodeUtfText', () => {
|
||||
it('decodes UTF-16 LE/BE and strips the BOM', () => {
|
||||
const le = Buffer.concat([Buffer.from([0xff, 0xfe]), utf16Le('你好\nworld')]);
|
||||
|
|
|
|||
|
|
@ -95,10 +95,11 @@ function fakeFs(
|
|||
};
|
||||
const lstatImpl = async (p: string) => {
|
||||
if (fileMap.has(p)) {
|
||||
const c = fileMap.get(p)!;
|
||||
return {
|
||||
isFile: true,
|
||||
isDirectory: false,
|
||||
size: fileMap.get(p)!.length,
|
||||
size: Buffer.isBuffer(c) ? c.length : Buffer.byteLength(c),
|
||||
mtimeMs: 1000,
|
||||
ino: 1,
|
||||
};
|
||||
|
|
@ -789,6 +790,30 @@ describe('WorkspaceFsService.read', () => {
|
|||
expect(result.content).toBe(utf16.toString('base64'));
|
||||
});
|
||||
|
||||
it('reads UTF-8 Chinese log content as text instead of throwing fs.is_binary', async () => {
|
||||
const log = '2026-08-16 INFO 启动完成 ✅\n2026-08-16 INFO 处理请求 🚀 成功\n'.repeat(50);
|
||||
const fs = makeSession({ 'app.log': log }, emptyHandler);
|
||||
const result = await fs.read({
|
||||
path: 'app.log',
|
||||
offset: 0,
|
||||
length: 1024 * 1024,
|
||||
encoding: 'utf-8',
|
||||
});
|
||||
expect(result.content).toBe(log);
|
||||
expect(result.encoding).toBe('utf-8');
|
||||
expect(result.is_binary).toBe(false);
|
||||
expect(result.mime).toBe('text/plain');
|
||||
expect(result.truncated).toBe(false);
|
||||
});
|
||||
|
||||
it('returns utf-8 rather than base64 for UTF-8 Chinese text in auto mode', async () => {
|
||||
const fs = makeSession({ 'app.log': '中文日志 ✅\n' }, emptyHandler);
|
||||
const result = await fs.read({ path: 'app.log', offset: 0, length: 1024, encoding: 'auto' });
|
||||
expect(result.content).toBe('中文日志 ✅\n');
|
||||
expect(result.encoding).toBe('utf-8');
|
||||
expect(result.is_binary).toBe(false);
|
||||
});
|
||||
|
||||
it('throws fs.is_directory for a directory', async () => {
|
||||
const fs = makeSession({ 'src/a.ts': '' }, emptyHandler);
|
||||
await expect(
|
||||
|
|
@ -875,6 +900,12 @@ describe('WorkspaceFsService.resolveDownload', () => {
|
|||
expect(res.modifiedAt).toBeInstanceOf(Date);
|
||||
});
|
||||
|
||||
it('resolves a UTF-8 Chinese log as text/plain', async () => {
|
||||
const fs = makeSession({ 'app.log': '启动完成 ✅ 中文日志内容\n'.repeat(20) }, emptyHandler);
|
||||
const res = await fs.resolveDownload('app.log');
|
||||
expect(res.mime).toBe('text/plain');
|
||||
});
|
||||
|
||||
it('throws fs.is_directory for a directory', async () => {
|
||||
const fs = makeSession({ 'src/a.ts': '' }, emptyHandler);
|
||||
await expect(fs.resolveDownload('src')).rejects.toMatchObject({ code: 'fs.is_directory' });
|
||||
|
|
|
|||
|
|
@ -76,10 +76,10 @@ import {
|
|||
} from '@moonshot-ai/agent-core-v2/app/hostFolderBrowser/hostFolderBrowser';
|
||||
import {
|
||||
buildEtag,
|
||||
detectBinary,
|
||||
FS_BINARY_SAMPLE_BYTES,
|
||||
guessMime,
|
||||
} from '@moonshot-ai/agent-core-v2/_base/utils/fileMeta';
|
||||
import { classifyTextSample } from '@moonshot-ai/agent-core-v2/_base/text/encoding';
|
||||
import { z } from 'zod';
|
||||
|
||||
import { errEnvelope, okEnvelope } from '../envelope';
|
||||
|
|
@ -289,7 +289,8 @@ async function handleFsContent(
|
|||
const sampleSize = Math.min(FS_BINARY_SAMPLE_BYTES, st.size);
|
||||
const sample =
|
||||
sampleSize === 0 ? new Uint8Array() : await hostFs.readBytes(abs, sampleSize);
|
||||
isBinary = detectBinary(sample);
|
||||
const classification = classifyTextSample(sample);
|
||||
isBinary = classification.isBinary || classification.encoding !== 'utf-8';
|
||||
} catch (err) {
|
||||
sendOsFsError(reply, requestId, err, path);
|
||||
return;
|
||||
|
|
|
|||
|
|
@ -358,6 +358,17 @@ describe('server-v2 /api/v1 fs:content', () => {
|
|||
expect(res.headers.get('content-type')).toContain('text/plain');
|
||||
});
|
||||
|
||||
it('serves a UTF-8 Chinese .log file as text/plain', async () => {
|
||||
const file = join(dir as string, 'server.log');
|
||||
const log = '2026-08-16 INFO 启动完成 ✅\n'.repeat(100);
|
||||
await writeFile(file, log);
|
||||
|
||||
const res = await getContent(file);
|
||||
expect(res.status).toBe(200);
|
||||
expect(res.headers.get('content-type')).toContain('text/plain');
|
||||
expect(await res.text()).toBe(log);
|
||||
});
|
||||
|
||||
it('serves binary files byte-for-byte with an octet-stream fallback mime', async () => {
|
||||
const file = join(dir as string, 'blob.bin');
|
||||
const original = Buffer.from([0x00, 0x01, 0x02, 0xff, 0xfe, 0x00, 0x10, 0x80]);
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue