fix(desktop): read CJK file names in prose whole, settled by the disk (#1431)

The prose scan excludes CJK so a verb is not swallowed into a path, which
cut an unquoted `测试文档1.docx` to `1.docx` and `D:/资料/测试文档1.docx` to
`1.docx`. A match that is recognisably the tail of a CJK name is now
widened to the whole token.

Names the text cannot bound are settled against what exists: `报告v2.docx`,
names glued together without a space, names with spaces or full-width
brackets, and names made only of CJK plus an extension. The extractor
offers the other readings; a changed file settles them, otherwise one
cached workspace listing per folder does, longest existing name first.
CJK-only names read like prose about formats (`后缀为.docx的文件`), so they
are never linked and appear as cards or images only once confirmed.

Fixes #1423
This commit is contained in:
程序员阿江-Relakkes
2026-10-03 22:27:55 +08:00
committed by GitHub
parent 66f9938f36
commit 54a77db95d
10 changed files with 698 additions and 22 deletions
@@ -49,8 +49,9 @@ vi.mock('../../stores/workspaceContentStore', () => {
})
const getWorkspaceFile = vi.hoisted(() => vi.fn().mockResolvedValue({ state: 'ok', content: 'file body' }))
const getWorkspaceTree = vi.hoisted(() => vi.fn().mockResolvedValue({ state: 'missing', path: '', entries: [] }))
vi.mock('../../api/sessions', () => ({
sessionsApi: { getWorkspaceFile },
sessionsApi: { getWorkspaceFile, getWorkspaceTree },
}))
const copyTextToClipboard = vi.hoisted(() => vi.fn().mockResolvedValue(true))
@@ -75,6 +76,7 @@ vi.mock('../../stores/settingsStore', () => ({
}))
import { AssistantMessage } from './AssistantMessage'
import { resetDiskListingCacheForTests } from '../../hooks/useDiskConfirmedTargets'
afterEach(() => {
openPath.mockClear()
@@ -85,6 +87,8 @@ afterEach(() => {
openPreviewFn.mockReset().mockResolvedValue(undefined)
copyTextToClipboard.mockReset().mockResolvedValue(true)
getWorkspaceFile.mockReset().mockResolvedValue({ state: 'ok', content: 'file body' })
getWorkspaceTree.mockReset().mockResolvedValue({ state: 'missing', path: '', entries: [] })
resetDiskListingCacheForTests()
})
describe('AssistantMessage file references', () => {
@@ -134,6 +138,33 @@ describe('AssistantMessage file references', () => {
await waitFor(() => expect(openPath).toHaveBeenCalledWith('/other/promo/public/audio/track.wav'))
})
it('names and opens the card by the file that exists when the prose could not bound it (#1423)', async () => {
getWorkspaceTree.mockResolvedValue({
state: 'ok',
path: '',
entries: [{ name: '报告v2.docx', path: '报告v2.docx', isDirectory: false }],
})
render(<AssistantMessage sessionId="s1" content={'已生成报告v2.docx'} turnChangedFiles={[]} />)
const card = await screen.findByText('报告v2.docx', { selector: 'span' })
fireEvent.click(card.closest('button')!)
await waitFor(() => expect(openPreviewFn).toHaveBeenCalledWith('s1', '报告v2.docx', {}))
})
it('shows a card for a CJK-and-extension name only once it is on disk, and links neither', async () => {
getWorkspaceTree.mockResolvedValue({
state: 'ok',
path: '',
entries: [{ name: '开题报告.docx', path: '开题报告.docx', isDirectory: false }],
})
render(<AssistantMessage sessionId="s1" content={'已生成 开题报告.docx。只支持后缀为.docx的文件'} turnChangedFiles={[]} />)
expect(await screen.findByText('开题报告.docx', { selector: 'span' })).toBeInTheDocument()
expect(screen.queryByText(/后缀为\.docx/, { selector: 'span' })).not.toBeInTheDocument()
expect(screen.queryByRole('link', { name: /docx/ })).not.toBeInTheDocument()
})
it('keeps prose and card destinations equal when the project root is stated later', async () => {
render(<AssistantMessage sessionId="s1" content={'`track.wav`\n\n项目根目录是 `/other/promo/`'} />)
fireEvent.click(screen.getByRole('link', { name: 'track.wav' }))
@@ -21,6 +21,7 @@ import { getServerBaseUrl } from '../../lib/desktopRuntime'
import { isManagedGeneratedImagePath } from '../../lib/attachmentImages'
import { useWorkspaceContentStore } from '../../stores/workspaceContentStore'
import { useTranslation, type TranslationKey } from '../../i18n'
import { useDiskConfirmedTargets } from '../../hooks/useDiskConfirmedTargets'
type Props = {
content: string
@@ -90,7 +91,7 @@ export const AssistantMessage = memo(function AssistantMessage({
[content, sessionId, t, workDir],
)
const outputTargets = useMemo(
const extractedTargets = useMemo(
() =>
isStreaming || !sessionId
? []
@@ -99,11 +100,15 @@ export const AssistantMessage = memo(function AssistantMessage({
workDir,
changedFiles: turnChangedFiles,
includeChangedFileFallback: isTurnOutputOwner,
// Confirmed against the disk by useDiskConfirmedTargets before showing.
includeUnconfirmedNames: true,
}).filter(
(target) => target.kind !== 'image' && target.kind !== 'video',
),
[content, isStreaming, isTurnOutputOwner, sessionId, workDir, turnChangedFiles],
)
// A bare name the text could not bound is settled against the workspace listing.
const outputTargets = useDiskConfirmedTargets(sessionId, extractedTargets)
const resolveAssistantImageSrc = useMemo(
() => {
if (isStreaming || !sessionId) return undefined
@@ -19,9 +19,16 @@ vi.mock('../../lib/desktopRuntime', () => ({
const fetchServerImageBlobUrl = vi.hoisted(() => vi.fn())
vi.mock('../../lib/authedImage', () => ({ fetchServerImageBlobUrl }))
// Bare names glued to prose are settled against a workspace listing.
const getWorkspaceTree = vi.hoisted(() => vi.fn())
vi.mock('../../api/sessions', () => ({ sessionsApi: { getWorkspaceTree } }))
import { InlineImageGallery } from './InlineImageGallery'
import { resetDiskListingCacheForTests } from '../../hooks/useDiskConfirmedTargets'
beforeEach(() => {
resetDiskListingCacheForTests()
getWorkspaceTree.mockReset().mockResolvedValue({ state: 'missing', path: '', entries: [] })
fetchServerImageBlobUrl.mockReset().mockRejectedValue(new Error('403'))
// jsdom ships no object-URL support.
Object.defineProperty(URL, 'revokeObjectURL', { value: vi.fn(), configurable: true, writable: true })
@@ -219,6 +226,31 @@ describe('InlineImageGallery', () => {
expect(await screen.findByRole('alert')).toHaveTextContent('frame.png')
})
it('shows the image the workspace really holds when a verb is glued to its name', async () => {
getWorkspaceTree.mockResolvedValue({
state: 'ok',
path: '',
entries: [{ name: '1.png', path: '1.png', isDirectory: false }],
})
render(<InlineImageGallery text={'截图保存为1.png'} sessionId="s1" workDir="/w" changedFiles={[]} />)
await waitFor(() => expect(imgSrcs()).toEqual(['http://127.0.0.1:4321/preview-fs/s1/1.png']))
})
it('shows a CJK-named image only once the workspace confirms it', async () => {
getWorkspaceTree.mockResolvedValue({
state: 'ok',
path: '',
entries: [{ name: '流程图.png', path: '流程图.png', isDirectory: false }],
})
render(<InlineImageGallery text={'已导出 流程图.png,格式选.png即可'} sessionId="s1" workDir="/w" changedFiles={[]} />)
expect(screen.queryAllByRole('img')).toHaveLength(0)
await waitFor(() => expect(imgSrcs()).toEqual([
`http://127.0.0.1:4321/preview-fs/s1/${encodeURIComponent('流程图.png')}`,
]))
})
it('still raises the error block for a name the turn really wrote', async () => {
render(
<InlineImageGallery
@@ -11,6 +11,7 @@ import {
import { isAbsoluteLocalPath, previewFsUrl } from '../../lib/handlePreviewLink'
import { getServerBaseUrl } from '../../lib/desktopRuntime'
import { resolveAbsoluteOpenPath } from '../../lib/systemFileOpen'
import { useDiskConfirmedTargets } from '../../hooks/useDiskConfirmedTargets'
const IMAGE_EXTENSIONS = /\.(png|jpe?g|gif|webp|svg|bmp|avif|ico)$/i
@@ -118,6 +119,24 @@ export function InlineImageGallery({ text, sessionId, workDir, changedFiles, sup
// back to text-only extraction instead of filtering every mention away.
const changedFileEvidence = changedFiles !== undefined && changedFiles.length === 0 ? undefined : changedFiles
const extractedRelativeTargets = useMemo(() => sessionId
? extractAssistantOutputTargets(text, {
workDir,
changedFiles: changedFileEvidence,
includeUnconfirmedNames: true,
}).filter(
(target) => (
target.kind === 'image' &&
target.source !== 'markdown-link' &&
!markdownImageSources.has(normalizeImageReference(target.href)) &&
!markdownImageSources.has(normalizeImageReference(target.normalizedPath ?? ''))
),
)
: [], [changedFileEvidence, markdownImageSources, sessionId, text, workDir])
// `截图保存为1.png` may be `1.png` saved by a glued verb, or one CJK name; the
// workspace listing settles which, as it does for the output cards.
const relativeTargets = useDiskConfirmedTargets(sessionId, extractedRelativeTargets)
const images = useMemo<GalleryImage[]>(() => {
// 1. Absolute paths (legacy behavior) — served via /api/filesystem/file.
const absolute: GalleryImage[] = imagePaths.map((p) => ({ src: localImageFileUrl(p), name: fileName(p), path: p }))
@@ -130,14 +149,6 @@ export function InlineImageGallery({ text, sessionId, workDir, changedFiles, sup
// build a /preview-fs URL. Reuses the sandboxed target extractor instead of
// a bespoke relative-path regex.
const base = getServerBaseUrl()
const relativeTargets = extractAssistantOutputTargets(text, { workDir, changedFiles: changedFileEvidence }).filter(
(target) => (
target.kind === 'image' &&
target.source !== 'markdown-link' &&
!markdownImageSources.has(normalizeImageReference(target.href)) &&
!markdownImageSources.has(normalizeImageReference(target.normalizedPath ?? ''))
),
)
// Dedup: an absolute path inside the workspace can be caught by BOTH sources.
// Skip a relative target whose basename already appears among the absolute
@@ -170,7 +181,7 @@ export function InlineImageGallery({ text, sessionId, workDir, changedFiles, sup
}
return [...absolute, ...relative]
}, [changedFileEvidence, imagePaths, markdownImageSources, sessionId, text, workDir])
}, [changedFileEvidence, imagePaths, relativeTargets, sessionId, text, workDir])
// A guessed image that failed to load leaves no trace: there is nothing to retry
// when the file was never claimed to be there.
@@ -0,0 +1,139 @@
import { renderHook, waitFor } from '@testing-library/react'
import { afterEach, beforeEach, describe, expect, it, vi, type MockInstance } from 'vitest'
import { sessionsApi, type WorkspaceTreeResult } from '@/api/sessions'
import { extractAssistantOutputTargets } from '@/lib/assistantOutputTargets'
import { resetDiskListingCacheForTests, useDiskConfirmedTargets } from './useDiskConfirmedTargets'
let getWorkspaceTree: MockInstance<typeof sessionsApi.getWorkspaceTree>
function listing(path: string, files: string[]): WorkspaceTreeResult {
return {
state: 'ok',
path,
entries: files.map((name) => ({ name, path: path ? `${path}/${name}` : name, isDirectory: false })),
} as WorkspaceTreeResult
}
function onDisk(files: Record<string, string[]>) {
getWorkspaceTree.mockImplementation(async (_sessionId, dir = '') => listing(dir, files[dir] ?? []))
}
// No changed file corroborates anything, so every reading is left for the disk.
const cards = (content: string) => extractAssistantOutputTargets(content, {
workDir: '/w',
changedFiles: [],
includeUnconfirmedNames: true,
})
const named = (targets: { title: string; href: string }[]) => targets.map((target) => [target.title, target.href])
beforeEach(() => {
resetDiskListingCacheForTests()
getWorkspaceTree = vi.spyOn(sessionsApi, 'getWorkspaceTree')
})
afterEach(() => {
getWorkspaceTree.mockRestore()
})
describe('useDiskConfirmedTargets', () => {
it('renames a card to the longest reading that exists on disk', async () => {
onDisk({ '': ['报告v2.docx', 'v2.docx.bak'] })
const targets = cards('已生成报告v2.docx')
const { result } = renderHook(() => useDiskConfirmedTargets('s1', targets))
expect(named(result.current)).toEqual([['v2.docx', 'v2.docx']])
await waitFor(() => expect(named(result.current)).toEqual([['报告v2.docx', '报告v2.docx']]))
expect(getWorkspaceTree).toHaveBeenCalledTimes(1)
expect(getWorkspaceTree).toHaveBeenCalledWith('s1', '')
})
it('recovers names glued together and names with spaces and full-width brackets', async () => {
onDisk({ '': ['测试文档1.docx', '测试文档2.docx', '毕业设计(论文)任务书 张三.docx'] })
const targets = cards('已找到测试文档1.docx和测试文档2.docx。任务书是 毕业设计(论文)任务书 张三.docx')
const { result } = renderHook(() => useDiskConfirmedTargets('s1', targets))
await waitFor(() => expect(named(result.current)).toEqual([
['测试文档1.docx', '测试文档1.docx'],
['测试文档2.docx', '测试文档2.docx'],
['毕业设计(论文)任务书 张三.docx', '毕业设计(论文)任务书 张三.docx'],
]))
})
it('prefers the longer name when a shorter one exists too', async () => {
// `1.docx` also sits in the folder, but the prose spelled out more of the name.
onDisk({ '': ['1.docx', '测试文档1.docx'] })
const targets = cards('已找到测试文档1.docx')
const { result } = renderHook(() => useDiskConfirmedTargets('s1', targets))
await waitFor(() => expect(named(result.current)).toEqual([['测试文档1.docx', '测试文档1.docx']]))
})
it('matches a name the file system stored decomposed (NFD)', async () => {
onDisk({ '': ['résumé报告v2.docx'.normalize('NFD')] })
const targets = cards('见 résumé报告v2.docx')
const { result } = renderHook(() => useDiskConfirmedTargets('s1', targets))
await waitFor(() => expect(result.current[0]?.title).toBe('résumé报告v2.docx'))
})
it('keeps the scan reading when nothing exists or the listing fails', async () => {
onDisk({ '': ['other.docx'] })
const { result } = renderHook(() => useDiskConfirmedTargets('s1', cards('已生成报告v2.docx')))
await waitFor(() => expect(getWorkspaceTree).toHaveBeenCalled())
expect(named(result.current)).toEqual([['v2.docx', 'v2.docx']])
getWorkspaceTree.mockRejectedValue(new Error('403'))
const failed = renderHook(() => useDiskConfirmedTargets('s2', cards('已生成报告v3.docx')))
await waitFor(() => expect(getWorkspaceTree).toHaveBeenCalledWith('s2', ''))
expect(named(failed.result.current)).toEqual([['v3.docx', 'v3.docx']])
})
it('collapses two mentions that settle on the same file', async () => {
onDisk({ '': ['报告v2.docx'] })
const targets = cards('已生成报告v2.docx,即 `报告v2.docx`')
const { result } = renderHook(() => useDiskConfirmedTargets('s1', targets))
await waitFor(() => expect(named(result.current)).toEqual([['报告v2.docx', '报告v2.docx']]))
})
it('shows a CJK-and-extension name only once the disk confirms it', async () => {
onDisk({ '': ['开题报告.docx'] })
const targets = cards('已生成 开题报告.docx。只支持后缀为.docx的文件')
const { result } = renderHook(() => useDiskConfirmedTargets('s1', targets))
expect(result.current).toEqual([])
await waitFor(() => expect(named(result.current)).toEqual([['开题报告.docx', '开题报告.docx']]))
})
it('never shows an unconfirmed name when the listing fails', async () => {
getWorkspaceTree.mockRejectedValue(new Error('403'))
const { result } = renderHook(() => useDiskConfirmedTargets('s1', cards('已生成 开题报告.docx')))
await waitFor(() => expect(getWorkspaceTree).toHaveBeenCalled())
expect(result.current).toEqual([])
})
it('lists a folder once for every message that names files in it', async () => {
onDisk({ '': ['报告v2.docx', '报告v3.docx'] })
const first = renderHook(() => useDiskConfirmedTargets('s1', cards('已生成报告v2.docx')))
const second = renderHook(() => useDiskConfirmedTargets('s1', cards('已生成报告v3.docx')))
await waitFor(() => expect(second.result.current[0]?.title).toBe('报告v3.docx'))
expect(first.result.current[0]?.title).toBe('报告v2.docx')
expect(getWorkspaceTree).toHaveBeenCalledTimes(1)
})
it('asks the disk nothing when every name is already bounded', () => {
const targets = cards('见 `报告v2.docx` 和 out/a.docx')
renderHook(() => useDiskConfirmedTargets('s1', targets))
expect(getWorkspaceTree).not.toHaveBeenCalled()
})
})
@@ -0,0 +1,109 @@
import { useEffect, useMemo, useState } from 'react'
import { sessionsApi } from '@/api/sessions'
import type { AssistantOutputTarget } from '@/lib/assistantOutputTargets'
type Listing = { key: string; namesByDir: Map<string, Set<string>> }
const DIR_SEPARATOR = '\u0000'
const LISTING_TTL_MS = 5_000
// A long reply history mounts many messages naming files in the same folder; one
// listing serves them all for a few seconds, then a fresh one sees new files.
const listingCache = new Map<string, { at: number; names: Promise<Set<string> | null> }>()
export function resetDiskListingCacheForTests() {
listingCache.clear()
}
function listFileNames(sessionId: string, directory: string): Promise<Set<string> | null> {
const cacheKey = `${sessionId}${DIR_SEPARATOR}${directory}`
const cached = listingCache.get(cacheKey)
if (cached && Date.now() - cached.at < LISTING_TTL_MS) return cached.names
const names = sessionsApi.getWorkspaceTree(sessionId, directory)
.then((tree) => tree.state === 'ok'
? new Set(tree.entries.filter((entry) => !entry.isDirectory).map((entry) => entry.name.normalize('NFC')))
: null)
.catch(() => null)
listingCache.set(cacheKey, { at: Date.now(), names })
return names
}
function splitDirectory(path: string): [string, string] {
const cut = Math.max(path.lastIndexOf('/'), path.lastIndexOf('\\'))
return cut === -1 ? ['', path] : [path.slice(0, cut), path.slice(cut + 1)]
}
/**
* Let the disk settle names the text could not bound.
*
* A bare prose mention such as `报告v2.docx` reaches the card as the scan's best
* reading (`v2.docx`) plus the other readings it may be. One workspace listing of
* each directory involved tells which of them exists; the longest that does wins,
* as it is the one the prose spelled out in full. Without an answer — no session,
* a failed listing, no reading on disk — the card keeps the scan's reading, and a
* target that {@link AssistantOutputTarget.awaitsConfirmation awaits confirmation}
* (`开题报告.docx`, which reads like `后缀为.docx的文件`) is not shown at all.
*/
export function useDiskConfirmedTargets(
sessionId: string | undefined,
targets: AssistantOutputTarget[],
): AssistantOutputTarget[] {
const directories = useMemo(() => [...new Set(
targets
.filter((target) => target.nameCandidates?.length || target.awaitsConfirmation)
.map((target) => splitDirectory(target.normalizedPath ?? target.href)[0]),
)].sort(), [targets])
const key = sessionId && directories.length > 0 ? `${sessionId}${DIR_SEPARATOR}${directories.join(DIR_SEPARATOR)}` : ''
const [listing, setListing] = useState<Listing | null>(null)
useEffect(() => {
if (!key || !sessionId) return
let current = true
void Promise.all(directories.map(async (directory) => {
const names = await listFileNames(sessionId, directory)
return names ? [directory, names] as const : null
})).then((results) => {
if (!current) return
setListing({ key, namesByDir: new Map(results.filter((result) => result !== null)) })
})
return () => {
current = false
}
// `key` encodes the session and every directory.
// eslint-disable-next-line react-hooks/exhaustive-deps
}, [key])
return useMemo(() => {
if (!listing || listing.key !== key) return targets.filter((target) => !target.awaitsConfirmation)
const seen = new Set<string>()
return targets.map((target): AssistantOutputTarget | null => {
if (!target.nameCandidates?.length && !target.awaitsConfirmation) return target
const [directory, name] = splitDirectory(target.normalizedPath ?? target.href)
const names = listing.namesByDir.get(directory)
const confirmed = names && [...(target.nameCandidates ?? []), name]
.sort((left, right) => right.length - left.length)
.find((candidate) => names.has(candidate.normalize('NFC')))
// A name only the disk could vouch for is not shown on a guess.
if (!confirmed) return target.awaitsConfirmation ? null : target
const { awaitsConfirmation: _awaitsConfirmation, ...settled } = target
if (confirmed === name) return settled
const corrected = directory ? `${directory}/${confirmed}` : confirmed
return {
...settled,
id: `${target.kind}:${corrected}`,
title: confirmed,
href: corrected,
normalizedPath: corrected,
subtitle: corrected,
}
}).filter((target): target is AssistantOutputTarget => {
if (!target) return false
// `v2.docx` and `报告v2.docx` in one reply can settle on the same file.
if (seen.has(target.id)) return false
seen.add(target.id)
return true
})
}, [key, listing, targets])
}
@@ -443,6 +443,77 @@ describe('extractAssistantOutputTargets with changedFiles reconciliation', () =>
.toEqual(['开题报告2.docx', '开题报告3.docx', '开题报告_v2.docx'])
})
describe('a name the text alone cannot bound', () => {
const names = (content: string, changedFiles: string[]) =>
extractAssistantOutputTargets(content, {
workDir: '/w',
changedFiles,
includeChangedFileFallback: false,
includeUnconfirmedNames: true,
})
.map((target) => [target.title, target.normalizedPath])
it('takes the longer reading the turn really wrote', () => {
expect(names('已生成报告v2.docx', ['/w/报告v2.docx'])).toEqual([['报告v2.docx', '报告v2.docx']])
})
it('splits names glued together without a space', () => {
expect(names('已找到测试文档1.docx和测试文档2.docx', ['/w/测试文档1.docx', '/w/测试文档2.docx']))
.toEqual([['测试文档1.docx', '测试文档1.docx'], ['测试文档2.docx', '测试文档2.docx']])
})
it('recovers a name with spaces and full-width brackets', () => {
expect(names('已生成 毕业设计(论文)任务书 张三.docx', ['/w/毕业设计(论文)任务书 张三.docx']))
.toEqual([['毕业设计(论文)任务书 张三.docx', '毕业设计(论文)任务书 张三.docx']])
})
it('prefers the mention over a shorter name that also exists', () => {
expect(names('已找到 测试文档1.docx', ['/w/1.docx', '/w/测试文档1.docx']))
.toEqual([['测试文档1.docx', '测试文档1.docx']])
})
describe('a name made only of CJK and an extension', () => {
it('is not guessed at by default, as prose about formats looks the same', () => {
for (const content of ['已生成 开题报告.docx', '只支持后缀为.docx的文件']) {
expect(extractAssistantOutputTargets(content, { workDir: '/w', changedFiles: [] })).toEqual([])
}
})
it('is offered for confirmation when asked, and settled by a changed file', () => {
const unconfirmed = extractAssistantOutputTargets('已生成 开题报告.docx', {
workDir: '/w', changedFiles: [], includeUnconfirmedNames: true,
})
expect(unconfirmed).toMatchObject([{ normalizedPath: '开题报告.docx', awaitsConfirmation: true }])
const written = extractAssistantOutputTargets('已生成 开题报告.docx', {
workDir: '/w', changedFiles: ['/w/开题报告.docx'], includeUnconfirmedNames: true,
})
expect(written).toHaveLength(1)
expect(written[0]).toMatchObject({ title: '开题报告.docx', normalizedPath: '开题报告.docx' })
expect(written[0]!.awaitsConfirmation).toBeUndefined()
})
})
it('leaves the readings open for the disk when no changed file settles them', () => {
const [target] = extractAssistantOutputTargets('已生成报告v2.docx', { workDir: '/w', changedFiles: [] })
expect(target).toMatchObject({ normalizedPath: 'v2.docx' })
expect(target!.nameCandidates).toEqual(expect.arrayContaining(['已生成报告v2.docx', '报告v2.docx']))
})
})
it('keeps the whole CJK name when the prose does not quote it (#1423)', () => {
const targets = extractAssistantOutputTargets(
'已找到 测试文档1.docx 和 测试文档2.docx,另一份在 C:\\Users\\a\\Desktop\\资料\\测试文档3.docx',
{ workDir: 'C:\\Users\\a\\Desktop', changedFiles: [] },
)
expect(targets.map((target) => [target.title, target.normalizedPath])).toEqual([
['测试文档1.docx', '测试文档1.docx'],
['测试文档2.docx', '测试文档2.docx'],
['测试文档3.docx', 'C:/Users/a/Desktop/资料/测试文档3.docx'],
])
})
it('keeps a CJK directory', () => {
const targets = extractAssistantOutputTargets(
'见 `论文/开题报告终稿.docx`',
+68 -5
View File
@@ -1,6 +1,12 @@
import { resolveAssistantFileHref } from './assistantFileContext'
import { trimTrailingPunctuation } from './urlBoundary'
import { isLinkableFilePath, parseFilePathRef, splitTextByFilePaths } from './filePathBoundary'
import {
bareNameCandidates,
findUnicodeExtensionNames,
isLinkableFilePath,
parseFilePathRef,
splitTextByFilePaths,
} from './filePathBoundary'
import { isGeneratedArtifactFile, isOutputResourceFile, isShellProducedDeliverable } from './fileCapabilities'
export type AssistantOutputTargetKind =
@@ -26,6 +32,14 @@ export type AssistantOutputTarget = {
normalizedPath?: string
confidence: 'high'
source: AssistantOutputTargetSource
/**
* Other names a bare prose mention may really be (`报告v2.docx` for `v2.docx`),
* longest first, in the same directory. Text cannot choose between them; the
* changed files or the disk can — see {@link useDiskConfirmedTargets}.
*/
nameCandidates?: string[]
/** Not to be shown until a changed file or the disk confirms one of its names. */
awaitsConfirmation?: boolean
}
export type ExtractAssistantOutputTargetOptions = {
@@ -48,6 +62,13 @@ export type ExtractAssistantOutputTargetOptions = {
* did not mention them. Reconciliation still runs when false. Defaults to true.
*/
includeChangedFileFallback?: boolean
/**
* Also return names made only of CJK and an extension (`开题报告.docx`), marked
* {@link AssistantOutputTarget.awaitsConfirmation}. Their text reads the same
* as prose about formats (`后缀为.docx的文件`), so only a caller that confirms
* them against the disk before showing them should ask. Defaults to false.
*/
includeUnconfirmedNames?: boolean
}
type FileTargetMatch = {
@@ -206,7 +227,12 @@ export function extractAssistantOutputTargets(
}, createFileKey(fileTarget), treeMatch.position)
}
const queuePlainPath = (path: string, position: number) => {
const queuePlainPath = (
path: string,
position: number,
nameCandidates?: string[],
awaitsConfirmation = false,
) => {
const href = resolveAssistantFileHref(path, content)
const fileTarget = toWorkspaceFileTarget(href, workDir)
@@ -223,6 +249,8 @@ export function extractAssistantOutputTargets(
normalizedPath: fileTarget.normalizedPath,
confidence: 'high',
source: 'plain-path',
...(nameCandidates?.length ? { nameCandidates } : {}),
...(awaitsConfirmation ? { awaitsConfirmation } : {}),
}, createFileKey(fileTarget), position)
}
@@ -240,6 +268,8 @@ export function extractAssistantOutputTargets(
}
let plainTextPosition = 0
let previousPathEnd = 0
const pathRanges: Array<{ start: number; end: number }> = []
for (const segment of splitTextByFilePaths(content)) {
const position = plainTextPosition
plainTextPosition += segment.value.length
@@ -247,6 +277,9 @@ export function extractAssistantOutputTargets(
if (segment.type !== 'path') {
continue
}
const floor = previousPathEnd
previousPathEnd = position + segment.value.length
pathRanges.push({ start: position, end: previousPathEnd })
if (isInMarkdownLink(position, markdownLinks)) {
continue
@@ -260,7 +293,25 @@ export function extractAssistantOutputTargets(
continue
}
queuePlainPath(segment.ref.path, position)
// Only a bare name is ambiguous at its edges; a path with a directory is not.
const nameCandidates = /[\\/]/.test(segment.ref.path)
? undefined
: bareNameCandidates(content, position, position + segment.ref.path.length, floor)
queuePlainPath(segment.ref.path, position, nameCandidates)
}
if (options.includeUnconfirmedNames) {
const overlaps = (start: number, end: number) => [...pathRanges, ...codeSpans]
.some((range) => start < range.end && end > range.start)
for (const name of findUnicodeExtensionNames(content)) {
if (overlaps(name.start, name.end)) continue
if (isInMarkdownLink(name.start, markdownLinks) || isInCodeBlock(name.start, codeBlocks)) continue
const nameCandidates = /[\\/]/.test(name.ref.path)
? undefined
: bareNameCandidates(content, name.start, name.start + name.ref.path.length, 0)
queuePlainPath(name.ref.path, name.start, nameCandidates, true)
}
}
candidates.sort((left, right) => {
@@ -370,7 +421,14 @@ function reconcileTargetsWithChangedFiles(
// An authored absolute path (including an explicit prose root) is identity,
// not a basename hint. A checkpoint cannot disprove a shell-created output.
const explicitPath = isAbsoluteFilePath(target.href)
const match = explicitPath ? null : matchChangedFile(mentioned, changedFiles)
// The longer reading of an ambiguous name wins when the turn really wrote it
// (`报告v2.docx`, not the `v2.docx` the prose scan settled on).
const match = explicitPath
? null
: [...(target.nameCandidates ?? []), mentioned]
.sort((left, right) => getBasename(right).length - getBasename(left).length)
.map((name) => matchChangedFile(name, changedFiles))
.find(Boolean) ?? null
// Checkpoints record editing tools, not arbitrary shell output. Explicit
// identities and document/media deliverables survive absent evidence; bare
// source-like mentions still need corroboration to avoid resource-strip noise.
@@ -391,12 +449,17 @@ function reconcileTargetsWithChangedFiles(
}
seen.add(key)
const { nameCandidates, awaitsConfirmation, ...rest } = target
out.push({
...target,
...rest,
id: createId(target.kind, corrected),
...(target.source === 'plain-path' ? { title: getBasename(corrected) } : {}),
href: corrected,
normalizedPath: corrected,
subtitle: corrected,
// A changed-file match is settled; only an unmatched guess stays open.
...(!match && nameCandidates ? { nameCandidates } : {}),
...(!match && awaitsConfirmation ? { awaitsConfirmation } : {}),
})
if (out.length >= limit) break
}
+87
View File
@@ -1,6 +1,8 @@
import { describe, expect, it } from 'vitest'
import {
AMBIGUOUS_STANDALONE_EXTENSIONS,
bareNameCandidates,
findUnicodeExtensionNames,
LINKABLE_FILE_EXTENSIONS,
isFilePathOnly,
isLinkableFilePath,
@@ -104,6 +106,91 @@ describe('splitTextByFilePaths', () => {
])
})
describe('a CJK file name in prose (#1423)', () => {
const paths = (text: string) => splitTextByFilePaths(text)
.filter((s) => s.type === 'path')
.map((s) => s.value)
it.each([
['已找到 测试文档1.docx 和 测试文档2.docx', ['测试文档1.docx', '测试文档2.docx']],
['- 测试文档1.docx', ['测试文档1.docx']],
['| 测试文档1.docx | 12KB |', ['测试文档1.docx']],
['已生成:开题报告_v2.docx。', ['开题报告_v2.docx']],
['见 资料/report.docx', ['资料/report.docx']],
['改了 src/中文/a.ts:3', ['src/中文/a.ts:3']],
['已找到 D:/资料/测试文档1.docx', ['D:/资料/测试文档1.docx']],
['已找到 C:\\Users\\a\\Desktop\\测试文档1.docx', ['C:\\Users\\a\\Desktop\\测试文档1.docx']],
['在 /Users/a/论文/开题报告4.pdf', ['/Users/a/论文/开题报告4.pdf']],
])('keeps the whole name: %s', (text, expected) => {
expect(paths(text)).toEqual(expected)
})
it('still leaves a CJK verb glued to an ASCII name in the sentence', () => {
expect(paths('修改了foo.ts')).toEqual(['foo.ts'])
expect(paths('修改了/Users/a/foo.ts')).toEqual(['/Users/a/foo.ts'])
})
it('preserves the original text exactly', () => {
const text = '已找到 测试文档1.docx,以及 D:/资料/测试文档2.docx。'
expect(splitTextByFilePaths(text).map((s) => s.value).join('')).toBe(text)
})
})
it('links no name made only of CJK and an extension: prose about formats reads the same', () => {
// `开题报告.docx` and `后缀为.docx的文件` cannot be told apart by their text.
for (const text of ['已生成 开题报告.docx', '只支持后缀为.docx的文件', '把它另存为.pdf格式']) {
expect(splitTextByFilePaths(text).some((s) => s.type === 'path')).toBe(false)
}
})
describe('findUnicodeExtensionNames', () => {
const names = (text: string) => findUnicodeExtensionNames(text).map((match) => text.slice(match.start, match.end))
it('reads the whole token in front of the extension, for a caller that checks the disk', () => {
expect(names('已生成 开题报告.docx 和 摘要.pdf。')).toEqual(['开题报告.docx', '摘要.pdf'])
expect(names('只支持后缀为.docx的文件')).toEqual(['只支持后缀为.docx'])
})
it('leaves names the prose scan already reads, and non-file extensions', () => {
expect(names('见 测试文档1.docx、report.docx、Node.js、版本2.0')).toEqual([])
})
})
describe('bareNameCandidates', () => {
const at = (text: string, name: string) => {
const start = text.indexOf(name)
return bareNameCandidates(text, start, start + name.length, 0)
}
it('offers the longer names a space or full-width bracket may belong to', () => {
const text = '任务书在 毕业设计(论文)任务书 张三.docx 里'
expect(at(text, '三.docx')).toEqual(expect.arrayContaining([
'毕业设计(论文)任务书 张三.docx',
'任务书 张三.docx',
'张三.docx',
]))
})
it('offers the shorter names at each CJK boundary of a glued token', () => {
const text = '已找到测试文档1.docx和测试文档2.docx'
const candidates = at(text, '已找到测试文档1.docx')
expect(candidates).toEqual(expect.arrayContaining(['测试文档1.docx', '1.docx']))
expect(candidates).not.toContain('已找到测试文档1.docx')
})
it('lists longer names first and never crosses a sentence mark, a line or the previous path', () => {
const text = '前一句。\n见:报告 v2.docx'
const candidates = at(text, 'v2.docx')
expect(candidates[0]).toBe('报告 v2.docx')
expect(candidates.every((name) => !/[。:\n]/.test(name))).toBe(true)
expect(bareNameCandidates('a.docx 报告v2.docx', 9, 16, 6)).toEqual(['报告v2.docx', '告v2.docx'])
})
it('does not trim inside an ASCII word', () => {
expect(at('见 report.docx', 'report.docx')).toEqual(['见 report.docx'])
})
})
it('finds every path in a sentence', () => {
const segments = splitTextByFilePaths('先看 a/b.ts:10,再看 c/d.py')
expect(segments.filter((s) => s.type === 'path').map((s) => s.value)).toEqual(['a/b.ts:10', 'c/d.py'])
+134 -6
View File
@@ -20,6 +20,11 @@
* from `修`), while {@link parseFilePathRef} — an href we generated ourselves,
* a backtick-quoted span — accepts them, because `README-拍摄大纲.md` is a real
* file the turn really wrote and its delimiter is the span, not the sentence.
* When the ASCII match is recognisably the tail of a CJK name (`测试文档1.docx`),
* the prose scanner widens it to the whole token — see
* {@link widenAcrossUnicodeLetters}. A name with no ASCII part at all
* (`开题报告.docx`) reads exactly like prose about formats, so it is never linked
* here; {@link findUnicodeExtensionNames} offers it to callers that check the disk.
* CJK *punctuation* is excluded in both — it is sentence material either way.
*/
@@ -314,15 +319,19 @@ export function splitTextByFilePaths(text: string): FilePathSegment[] {
continue
}
if (start > cursor) {
segments.push({ type: 'text', value: text.slice(cursor, start) })
const widened = 'url' in found ? null : widenAcrossUnicodeLetters(text, start, found, cursor)
const pathStart = widened?.start ?? start
const ref = widened?.ref ?? found
if (pathStart > cursor) {
segments.push({ type: 'text', value: text.slice(cursor, pathStart) })
}
segments.push(
'url' in found
? { type: 'github', value: found.raw, ref: found }
: { type: 'path', value: found.raw, ref: found },
'url' in ref
? { type: 'github', value: ref.raw, ref }
: { type: 'path', value: ref.raw, ref },
)
cursor = start + found.raw.length
cursor = pathStart + ref.raw.length
FILE_PATH_SCAN_RE.lastIndex = cursor
}
@@ -333,6 +342,125 @@ export function splitTextByFilePaths(text: string): FilePathSegment[] {
return segments
}
const UNICODE_LETTER_RE = /[^\x00-\x7F]/
const IS_UNICODE_WORD_RE = /[\p{L}\p{N}]/u
const TOKEN_CHAR_RE = /[\p{L}\p{N}_.\-@+/\\~]/u
/**
* The ASCII scan stops at the first CJK letter, so a match flush against one is
* either a path glued to a CJK verb (`修改了lib/foo.ts`, keep it) or the ASCII
* tail of a CJK name (`测试文档1.docx` → `1.docx`, #1423). A tail is recognisable:
* it opens with a digit or symbol, sits in a later segment of the token
* (`资料/测试文档v1.docx`), or is a lone `/name` cut from a relative path
* (`资料/report.docx`). Then the whole whitespace/punctuation-bounded token is
* read with the Unicode segment set, and used only if it is one linkable path
* ending exactly where the ASCII match did.
*
* A verb glued to a name that opens with an ASCII letter (`报告v2.docx`) stays
* ambiguous with `修改了foo.ts`, and is left as the ASCII match.
*/
function widenAcrossUnicodeLetters(
text: string,
start: number,
found: FilePathRef,
floor: number,
): { start: number; ref: FilePathRef } | null {
const previous = start > 0 ? text.charAt(start - 1) : ''
if (!UNICODE_LETTER_RE.test(previous) || !IS_UNICODE_WORD_RE.test(previous)) return null
let tokenStart = start
while (tokenStart > floor && TOKEN_CHAR_RE.test(text.charAt(tokenStart - 1))) tokenStart -= 1
// A drive root: `:` is not a token character, but `D:` opens the path.
if (
tokenStart - 2 >= floor
&& text.charAt(tokenStart - 1) === ':'
&& /[A-Za-z]/.test(text.charAt(tokenStart - 2))
&& (tokenStart - 2 === floor || !TOKEN_CHAR_RE.test(text.charAt(tokenStart - 3)))
) {
tokenStart -= 2
}
const leading = text.slice(tokenStart, start)
const rooted = /^(?:~|\.{1,2})?[\\/]|^[A-Za-z]:[\\/]/.test(leading)
const startsAtSeparator = /^[\\/]/.test(found.raw)
const isTail = rooted
|| /[\\/]/.test(leading)
|| /^[\d_.\-@+]/.test(found.raw)
|| (startsAtSeparator && (found.path.match(/[\\/]/g)?.length ?? 0) === 1)
if (!isTail) return null
const end = start + found.raw.length
const ref = parseFilePathRef(text.slice(tokenStart, end))
return ref && ref.raw.length === end - tokenStart ? { start: tokenStart, ref } : null
}
export type UnicodeExtensionName = { start: number; end: number; ref: FilePathRef }
/**
* Names made only of non-ASCII letters and an extension: `开题报告.docx`.
*
* The prose scan never links these, because their text is the same as prose
* about formats — `只支持后缀为.docx的文件`, `另存为.pdf格式`. They are returned
* for a caller that can confirm them against the disk before showing anything.
* A name with an ASCII part in front of the dot (`测试文档1.docx`) is the prose
* scan's to read.
*/
export function findUnicodeExtensionNames(text: string): UnicodeExtensionName[] {
const names: UnicodeExtensionName[] = []
let floor = 0
for (const match of text.matchAll(/\.[A-Za-z0-9]+(?![A-Za-z0-9_])/g)) {
const start = match.index ?? 0
if (start < floor) continue
const previous = start > 0 ? text.charAt(start - 1) : ''
if (!UNICODE_LETTER_RE.test(previous) || !IS_UNICODE_WORD_RE.test(previous)) continue
const widened = widenAcrossUnicodeLetters(text, start, { raw: match[0], path: match[0] }, floor)
if (!widened || 'url' in widened.ref) continue
const end = widened.start + widened.ref.raw.length
names.push({ start: widened.start, end, ref: widened.ref })
floor = end
}
return names
}
const CANDIDATE_STOP_RE = /[\n\r|`*"'<>::,,.。;;、!!??“”‘’「」『』《》\[\]\\/]/
const CANDIDATE_REACH = 60
/**
* Every name a bare file mention in prose may really be, longest first, for a
* caller that can check them against the disk.
*
* Text alone cannot settle `报告v2.docx` (verb glued to `v2.docx`, or one CJK
* name?), `测试文档1.docx和测试文档2.docx` (where does the second start?) or
* `毕业设计(论文)任务书 张三.docx` (spaces and full-width brackets are sentence
* material in the scan). So this widens leftwards across spaces and brackets up
* to a sentence mark, a line, `floor` or {@link CANDIDATE_REACH} characters, and
* trims rightwards at each CJK boundary inside the name. The extension never
* moves, and the mention itself (`start`..`end`) is not listed.
*/
export function bareNameCandidates(text: string, start: number, end: number, floor: number): string[] {
const name = text.slice(start, end)
const dot = name.lastIndexOf('.')
if (dot <= 0 && !name.startsWith('.')) return []
let left = start
const reach = Math.max(floor, start - CANDIDATE_REACH)
while (left > reach && !CANDIDATE_STOP_RE.test(text.charAt(left - 1))) left -= 1
const candidates: string[] = []
for (let i = left; i < start; i += 1) {
if (!/\s/.test(text.charAt(i))) candidates.push(text.slice(i, end))
}
for (let i = start + 1; i < start + Math.max(dot, 0); i += 1) {
if (UNICODE_LETTER_RE.test(text.charAt(i - 1)) && !/\s/.test(text.charAt(i))) {
candidates.push(text.slice(i, end))
}
}
return candidates
}
/**
* Parse a reference that is already known to be one — the href/data attribute
* round-trip, and inline code spans.