Fix links to other site-spaces / sections in llms.txt (#4222)

This commit is contained in:
Samy Pessé
2026-04-30 08:43:32 +02:00
committed by GitHub
parent 10289e4881
commit 4b78672135
15 changed files with 307 additions and 523 deletions
+5
View File
@@ -0,0 +1,5 @@
---
"gitbook": patch
---
Fix links to other spaces/sections in the llms.txt.
+2 -2
View File
@@ -356,7 +356,7 @@
"react-dom": "catalog:",
},
"catalog": {
"@gitbook/api": "0.179.0",
"@gitbook/api": "0.180.0",
"@scalar/api-client-react": "^1.3.46",
"@tsconfig/node20": "^20.1.6",
"@tsconfig/strictest": "^2.0.6",
@@ -752,7 +752,7 @@
"@fortawesome/fontawesome-svg-core": ["@fortawesome/fontawesome-svg-core@7.2.0", "", { "dependencies": { "@fortawesome/fontawesome-common-types": "7.2.0" } }, "sha512-6639htZMjEkwskf3J+e6/iar+4cTNM9qhoWuRfj9F3eJD6r7iCzV1SWnQr2Mdv0QT0suuqU8BoJCZUyCtP9R4Q=="],
"@gitbook/api": ["@gitbook/api@0.179.0", "", { "dependencies": { "event-iterator": "^2.0.0", "eventsource-parser": "^3.0.0" } }, "sha512-DrD/Pdfkdv6WWQb1nE0o91O4+gzIhRh+v8l9fhL5Wq57yt0Tk0kTCUycVpI1+H9qUt39bopN4O+KTns4mR2eLQ=="],
"@gitbook/api": ["@gitbook/api@0.180.0", "", { "dependencies": { "event-iterator": "^2.0.0", "eventsource-parser": "^3.0.0" } }, "sha512-jGPP9cVGDLqVv1YjuZMpoVNfD0Yj30vVbkYwnVc4B2GRhYENJm/27x+nAzG7379YNilWOneqE245NyIWNgTu1w=="],
"@gitbook/browser-types": ["@gitbook/browser-types@workspace:packages/browser-types"],
+1 -1
View File
@@ -43,7 +43,7 @@
"catalog": {
"@tsconfig/strictest": "^2.0.6",
"@tsconfig/node20": "^20.1.6",
"@gitbook/api": "0.179.0",
"@gitbook/api": "0.180.0",
"@scalar/api-client-react": "^1.3.46",
"@types/react": "^19.0.0",
"@types/react-dom": "^19.0.0",
+54 -1
View File
@@ -5,7 +5,12 @@ import {
throwIfDataError,
} from '@/lib/data';
import { getLogger } from '@/lib/logger';
import { getLocalizedTitle, getSiteStructureSections } from '@/lib/sites';
import {
findSiteSpaceBy,
getFallbackSiteSpacePath,
getLocalizedTitle,
getSiteStructureSections,
} from '@/lib/sites';
import type {
ChangeRequest,
PublishedSiteContent,
@@ -386,6 +391,54 @@ export async function fetchSiteContextByIds(
};
}
/**
* Create a site context scoped to a specific site space.
* This keeps the site structure from the current context while resolving content
* against the target space revision.
*/
export async function fetchSiteContextForSiteSpace(
baseContext: GitBookSiteContext,
siteSpace: SiteSpace
): Promise<GitBookSiteContext> {
const found = findSiteSpaceBy(baseContext.structure, (entry) => entry.id === siteSpace.id);
if (!found) {
throw new Error(`Site space "${siteSpace.id}" not found in site structure`);
}
const spaceContext = await fetchSpaceContextByIds(baseContext, {
space: siteSpace.space.id,
shareKey: baseContext.shareKey,
changeRequest: undefined,
revision: siteSpace.space.revision,
});
const siteSpaces =
baseContext.structure.type === 'siteSpaces'
? baseContext.structure.structure
: (found.siteSection?.siteSpaces ?? baseContext.siteSpaces);
return {
...baseContext,
...spaceContext,
locale: siteSpace.space.language ?? spaceContext.locale,
linker: baseContext.linker.withOtherSiteSpace({
spaceBasePath: getFallbackSiteSpacePath(baseContext, siteSpace),
}),
siteSpace,
siteSpaces,
visibleSiteSpaces: filterHiddenSiteSpaces(siteSpaces),
sections:
baseContext.sections && found.siteSection
? { ...baseContext.sections, current: found.siteSection }
: baseContext.sections,
visibleSections:
baseContext.visibleSections && found.siteSection
? { ...baseContext.visibleSections, current: found.siteSection }
: baseContext.visibleSections,
};
}
/**
* Fetch a space context by IDs.
*/
+51 -1
View File
@@ -10,6 +10,7 @@ import { getCacheTag, getComputedContentSourceCacheTags } from '@gitbook/cache-t
import { parse as parseCacheControl } from '@tusbar/cache-control';
import { cacheLife, cacheTag } from 'next/cache';
import { cache } from '../cache';
import { isRollout } from '../rollout';
import { DataFetcherError, wrapCacheDataFetcherError } from './errors';
import type { GitBookDataFetcher } from './types';
@@ -87,7 +88,19 @@ export function createDataFetcher(
});
},
getRevisionPageMarkdown(params) {
return getRevisionPageMarkdown(input, {
if (
isRollout({
discriminator: params.spaceId,
percentageRollout: 20,
})
) {
return getRevisionPageMarkdown(input, {
spaceId: params.spaceId,
revisionId: params.revisionId,
pageId: params.pageId,
});
}
return getRevisionPageMarkdownV1(input, {
spaceId: params.spaceId,
revisionId: params.revisionId,
pageId: params.pageId,
@@ -308,6 +321,42 @@ const getRevision = cache(
}
);
const getRevisionPageMarkdownV1 = cache(
async (
input: DataFetcherInput,
params: { spaceId: string; revisionId: string; pageId: string }
) => {
'use cache: remote';
return wrapCacheDataFetcherError(async () => {
return trace(
`getRevisionPageMarkdown(${params.spaceId}, ${params.revisionId}, ${params.pageId})`,
async () => {
const api = apiClient(input);
const res = await api.spaces.getPageInRevisionById(
params.spaceId,
params.revisionId,
params.pageId,
{
format: 'markdown',
},
{
...noCacheFetchOptions,
}
);
cacheTag(...getCacheTagsFromResponse(res));
cacheLife('max');
if (!('markdown' in res.data)) {
throw new DataFetcherError('Page is not a document', 404);
}
return res.data.markdown;
}
);
});
}
);
const getRevisionPageMarkdown = cache(
async (
input: DataFetcherInput,
@@ -325,6 +374,7 @@ const getRevisionPageMarkdown = cache(
params.pageId,
{
format: 'markdown',
'format.markdown.refs': 'stable',
},
{
...noCacheFetchOptions,
+3
View File
@@ -257,6 +257,9 @@ export function linkerWithAbsoluteURLs(linker: GitBookLinker): GitBookLinker {
export function linkerWithMarkdownPages(linker: GitBookLinker): GitBookLinker {
const self: GitBookLinker = {
...linker,
fork: (override) => linkerWithMarkdownPages(linker.fork(override)),
withOtherSiteSpace: (override) =>
linkerWithMarkdownPages(linker.withOtherSiteSpace(override)),
toPathForPage: (input) => {
return self.toPathForPagePath({
path: input.page.path,
+51 -13
View File
@@ -1,9 +1,12 @@
import path from 'node:path';
import type { GitBookAnyContext, GitBookSiteContext } from '@/lib/context';
import {
type GitBookAnyContext,
type GitBookSiteContext,
fetchSiteContextForSiteSpace,
} from '@/lib/context';
import { DataFetcherError, throwIfDataError } from '@/lib/data';
import type { ResolvedPagePath } from '@/lib/pages';
import { getIndexablePages } from '@/lib/sitemap';
import { getFallbackSiteSpacePath } from '@/lib/sites';
import { getMarkdownForPagesTree } from '@/routes/llms';
import {
type RevisionPageDocument,
@@ -78,33 +81,29 @@ export async function getMarkdownForPageInSpace(
siteSpace: SiteSpace,
page: RevisionPageDocument | RevisionPageGroup
): Promise<string> {
const { dataFetcher } = context;
const spaceBasePath = getFallbackSiteSpacePath(context, siteSpace);
const linker = context.linker.withOtherSiteSpace({
spaceBasePath,
});
const siteSpaceContext = await fetchSiteContextForSiteSpace(context, siteSpace);
// Handle group pages (pages with no content that list their children)
if (page.type === RevisionPageType.Group) {
return renderGroupPageMarkdown({ linker, page });
return renderGroupPageMarkdown({ linker: siteSpaceContext.linker, page });
}
const rawMarkdown = await throwIfDataError(
dataFetcher.getRevisionPageMarkdown({
spaceId: siteSpace.space.id,
revisionId: siteSpace.space.revision,
siteSpaceContext.dataFetcher.getRevisionPageMarkdown({
spaceId: siteSpaceContext.space.id,
revisionId: siteSpaceContext.revisionId,
pageId: page.id,
})
);
const tree = await fromPageMarkdown(context, {
const tree = await fromPageMarkdown(siteSpaceContext, {
markdown: rawMarkdown,
pagePath: page.path,
});
// Handle empty document pages which have children (same as getMarkdownForPage)
if (isEmptyMarkdownPage(tree) && page.pages.length > 0) {
return renderGroupPageMarkdown({ linker, page });
return renderGroupPageMarkdown({ linker: siteSpaceContext.linker, page });
}
return toPageMarkdown(tree);
@@ -231,6 +230,7 @@ async function rewriteMarkdownLinks(
const pending: Array<Promise<void>> = [];
visit(tree, 'link', (node: Link) => {
const isMention = isMentionLike(node);
const original = node.url;
// Skip anchors, mailto:, http(s):, protocol-like
@@ -246,10 +246,35 @@ async function rewriteMarkdownLinks(
const resolved = await resolveContentRef(contentRef, context);
if (resolved?.href) {
node.url = resolved.href;
} else {
// We use an absolute URL so that crawler don't follow it.
node.url = `broken://${original.startsWith('/') ? original.slice(1) : original}`;
}
if (isMention) {
// Replace the text for mentions as otherwise it contains the raw ref
if (resolved) {
node.children = [
{
type: 'text',
value: resolved.text,
},
];
} else {
node.children = [
{
type: 'text',
value: 'Broken mention',
},
];
}
node.title = undefined;
}
})()
);
} else {
// DEPRECATED: to be removed once rollout for getRevisionPageMarkdown is done
//
// Resolve against the current page’s directory and strip any leading “/” or "../"
// Sometimes the path can be "../" if we are on the default section
// but it means we are just at the root of the site.
@@ -267,3 +292,16 @@ async function rewriteMarkdownLinks(
return tree;
}
function isMentionLike(node: Link) {
if (node.title === 'mention') {
return true;
}
const singleText =
node.children.length === 1 && node.children[0]?.type === 'text' ? node.children[0] : null;
if (!singleText) {
return false;
}
return singleText?.value === node.url;
}
+1 -1
View File
@@ -586,7 +586,7 @@ async function createContextForSpace(
}
/**
* When the API outputs markdown, it can sometimes format the content-ref into a strings that can be parsed back.
* When the API outputs markdown with `format.markdown.refs: stable`, the content refs are formatted this way.
*/
export function resolveStringContentRef(src: string): ContentRef | null {
for (const resolver of Object.values(RESOLVERS)) {
+1 -1
View File
@@ -8,7 +8,7 @@ export function isRollout({
discriminator: string;
percentageRollout: number;
}): boolean {
if (process.env.NODE_ENV === 'development') {
if (process.env.NODE_ENV === 'development' || process.env.VERCEL_ENV === 'preview') {
return true;
}
@@ -1,444 +0,0 @@
import { describe, expect, it, mock } from 'bun:test';
import type { GitBookSiteContext } from '@/lib/context';
import type { SiteSpace } from '@gitbook/api';
import { streamMarkdownFromSiteSpaces } from './llms-full';
function createMockLinker(args?: { spaceBasePath?: string }) {
return {
toAbsoluteURL: mock((path: string) => `https://example.com${path}`),
toPathInSite: mock((path: string) => `/site/${args?.spaceBasePath ?? ''}${path}`),
fork: (args: { spaceBasePath: string }) => createMockLinker(args),
};
}
describe('streamMarkdownFromSiteSpaces', () => {
// Test with real mocks of the dependencies
it('processes pages correctly with pagination', async () => {
// Mock the dependencies by replacing them in the module
const mockDataFetcher = {
getRevision: mock(() =>
Promise.resolve({
data: {
id: 'revision-1',
pages: [
{
id: 'page-1',
type: 'document',
title: 'Page 1',
path: 'page-1',
pages: [],
hidden: false,
},
{
id: 'page-2',
type: 'document',
title: 'Page 2',
path: 'page-2',
pages: [],
hidden: false,
},
{
id: 'page-3',
type: 'document',
title: 'Page 3',
path: 'page-3',
pages: [],
hidden: false,
},
{
id: 'page-4',
type: 'document',
title: 'Page 4',
path: 'page-4',
pages: [],
hidden: false,
},
{
id: 'page-5',
type: 'document',
title: 'Page 5',
path: 'page-5',
pages: [],
hidden: false,
},
],
},
})
),
getRevisionPageMarkdown: mock(() =>
Promise.resolve({
data: '# Test Page\n\nSome content\n',
})
),
};
const mockContext: GitBookSiteContext = {
dataFetcher: mockDataFetcher,
linker: createMockLinker(),
} as unknown as GitBookSiteContext;
const mockSiteSpace: SiteSpace = {
id: 'space-1',
space: {
id: 'space-1',
revision: 'rev-1',
},
urls: {
published: 'https://example.com',
},
path: 'test-space',
} as SiteSpace;
// Capture stream output
const chunks: string[] = [];
const mockController = {
enqueue: mock((chunk: Uint8Array) => {
chunks.push(new TextDecoder().decode(chunk));
}),
} as unknown as ReadableStreamDefaultController<Uint8Array>;
const result = await streamMarkdownFromSiteSpaces(
mockContext,
mockController,
[mockSiteSpace],
'base-path',
0,
0
);
// Verify results
expect(result.currentPageIndex).toBe(5); // Should process 5 pages
expect(result.reachedLimit).toBe(false); // Under limit
expect(chunks.length).toBe(5); // Should have 5 markdown chunks
expect(mockDataFetcher.getRevision).toHaveBeenCalledTimes(1);
expect(mockDataFetcher.getRevisionPageMarkdown).toHaveBeenCalledTimes(5);
});
it('applies offset correctly', async () => {
const mockDataFetcher = {
getRevision: mock(() =>
Promise.resolve({
data: {
pages: Array.from({ length: 10 }, (_, i) => ({
id: `page-${i + 1}`,
type: 'document',
title: `Page ${i + 1}`,
path: `page-${i + 1}`,
pages: [],
hidden: false,
})),
},
})
),
getRevisionPageMarkdown: mock(() => Promise.resolve({ data: 'content\n' })),
};
const mockContext: GitBookSiteContext = {
dataFetcher: mockDataFetcher,
linker: createMockLinker(),
} as unknown as GitBookSiteContext;
const mockSiteSpace: SiteSpace = {
space: { id: 'space-1', revision: 'rev-1' },
urls: { published: 'https://example.com' },
path: 'test-space',
} as SiteSpace;
const chunks: string[] = [];
const mockController = {
enqueue: mock((chunk: Uint8Array) => {
chunks.push(new TextDecoder().decode(chunk));
}),
} as unknown as ReadableStreamDefaultController<Uint8Array>;
const result = await streamMarkdownFromSiteSpaces(
mockContext,
mockController,
[mockSiteSpace],
'base-path',
3, // offset = 3
0
);
// Should process pages from index 3 onwards (7 pages)
expect(result.currentPageIndex).toBe(10);
expect(chunks.length).toBe(7); // 10 total - 3 offset = 7 processed
});
it('handles pagination when there are more than 100 pages', async () => {
const mockDataFetcher = {
getRevision: mock(() =>
Promise.resolve({
data: {
pages: Array.from({ length: 150 }, (_, i) => ({
id: `page-${i + 1}`,
type: 'document',
title: `Page ${i + 1}`,
path: `page-${i + 1}`,
pages: [],
hidden: false,
})),
},
})
),
getRevisionPageMarkdown: mock(() => Promise.resolve({ data: 'content\n' })),
};
const mockContext: GitBookSiteContext = {
dataFetcher: mockDataFetcher,
linker: createMockLinker(),
} as unknown as GitBookSiteContext;
const mockSiteSpace: SiteSpace = {
space: { id: 'space-1', revision: 'rev-1' },
urls: { published: 'https://example.com' },
path: 'test-space',
} as SiteSpace;
const chunks: string[] = [];
const mockController = {
enqueue: mock((chunk: Uint8Array) => {
chunks.push(new TextDecoder().decode(chunk));
}),
} as unknown as ReadableStreamDefaultController<Uint8Array>;
const result = await streamMarkdownFromSiteSpaces(
mockContext,
mockController,
[mockSiteSpace],
'base-path',
0,
0
);
// Should only process 100 pages (default limit)
expect(result.currentPageIndex).toBe(100);
expect(result.reachedLimit).toBe(true);
expect(chunks.length).toBe(101); // 100 pages + 1 next page link
// Check that next page link is included
const fullContent = chunks.join('');
expect(fullContent).toContain('[Next Page]');
expect(fullContent).toContain('/site/llms-full.txt/1');
});
it('handles multiple site spaces', async () => {
const mockDataFetcher = {
getRevision: mock()
.mockReturnValueOnce(
Promise.resolve({
data: {
pages: [
{
id: 'page-1',
type: 'document',
title: 'Space 1 Page 1',
path: 'page-1',
pages: [],
hidden: false,
},
{
id: 'page-2',
type: 'document',
title: 'Space 1 Page 2',
path: 'page-2',
pages: [],
hidden: false,
},
],
},
})
)
.mockReturnValueOnce(
Promise.resolve({
data: {
pages: [
{
id: 'page-3',
type: 'document',
title: 'Space 2 Page 1',
path: 'page-3',
pages: [],
hidden: false,
},
{
id: 'page-4',
type: 'document',
title: 'Space 2 Page 2',
path: 'page-4',
pages: [],
hidden: false,
},
{
id: 'page-5',
type: 'document',
title: 'Space 2 Page 3',
path: 'page-5',
pages: [],
hidden: false,
},
],
},
})
),
getRevisionPageMarkdown: mock(() => Promise.resolve({ data: 'content\n' })),
};
const mockContext: GitBookSiteContext = {
dataFetcher: mockDataFetcher,
linker: createMockLinker(),
} as unknown as GitBookSiteContext;
const mockSiteSpaces: SiteSpace[] = [
{
space: { id: 'space-1', revision: 'rev-1' },
urls: { published: 'https://example1.com' },
path: 'space-1',
},
{
space: { id: 'space-2', revision: 'rev-2' },
urls: { published: 'https://example2.com' },
path: 'space-2',
},
] as SiteSpace[];
const chunks: string[] = [];
const mockController = {
enqueue: mock((chunk: Uint8Array) => {
chunks.push(new TextDecoder().decode(chunk));
}),
} as unknown as ReadableStreamDefaultController<Uint8Array>;
const { streamMarkdownFromSiteSpaces } = await import('./llms-full');
const result = await streamMarkdownFromSiteSpaces(
mockContext,
mockController,
mockSiteSpaces,
'base-path',
0,
0
);
// Should process all pages from both spaces (2 + 3 = 5)
expect(result.currentPageIndex).toBe(5);
expect(result.reachedLimit).toBe(false);
expect(chunks.length).toBe(5);
expect(mockDataFetcher.getRevision).toHaveBeenCalledTimes(2);
});
it('skips site spaces without published URLs', async () => {
const mockDataFetcher = {
getRevision: mock(),
getRevisionPageMarkdown: mock(),
};
const mockContext: GitBookSiteContext = {
dataFetcher: mockDataFetcher,
linker: createMockLinker(),
} as unknown as GitBookSiteContext;
const mockSiteSpace: SiteSpace = {
space: { id: 'space-1', revision: 'rev-1' },
urls: { published: undefined }, // No published URL
path: 'test-space',
} as SiteSpace;
const mockController = {
enqueue: mock(),
} as unknown as ReadableStreamDefaultController<Uint8Array>;
const result = await streamMarkdownFromSiteSpaces(
mockContext,
mockController,
[mockSiteSpace],
'base-path',
0,
0
);
// Should not process any pages
expect(result.currentPageIndex).toBe(0);
expect(result.reachedLimit).toBe(false);
expect(mockDataFetcher.getRevision).not.toHaveBeenCalled();
});
it('filters only document type pages', async () => {
const mockDataFetcher = {
getRevision: mock(() =>
Promise.resolve({
data: {
pages: [
{
id: 'doc-1',
type: 'document',
title: 'Document 1',
path: 'doc-1',
pages: [],
hidden: false,
},
{
id: 'group-1',
type: 'group',
title: 'Group 1',
path: 'group-1',
pages: [],
hidden: false,
},
{
id: 'doc-2',
type: 'document',
title: 'Document 2',
path: 'doc-2',
pages: [],
hidden: false,
},
{
id: 'link-1',
type: 'link',
title: 'Link 1',
path: 'link-1',
pages: [],
hidden: false,
},
],
},
})
),
getRevisionPageMarkdown: mock(() => Promise.resolve({ data: 'content\n' })),
};
const mockContext: GitBookSiteContext = {
dataFetcher: mockDataFetcher,
linker: createMockLinker(),
} as unknown as GitBookSiteContext;
const mockSiteSpace: SiteSpace = {
space: { id: 'space-1', revision: 'rev-1' },
urls: { published: 'https://example.com' },
path: 'test-space',
} as SiteSpace;
const chunks: string[] = [];
const mockController = {
enqueue: mock((chunk: Uint8Array) => {
chunks.push(new TextDecoder().decode(chunk));
}),
} as unknown as ReadableStreamDefaultController<Uint8Array>;
const result = await streamMarkdownFromSiteSpaces(
mockContext,
mockController,
[mockSiteSpace],
'base-path',
0,
0
);
// Should only process the 2 document pages
expect(result.currentPageIndex).toBe(2);
expect(chunks.length).toBe(2);
expect(mockDataFetcher.getRevisionPageMarkdown).toHaveBeenCalledTimes(2);
});
});
+18 -36
View File
@@ -1,7 +1,10 @@
import { type GitBookSiteContext, checkIsRootSiteContext } from '@/lib/context';
import {
type GitBookSiteContext,
checkIsRootSiteContext,
fetchSiteContextForSiteSpace,
} from '@/lib/context';
import { throwIfDataError } from '@/lib/data';
import { fromPageMarkdown, toPageMarkdown } from '@/lib/markdownPage';
import { joinPath } from '@/lib/paths';
import { getIndexablePages } from '@/lib/sitemap';
import { filterSiteSpacesByLocale, getSiteStructureSections } from '@/lib/sites';
import type { RevisionPageDocument, SiteSection, SiteSpace } from '@gitbook/api';
@@ -63,7 +66,6 @@ async function streamMarkdownFromSiteStructure(
context,
stream,
context.structure.structure,
'',
offset
);
return;
@@ -88,7 +90,6 @@ async function streamMarkdownFromSections(
context,
stream,
siteSection.siteSpaces,
siteSection.path,
offset,
currentPageIndex
);
@@ -107,16 +108,13 @@ export async function streamMarkdownFromSiteSpaces(
context: GitBookSiteContext,
stream: ReadableStreamDefaultController<Uint8Array>,
siteSpaces: SiteSpace[],
basePath: string,
offset = 0,
initialPageIndex = 0
): Promise<{ currentPageIndex: number; reachedLimit: boolean }> {
const { dataFetcher } = context;
let totalPagesProcessed = initialPageIndex;
// Collect all pages first
const allPages: Array<{ page: RevisionPageDocument; siteSpace: SiteSpace; basePath: string }> =
[];
const allPages: Array<{ context: GitBookSiteContext; page: RevisionPageDocument }> = [];
const filteredSiteSpaces = filterSiteSpacesByLocale(siteSpaces, context.locale);
@@ -125,21 +123,15 @@ export async function streamMarkdownFromSiteSpaces(
if (!siteSpaceUrl) {
continue;
}
const revision = await throwIfDataError(
dataFetcher.getRevision({
spaceId: siteSpace.space.id,
revisionId: siteSpace.space.revision,
})
);
const pages = getIndexablePages(revision.pages);
const siteSpaceContext = await fetchSiteContextForSiteSpace(context, siteSpace);
const pages = getIndexablePages(siteSpaceContext.revision.pages);
// Add document pages to our collection
for (const { page } of pages) {
if (page.type === 'document') {
allPages.push({
context: siteSpaceContext,
page,
siteSpace,
basePath,
});
}
}
@@ -152,8 +144,8 @@ export async function streamMarkdownFromSiteSpaces(
// Process the pages
for await (const markdown of pMapIterable(
pagesToProcess,
async ({ page, siteSpace, basePath }) => {
return getMarkdownForPage(context, siteSpace, page, basePath);
async ({ context: siteSpaceContext, page }) => {
return getMarkdownForPage(siteSpaceContext, page);
},
{
concurrency: MAX_CONCURRENCY,
@@ -180,32 +172,22 @@ export async function streamMarkdownFromSiteSpaces(
*/
async function getMarkdownForPage(
context: GitBookSiteContext,
siteSpace: SiteSpace,
page: RevisionPageDocument,
basePath: string
page: RevisionPageDocument
): Promise<string> {
const { dataFetcher } = context;
const pageMarkdown = await throwIfDataError(
dataFetcher.getRevisionPageMarkdown({
spaceId: siteSpace.space.id,
revisionId: siteSpace.space.revision,
spaceId: context.space.id,
revisionId: context.revisionId,
pageId: page.id,
})
);
const tree = await fromPageMarkdown(
{
...context,
linker: context.linker.fork({
spaceBasePath: joinPath(context.linker.siteBasePath, basePath),
}),
},
{
markdown: pageMarkdown,
pagePath: page.path,
}
);
const tree = await fromPageMarkdown(context, {
markdown: pageMarkdown,
pagePath: page.path,
});
if (page.description) {
// The first node is the page title as a H1, we insert the description as a paragraph
+13 -2
View File
@@ -3,7 +3,12 @@ import { throwIfDataError } from '@/lib/data';
import { type GitBookLinker, linkerWithMarkdownPages } from '@/lib/links';
import { resolveFirstDocument } from '@/lib/pages';
import { type FlatPageEntry, getIndexablePages } from '@/lib/sitemap';
import { filterSiteSpacesByLocale, getLocalizedTitle, getSiteStructureSections } from '@/lib/sites';
import {
filterSiteSpacesByLocale,
getFallbackSiteSpacePath,
getLocalizedTitle,
getSiteStructureSections,
} from '@/lib/sites';
import type { SiteSection, SiteSpace } from '@gitbook/api';
import assertNever from 'assert-never';
import type { ListItem, Paragraph, Root, RootContent } from 'mdast';
@@ -143,8 +148,14 @@ async function getNodesFromSiteSpaces(
});
}
const siteSpaceLinker = linkerWithMarkdownPages(
linker.withOtherSiteSpace({
spaceBasePath: getFallbackSiteSpacePath(context, siteSpace),
})
);
// Add the pages as a list
nodes.push(...(await getMarkdownForPagesTree(pages, linker)));
nodes.push(...(await getMarkdownForPagesTree(pages, siteSpaceLinker)));
return nodes;
})
+64 -21
View File
@@ -12,6 +12,18 @@ describe('llms.txt', () => {
expect(await response.text()).toContain('# E2E Tests GitBook Open');
});
it('should properly format links', async () => {
const response = await fetch(
getContentTestURL('https://gitbook-open-e2e-sites.gitbook.io/sections/llms.txt')
);
expect(response.status).toBe(200);
expect(response.headers.get('content-type')).toContain('text/markdown');
const content = await response.text();
expect(content).toContain('/sections/sections-3/readme.md');
expect(content).toContain('/sections/sections-4/getting-started/quickstart.md');
});
it('should expose a llms.txt file with the accept header', async () => {
const response = await fetch(
getContentTestURL('https://gitbook.gitbook.io/test-gitbook-open/llms.txt'),
@@ -49,28 +61,59 @@ describe('llms.txt', () => {
});
describe('llms-full.txt', () => {
it('should expose a llms-full.txt file', async () => {
const response = await fetch(
getContentTestURL('https://gitbook.gitbook.io/test-gitbook-open/llms-full.txt')
);
it(
'should expose a llms-full.txt file',
async () => {
const response = await fetch(
getContentTestURL('https://gitbook.gitbook.io/test-gitbook-open/llms-full.txt')
);
expect(response.status).toBe(200);
expect(response.headers.get('content-type')).toContain('text/markdown');
expect(await response.text()).toContain('# Welcome');
});
expect(response.status).toBe(200);
expect(response.headers.get('content-type')).toContain('text/markdown');
expect(await response.text()).toContain('# Welcome');
},
{ timeout: 30_000 }
);
it('should expose a llms-full.txt file with the accept header', async () => {
const response = await fetch(
getContentTestURL('https://gitbook.gitbook.io/test-gitbook-open/llms-full.txt'),
{
headers: {
Accept: 'text/markdown',
},
}
);
it(
'should expose cross-space pages from a multi-version site',
async () => {
const response = await fetch(
getContentTestURL(
'https://gitbook-open-e2e-sites.gitbook.io/api-multi-versions-share-links/8tNo6MeXg7CkFMzSSz81/llms-full.txt'
)
);
const text = await response.text();
expect(response.status).toBe(200);
expect(response.headers.get('content-type')).toContain('text/markdown');
expect(await response.text()).toContain('# Welcome');
});
expect(response.status).toBe(200);
expect(response.headers.get('content-type')).toContain('text/markdown');
expect(text).toContain(
'gitbook-open-e2e-sites.gitbook.io/api-multi-versions-share-links/8tNo6MeXg7CkFMzSSz81/2.0/quick-start'
);
expect(text).toContain(
'gitbook-open-e2e-sites.gitbook.io/api-multi-versions-share-links/8tNo6MeXg7CkFMzSSz81/3.0/other-page'
);
expect(text).not.toContain('broken://');
},
{ timeout: 30_000 }
);
it(
'should expose a llms-full.txt file with the accept header',
async () => {
const response = await fetch(
getContentTestURL('https://gitbook.gitbook.io/test-gitbook-open/llms-full.txt'),
{
headers: {
Accept: 'text/markdown',
},
}
);
expect(response.status).toBe(200);
expect(response.headers.get('content-type')).toContain('text/markdown');
expect(await response.text()).toContain('# Welcome');
},
{ timeout: 30_000 }
);
});
+12
View File
@@ -42,6 +42,18 @@ describe('markdown pages', () => {
expect(response.headers.get('x-robots-tag')).toBe('noindex');
expect(text).toContain('# Page Not Found');
});
it('should rewrite links to markdown URLs', async () => {
const response = await fetch(
getContentTestURL('https://gitbook.gitbook.io/test-gitbook-open/blocks/links.md')
);
const text = await response.text();
expect(response.status).toBe(200);
expect(response.headers.get('content-type')).toContain('text/markdown');
expect(response.headers.get('x-robots-tag')).toBe('noindex');
expect(text).toContain('gitbook.gitbook.io/test-gitbook-open/text-page.md');
});
});
describe('markdown ask responses', () => {
+31
View File
@@ -53,3 +53,34 @@ it(
},
{ timeout: 10_000 }
);
it(
'should get a page from another site space through MCP',
async () => {
const client = new Client({
name: 'test',
version: '1.0.0',
});
await client.connect(
new StreamableHTTPClientTransport(
new URL(
getContentTestURL(
'https://gitbook-open-e2e-sites.gitbook.io/api-multi-versions-share-links/8tNo6MeXg7CkFMzSSz81/~gitbook/mcp/auth'
)
)
)
);
const response = await client.callTool({
name: 'getPage',
arguments: {
url: 'https://gitbook-open-e2e-sites.gitbook.io/api-multi-versions-share-links/8tNo6MeXg7CkFMzSSz81/3.0/other-page',
},
});
// @ts-expect-error - response.content is of type unknown
expect(response.content[0]?.text).toContain('# Other Page');
},
{ timeout: 15_000 }
);