docs: add llm txt (#5009)

This commit is contained in:
Alireza authored and GitHub committed 2025-05-01 18:13:21 -04:00
1 parent 4732e8093e
commit 4e185e4b21
149 files changed
+865 -7

No files matched your search

+101
View File
@@ -0,0 +1,101 @@
# OHIF Viewer Documentation Scripts
This directory contains utility scripts for the OHIF Viewer documentation.
## prepare-markdown-files.js
This script copies all markdown files from the docs directory (excluding `api/` and `assets/`) to a `/llm` directory in the build output, making them available as static markdown files that can be accessed directly without the Docusaurus UI.
### Purpose
These files are made available for easy access by LLMs and other tools that need to retrieve the raw markdown content without the surrounding Docusaurus UI elements.
### Access URLs
After deployment, the markdown files can be accessed at URLs like:
- `https://docs.ohif.org/llm/platform/extensions/modules/commands.md`
- `https://docs.ohif.org/llm/platform/services/ui/index.md`
### Implementation
The script is automatically run as part of the `build:docs` command and will:
1. Find all markdown files in the `/platform/docs/docs/` directory (excluding api/ and assets/)
2. Copy them to `/platform/docs/build/llm/` preserving the directory structure
3. These files are then deployed to the website along with the rest of the built documentation
You can also run this script manually with:
```bash
cd platform/docs
yarn run prepare-markdown-files
```
## generate-llms-txt.js
This script generates a llms.txt file that follows the [llms.txt specification](https://llmstxt.site) by creating an index of all the markdown files that have been copied to the `/llm` directory.
### Purpose
The llms.txt file provides an overview of all available documentation in a format that's optimized for use with Large Language Models (LLMs). It creates a structured index that LLMs can use to efficiently navigate and find information in the documentation.
### Access URL
After deployment, the llms.txt file can be accessed at:
- `https://docs.ohif.org/llms.txt`
### Implementation
The script is automatically run as part of the `build:docs` command (after `prepare-markdown-files.js`) and will:
1. Scan all the markdown files that were copied to the `/llm` directory
2. Extract titles and summaries from the frontmatter of each file
3. Organize them into sections based on their directory structure
4. Generate a single llms.txt file in the standard format with links to all the documentation files
5. Save the file to the build directory root so it will be accessible at the root of the website
You can also run this script manually with (after running `prepare-markdown-files.js`):
```bash
cd platform/docs
yarn run generate-llms-txt
```
## generate-llms-full-txt.js
This script generates an llms-full.txt file by concatenating the content of all markdown files from the `/llm` directory into a single giant file, similar to the approach used by bun.sh.
### Purpose
The llms-full.txt file provides the complete content of all documentation in a single file, which can be useful for LLMs that need to search or reference the entire documentation at once. This approach allows for contextual understanding across multiple documentation pages.
### Access URL
After deployment, the llms-full.txt file can be accessed at:
- `https://docs.ohif.org/llms-full.txt`
### Implementation
The script is automatically run as part of the `build:docs` command (after `generate-llms-txt.js`) and will:
1. Scan all the markdown files that were copied to the `/llm` directory
2. Remove frontmatter from each file to clean up the content
3. Add clear section headers and source URLs for each file
4. Concatenate all the content into a single file with proper organization by section
5. Save the file to the build directory root so it will be accessible at the root of the website
The output file has the following structure:
- Title and introduction for OHIF Viewer
- Files are organized by their directory structure (e.g., platform, extensions, services)
- Each file has a header with its title and source URL
- Separator lines between files for clarity
You can also run this script manually with (after running `prepare-markdown-files.js`):
```bash
cd platform/docs
yarn run generate-llms-full-txt
```
@@ -0,0 +1,291 @@
/**
* Script to generate an llms-full.txt file that concatenates all markdown content
* This script creates a single giant file with all markdown content from the /llm directory
* with proper heading hierarchy preserved
*/
const fs = require('fs');
const path = require('path');
const { glob } = require('glob');
// Base docs directory
const baseDocsDir = path.join(__dirname, '../docs');
// Output directory for LLM markdown files
const llmDir = path.join(__dirname, '../build/llm');
// Output file path for llms-full.txt
const outputFilePath = path.join(__dirname, '../build/llms-full.txt');
// Get all subdirectories to create sections
async function getDirectorySections() {
const directories = new Set();
// Find all subdirectories that contain markdown files
const markdownFiles = await glob(`${llmDir}/**/*.md`);
markdownFiles.forEach((filePath) => {
// Get the relative directory from the llm dir
const relativePath = path.relative(llmDir, filePath);
const dirPath = path.dirname(relativePath);
// Add to set if it's not in the root
if (dirPath !== '.') {
// Get the top-level directory
const topDir = dirPath.split(path.sep)[0];
directories.add(topDir);
}
});
return Array.from(directories).sort();
}
// Get all subdirectories for a section
async function getSubdirectories(section) {
const sectionPath = path.join(llmDir, section);
const subdirectories = new Set();
// Find all markdown files in this section
const markdownFiles = await glob(`${sectionPath}/**/*.md`);
markdownFiles.forEach((filePath) => {
// Get the relative directory from the section path
const relativePath = path.relative(sectionPath, filePath);
const dirPath = path.dirname(relativePath);
// Add to set if it's not in the root
if (dirPath !== '.') {
subdirectories.add(dirPath);
}
});
return Array.from(subdirectories).sort();
}
// Process content to adjust heading levels
function adjustHeadingLevels(content, baseLevel) {
// Replace headings with adjusted levels while limiting max depth to h4
let processedContent = content;
// Process all heading levels from h1 to h6
for (let i = 1; i <= 6; i++) {
// Limit maximum heading level to h4
const newLevel = Math.min(i + baseLevel, 4);
const pattern = new RegExp(`^(#{${i}})\\s+(.+)$`, 'gm');
processedContent = processedContent.replace(pattern, `${'#'.repeat(newLevel)} $2`);
}
return processedContent;
}
// Process a single markdown file
function processMarkdownFile(filePath, baseLevel) {
// Read the file content
const content = fs.readFileSync(filePath, 'utf8');
// Process content to remove frontmatter
let processedContent = content;
const frontmatterMatch = content.match(/^---\n([\s\S]*?)\n---\n/);
if (frontmatterMatch) {
processedContent = content.replace(/^---\n[\s\S]*?\n---\n/, '');
}
// Adjust heading levels
processedContent = adjustHeadingLevels(processedContent, baseLevel);
return processedContent;
}
// Process all files in a subsection
async function processSubsectionFiles(section, subsection) {
const subsectionPath = path.join(llmDir, section, subsection);
const subsectionContent = [];
// Get all markdown files for this subsection
const markdownFiles = await glob(`${subsectionPath}/*.md`);
// Sort files alphabetically
markdownFiles.sort();
for (const filePath of markdownFiles) {
// Get the file name
const fileName = path.basename(filePath, '.md');
const relativePath = path.relative(llmDir, filePath);
// Extract title from frontmatter or filename
const content = fs.readFileSync(filePath, 'utf8');
let title = '';
const frontmatterTitleMatch = content.match(/title:\s*([^\n]+)/);
const firstHeadingMatch = content.match(/# ([^\n]+)/);
if (frontmatterTitleMatch) {
title = frontmatterTitleMatch[1].trim();
// Remove quotes if present
title = title.replace(/^["'](.*)["']$/, '$1');
} else if (firstHeadingMatch) {
title = firstHeadingMatch[1].trim();
} else {
// Use filename as fallback
title = fileName;
}
// Add file header (as h3, since section is h1 and subsection is h2)
subsectionContent.push(`\n\n### ${title}\n`);
subsectionContent.push(`Source: https://docs.ohif.org/llm/${relativePath}\n`);
// Process the file content with adjusted heading levels (h1->h4, h2->h5, etc.)
const processedContent = processMarkdownFile(filePath, 3);
subsectionContent.push(processedContent);
// Add separator
subsectionContent.push('\n\n---\n');
}
return subsectionContent.join('\n');
}
// Process files directly in the section root (not in subsections)
async function processSectionRootFiles(section) {
const sectionPath = path.join(llmDir, section);
const sectionContent = [];
// Get all markdown files directly in the section root
const markdownFiles = await glob(`${sectionPath}/*.md`);
// Sort files alphabetically
markdownFiles.sort();
for (const filePath of markdownFiles) {
// Get the file name
const fileName = path.basename(filePath, '.md');
const relativePath = path.relative(llmDir, filePath);
// Extract title from frontmatter or filename
const content = fs.readFileSync(filePath, 'utf8');
let title = '';
const frontmatterTitleMatch = content.match(/title:\s*([^\n]+)/);
const firstHeadingMatch = content.match(/# ([^\n]+)/);
if (frontmatterTitleMatch) {
title = frontmatterTitleMatch[1].trim();
// Remove quotes if present
title = title.replace(/^["'](.*)["']$/, '$1');
} else if (firstHeadingMatch) {
title = firstHeadingMatch[1].trim();
} else {
// Use filename as fallback
title = fileName;
}
// Add file header (as h2, since section is h1)
sectionContent.push(`\n\n## ${title}\n`);
sectionContent.push(`Source: https://docs.ohif.org/llm/${relativePath}\n`);
// Process the file content with adjusted heading levels (h1->h3, h2->h4, etc.)
const processedContent = processMarkdownFile(filePath, 2);
sectionContent.push(processedContent);
// Add separator
sectionContent.push('\n\n---\n');
}
return sectionContent.join('\n');
}
// Process root files
async function processRootFiles() {
const rootContent = [];
// Get all markdown files in the root directory
const rootFiles = await glob(`${llmDir}/*.md`);
// Sort files alphabetically
rootFiles.sort();
for (const filePath of rootFiles) {
// Get the file name
const fileName = path.basename(filePath, '.md');
// Extract title from frontmatter or filename
const content = fs.readFileSync(filePath, 'utf8');
let title = '';
const frontmatterTitleMatch = content.match(/title:\s*([^\n]+)/);
const firstHeadingMatch = content.match(/# ([^\n]+)/);
if (frontmatterTitleMatch) {
title = frontmatterTitleMatch[1].trim();
// Remove quotes if present
title = title.replace(/^["'](.*)["']$/, '$1');
} else if (firstHeadingMatch) {
title = firstHeadingMatch[1].trim();
} else {
// Use filename as fallback
title = fileName;
}
// Add file header (as h2, since root is h1)
rootContent.push(`\n\n## ${title}\n`);
rootContent.push(`Source: https://docs.ohif.org/llm/${fileName}\n`);
// Process the file content with adjusted heading levels (h1->h3, h2->h4, etc.)
const processedContent = processMarkdownFile(filePath, 2);
rootContent.push(processedContent);
// Add separator
rootContent.push('\n\n---\n');
}
return rootContent.join('\n');
}
// Generate the full concatenated content
async function generateLlmsFullTxt() {
let content = '';
// Add title and introduction
content += '# OHIF Documentation\n\n';
content += '> OHIF (Open Health Imaging Foundation) Viewer is an open-source, web-based, zero-footprint DICOM viewer platform designed for medical imaging. It provides a highly configurable and extensible framework for building diagnostic quality medical imaging applications. OHIF Viewer supports various imaging formats (primarily DICOM), offers advanced visualization tools, customizable workflows, and integration capabilities with different data sources.\n\n';
content += 'This file contains the complete documentation for OHIF Viewer, concatenated for easy reference and searching. Each section is clearly marked with its source URL.\n\n';
// Process root files first (if any)
const rootFiles = await glob(`${llmDir}/*.md`);
if (rootFiles.length > 0) {
content += '# Root Documentation\n\n';
// Process root files
const rootContent = await processRootFiles();
content += rootContent;
}
// Get all sections
const sections = await getDirectorySections();
// Process each section
for (const section of sections) {
// Add section header (as h1)
content += `\n\n# ${section.charAt(0).toUpperCase() + section.slice(1)}\n\n`;
// Process files directly in the section root
const sectionRootContent = await processSectionRootFiles(section);
content += sectionRootContent;
// Process subsections if any
const subsections = await getSubdirectories(section);
for (const subsection of subsections) {
// Add subsection header (as h2)
content += `\n\n## ${subsection.charAt(0).toUpperCase() + subsection.slice(1)}\n\n`;
// Process files in this subsection
const subsectionContent = await processSubsectionFiles(section, subsection);
content += subsectionContent;
}
}
// Write the llms-full.txt file
fs.writeFileSync(outputFilePath, content);
console.log(`Generated llms-full.txt file at ${outputFilePath}`);
}
// Run the script
generateLlmsFullTxt();
+142
View File
@@ -0,0 +1,142 @@
/**
* Script to generate an llms.txt file following the specification
* This script generates a single llms.txt file in the build directory
* with links to all the markdown files that have been copied to the /llm directory
*/
const fs = require('fs');
const path = require('path');
const { glob } = require('glob');
// Base docs directory
const baseDocsDir = path.join(__dirname, '../docs');
// Output directory for LLM markdown files
const llmDir = path.join(__dirname, '../build/llm');
// Output file path for llms.txt
const outputFilePath = path.join(__dirname, '../build/llms.txt');
// Get all subdirectories to create sections
async function getDirectorySections() {
const directories = new Set();
// Find all subdirectories that contain markdown files
const markdownFiles = await glob(`${llmDir}/**/*.md`);
markdownFiles.forEach((filePath) => {
// Get the relative directory from the llm dir
const relativePath = path.relative(llmDir, filePath);
const dirPath = path.dirname(relativePath);
// Add to set if it's not in the root
if (dirPath !== '.') {
// Get the top-level directory
const topDir = dirPath.split(path.sep)[0];
directories.add(topDir);
}
});
return Array.from(directories).sort();
}
// Generate links for a specific section
async function generateSectionLinks(section) {
const sectionPath = path.join(llmDir, section);
const links = [];
// Get all markdown files for this section
const markdownFiles = await glob(`${sectionPath}/**/*.md`);
for (const filePath of markdownFiles) {
// Get the relative path from the llm dir
const relativePath = path.relative(llmDir, filePath);
// Read the file to extract title
const content = fs.readFileSync(filePath, 'utf8');
// Try to extract title from frontmatter or first heading
let title = '';
const frontmatterTitleMatch = content.match(/title:\s*([^\n]+)/);
const firstHeadingMatch = content.match(/# ([^\n]+)/);
if (frontmatterTitleMatch) {
title = frontmatterTitleMatch[1].trim();
// Remove quotes if present
title = title.replace(/^["'](.*)["']$/, '$1');
} else if (firstHeadingMatch) {
title = firstHeadingMatch[1].trim();
} else {
// Use filename as fallback
title = path.basename(filePath, '.md');
}
// Get summary from frontmatter if available
let summary = '';
const summaryMatch = content.match(/summary:\s*([^\n]+)/);
if (summaryMatch) {
summary = summaryMatch[1].trim();
// Remove quotes if present
summary = summary.replace(/^["'](.*)["']$/, '$1');
}
// Create URL for the file (relative to site root)
const url = `/llm/${relativePath}`;
// Add to links
links.push({
title,
url,
summary,
});
}
return links.sort((a, b) => a.title.localeCompare(b.title));
}
// Generate the full llms.txt content
async function generateLlmsTxt() {
let content = '';
// Add title
content += '# OHIF Viewer\n\n';
// Add description blockquote
content +=
'> OHIF (Open Health Imaging Foundation) Viewer is an open-source, web-based, zero-footprint DICOM viewer platform designed for medical imaging. It provides a highly configurable and extensible framework for building diagnostic quality medical imaging applications. OHIF Viewer supports various imaging formats (primarily DICOM), offers advanced visualization tools, customizable workflows, and integration capabilities with different data sources.\n\n';
// Add general information
content +=
'The OHIF Viewer is built with a modular architecture that includes extensions, modes, and services. It leverages Cornerstone3D for rendering capabilities and provides a comprehensive framework for building medical imaging applications with features like hanging protocols, segmentation, measurements, and advanced visualization tools.\n\n';
content += 'The documentation is organized into the following sections:\n\n';
// Get all sections
const sections = await getDirectorySections();
// Process each section
for (const section of sections) {
// Get all links for this section
const links = await generateSectionLinks(section);
if (links.length > 0) {
// Add section header
content += `## ${section.charAt(0).toUpperCase() + section.slice(1)}\n\n`;
// Add links for this section
for (const link of links) {
content += `- [${link.title}](https://docs.ohif.org${link.url})`;
if (link.summary) {
content += `: ${link.summary}`;
}
content += '\n';
}
content += '\n';
}
}
// Write the llms.txt file
fs.writeFileSync(outputFilePath, content);
console.log(`Generated llms.txt file at ${outputFilePath}`);
}
// Run the script
generateLlmsTxt();
@@ -0,0 +1,67 @@
/**
* Script to copy and prepare markdown files for static hosting
* This script copies all markdown files from the docs directory (excluding api and assets)
* to a /llm directory where they can be accessed directly without Docusaurus rendering
*/
const fs = require('fs');
const path = require('path');
const { glob } = require('glob');
// Base docs directory
const baseDocsDir = path.join(__dirname, '../docs');
// Output directory for LLM markdown files
const outputDir = path.join(__dirname, '../build/llm');
// Create the output directory structure
function createDirectoryIfNotExists(directoryPath) {
if (!fs.existsSync(directoryPath)) {
fs.mkdirSync(directoryPath, { recursive: true });
}
}
// Process a markdown file
function processMarkdownFile(filePath) {
try {
// Read the content of the markdown file
const content = fs.readFileSync(filePath, 'utf8');
// Get the relative path from baseDocsDir
const relativePath = path.relative(baseDocsDir, filePath);
// Determine the output directory
const outputFilePath = path.join(outputDir, relativePath);
const outputFileDir = path.dirname(outputFilePath);
// Create the directory if it doesn't exist
createDirectoryIfNotExists(outputFileDir);
// Write the content to the new location
fs.writeFileSync(outputFilePath, content);
console.log(`Processed: ${relativePath}`);
} catch (error) {
console.error(`Error processing file ${filePath}:`, error);
}
}
// Main function
async function prepareMarkdownFiles() {
// Create the base output directory
createDirectoryIfNotExists(outputDir);
// Find all markdown files in the docs directory
const markdownFiles = await glob(`${baseDocsDir}/**/*.md`, {
ignore: [`${baseDocsDir}/api/**/*.md`, `${baseDocsDir}/assets/**/*.md`],
});
console.log(`Found ${markdownFiles.length} markdown files to process.`);
// Process each markdown file
markdownFiles.forEach(processMarkdownFile);
console.log('Markdown file preparation complete.');
}
// Run the script
prepareMarkdownFiles();