Change to use DuckDB instead of data-forge for data analisis

This commit is contained in:
Jorge Cabiedes Acosta
2025-06-11 10:18:53 -07:00
parent 3f2b86b5b1
commit b3f66b9dd2
5 changed files with 659 additions and 200 deletions
+1 -1
View File
@@ -9,7 +9,7 @@ dist
!packages/playground/.vscode
testfilter.txt
packages/react-mcp-server/artifacts
packages/react-mcp-server/src/artifacts
# forgive
*.vsix
@@ -21,8 +21,7 @@
"@modelcontextprotocol/sdk": "^1.9.0",
"algoliasearch": "^5.23.3",
"cheerio": "^1.0.0",
"data-forge": "^1.10.4",
"data-forge-fs": "^0.0.9",
"duckdb": "^1.3.0",
"html-to-text": "^9.0.5",
"prettier": "^3.3.3",
"puppeteer": "^24.7.2",
+17 -15
View File
@@ -512,34 +512,36 @@ server.tool(
'interpret-react-performance-data',
`
<description>
Pass in a Javascript script using the data-forge library to analyze the performance data captured by the start-react-performance-recording tool.
The script should use the data-forge library to process the data and provide insights and solutions to the user.
Your scripts should be very atomical so you can:
Pass in a SQL query to analyze the performance data captured by the start-react-performance-recording tool.
The query will be executed against a DuckDB database containing the performance data.
Your queries should be focused and specific to extract meaningful insights from the data.
</description>
<requirements>
- On your first iteration your script should just try to understand the shape of the dataFrames object
- the script should return a string with the analysis at the end
- Do not use console.log this will only cause noise
- only analyze the data frames within the \`dataFrames\` parameter, you should assume you always receive this parameter.
- Remember to ask yourself 10 questions about the dataframe and then narrow down to 5 most important about the dataset
- dataFrames should never use .get method
- On your first iteration your query should just try to understand the structure of the tables
- The query should return a result set that provides meaningful analysis
- Queries should be well-structured and optimized for performance
- Remember to ask yourself 10 questions about the data and then narrow down to the 5 most important about the dataset
</requirements>
Allowed Actions
1. Print Results: Output will be displayed as the script's stdout.
1. Print Results: Output will be displayed as the query result.
Prohibited Actions
1. Overwriting Original DataFrames: Do not modify existing DataFrames to preserve their integrity for future tasks.
1. Modifying the database schema or data: Queries should be read-only.
2. Creating Charts: Chart generation is not permitted.
3. Using console.log()
When using this tool you are a senior React Performance Engineer tasked with performing exploratory data analysis on a dataset of React performance data. Your goal is to provide insightful analysis while ensuring stability and manageable result sizes.
<usage>
Your script will be passed into a Function() contructor, you should only use data-forge import, the function will receive the paremeters:
- dataForge - the dataFroge import
- dataFrames - A map of the currently loaded dataFrames meant to be analyzed by the tool. This is just a \`Record<string, any>\`
Your SQL query will be executed against a DuckDB database containing the following tables:
- component_track: Contains data from the Components track in React DevTools
- scheduler_track: Contains data from the Scheduler track in React DevTools
Example queries:
- "SELECT * FROM component_track LIMIT 10" - View the first 10 rows of the component track data
- "SELECT name, COUNT(*) as render_count FROM component_track GROUP BY name ORDER BY render_count DESC LIMIT 10" - Find the components with the most renders
- "SELECT name, AVG(endTime - startTime) as avg_duration FROM component_track GROUP BY name ORDER BY avg_duration DESC LIMIT 10" - Find the components with the longest average render time
</usage>
`,
{
@@ -1,8 +1,7 @@
import {hookIntoPage} from '../utils/puppeteerUtils';
import fs from 'fs/promises';
import path from 'path';
import dataForge from 'data-forge';
import {readFileSync} from 'data-forge-fs';
import * as duckdb from 'duckdb';
/**
* Connects to a browser and patches the console.timeStamp method to capture data
@@ -24,7 +23,8 @@ export async function beginPerfRecording(url: string): Promise<string> {
// Monkey-patch console.timeStamp to capture data
console.timeStamp = function (...args: any) {
if ((window as any).__MCP_RECORDING_ACTIVE__) {
// Filter out 0 endTimes, this is used in devtools to hide duplicated render blocks but is noise for us
if ((window as any).__MCP_RECORDING_ACTIVE__ && args[2] !== 0) {
if ((args[4] as string)?.includes('Scheduler')) {
const timeStampData = {
name: args[0],
@@ -43,7 +43,7 @@ export async function beginPerfRecording(url: string): Promise<string> {
startTime: args[1],
endTime: args[2],
track: 'Components',
color: args[4],
color: args[5],
};
(window as any).__COMPONENT_TRACK_TIMESTAMP_DATA__.push(
timeStampData,
@@ -138,7 +138,7 @@ export async function beginPerfRecording(url: string): Promise<string> {
async function processTrackData(
trackData: any,
headers: string[],
): Promise<boolean> {
): Promise<string | null> {
const artifactsDir = path.join(__dirname, '../src/artifacts');
if (Array.isArray(trackData) && trackData.length > 0) {
@@ -168,12 +168,24 @@ async function processTrackData(
await fs.writeFile(componentFilePath, componentCsvContent, 'utf8');
return true;
return componentFilePath;
}
return false;
return null;
}
// Create a singleton DuckDB connection
let db: duckdb.Database | null = null;
let conn: duckdb.Connection | null = null;
function getDuckDB(): { db: duckdb.Database, conn: duckdb.Connection } {
if (!db) {
const dbPath = path.join(__dirname, '../src/artifacts/perf_data.db');
db = new duckdb.Database(dbPath);
conn = db.connect();
}
return { db, conn: conn! };
}
const dataFrames: Record<string, any> = {};
export async function getPerfData(url: string): Promise<string[]> {
try {
const page = await hookIntoPage(url);
@@ -204,16 +216,21 @@ export async function getPerfData(url: string): Promise<string[]> {
}
const result = [];
const csvFiles = [];
if (
await processTrackData(componentTrackData, [
'name',
'startTime',
'endTime',
'track',
'color',
])
) {
const componentCsvPath = await processTrackData(componentTrackData, [
'name',
'startTime',
'endTime',
'track',
'color',
]);
if (componentCsvPath) {
csvFiles.push({
path: componentCsvPath,
tableName: 'component_track',
});
result.push(
'Component track data saved in the artifacts directory it can now be accessed through the interpret-perf-data tool.',
);
@@ -221,16 +238,20 @@ export async function getPerfData(url: string): Promise<string[]> {
result.push('No component track data was available to save.');
}
if (
await processTrackData(schedulerTrackData, [
'name',
'startTime',
'endTime',
'type',
'track',
'color',
])
) {
const schedulerCsvPath = await processTrackData(schedulerTrackData, [
'name',
'startTime',
'endTime',
'type',
'track',
'color',
]);
if (schedulerCsvPath) {
csvFiles.push({
path: schedulerCsvPath,
tableName: 'scheduler_track',
});
result.push(
'Scheduler track data saved in the artifacts directory it can now be accessed through the interpret-perf-data tool.',
);
@@ -238,18 +259,17 @@ export async function getPerfData(url: string): Promise<string[]> {
result.push('No scheduler track data was available to save.');
}
// Get all CSV files in the artifacts directory
const files = await fs.readdir(artifactsDir);
const csvFiles = files.filter((file: string) => file.endsWith('.csv'));
// Initialize DuckDB and load CSV files
const { conn } = getDuckDB();
for (const file of csvFiles) {
const filePath = path.join(artifactsDir, file);
const df = readFileSync(filePath).parseCSV();
const dfName = file.replace(/\.csv$/, '').replace(/[^a-zA-Z0-9]/g, '_');
dataFrames[dfName] = df;
// Create tables and load data from CSV files
for (const csvFile of csvFiles) {
// Create table based on CSV structure
conn.exec(`DROP TABLE IF EXISTS ${csvFile.tableName}`);
conn.exec(`CREATE TABLE ${csvFile.tableName} AS SELECT * FROM read_csv_auto('${csvFile.path}')`);
}
result.push('DataFrames loaded successfully.');
result.push('DuckDB database created and data loaded successfully.');
return result;
} catch (error) {
@@ -258,21 +278,63 @@ export async function getPerfData(url: string): Promise<string[]> {
}
/**
* Executes a JavaScript script that uses the data-frame library to analyze CSV files
* in the artifacts directory.
* Executes a SQL query against the DuckDB database containing performance data.
*
* @param script The JavaScript script to execute
* @param saveToMemory Optional array of dataframe names to save to memory
* @returns A promise that resolves to the result of the script execution
* @param query The SQL query to execute
* @returns A promise that resolves to the result of the query execution
*/
export async function executeDataFrameScript(script: string): Promise<string> {
export async function executeDataFrameScript(query: string): Promise<string> {
try {
const executeScript = new Function('dataForge', 'dataFrames', script);
const { conn } = getDuckDB();
const result = executeScript(dataForge, dataFrames);
// Check if tables exist by querying the information schema
const tablesExist = await new Promise<boolean>((resolve) => {
conn.all(
`SELECT table_name FROM information_schema.tables
WHERE table_name IN ('component_track', 'scheduler_track')`,
(err: Error | null, result: any[]) => {
if (err || !result || result.length === 0) {
resolve(false);
} else {
resolve(true);
}
}
);
});
return result;
if (!tablesExist) {
return `No performance data tables found. Please run the process-react-performance-data tool first to capture and load performance data.
Example workflow:
1. Run 'start-react-performance-recording' to begin capturing data
2. Interact with your React application to generate performance data
3. Run 'process-react-performance-data' to process and load the data into DuckDB
4. Then run SQL queries using this tool`;
}
const result = await new Promise<any>((resolve, reject) => {
conn.all(query, (err: Error | null, result: any) => {
if (err) {
reject(err);
} else {
resolve(result);
}
});
});
if (Array.isArray(result)) {
if (result.length === 0) {
return 'Query executed successfully, but returned no results.';
}
const headers = Object.keys(result[0]);
const rows = result.map(row => headers.map(h => JSON.stringify(row[h])).join(','));
return [headers.join(','), ...rows].join('\n');
} else {
return JSON.stringify(result, null, 2);
}
} catch (error) {
throw new Error(`Error running script: ${error.message}`);
throw new Error(`Error running SQL query: ${error.message}`);
}
}
+532 -136
View File
File diff suppressed because it is too large Load Diff