| stats_tool_schemas = [ |
| { |
| "name": "descriptive_stats_func", |
| "description": ( |
| "Computes summary statistics for numeric columns in query.csv: " |
| "count, mean, std, min, 25th/50th/75th percentile, and max. " |
| "Use when the user asks for summary statistics, descriptive statistics, or a statistical overview. " |
| "Returns a formatted HTML table." |
| ), |
| "parameters": { |
| "type": "object", |
| "properties": { |
| "columns": { |
| "type": "array", |
| "description": "Optional list of column names to include. If omitted, all numeric columns from query.csv are used. Avoid ID or index columns.", |
| "items": {"type": "string"}, |
| }, |
| }, |
| "required": [], |
| }, |
| }, |
| { |
| "name": "kmeans_clustering_func", |
| "description": ( |
| "Runs K-Means clustering on numeric feature columns from query.csv. " |
| "Groups rows into k clusters, displays a scatter plot coloured by cluster assignment, " |
| "and returns a centroid summary table showing the mean of each feature per cluster. " |
| "Use when the user asks to cluster the data, find natural segments or groups, or apply K-Means. " |
| "Returns an HTML iframe and summary table." |
| ), |
| "parameters": { |
| "type": "object", |
| "properties": { |
| "feature_columns": { |
| "type": "array", |
| "description": "List of numeric column names from query.csv to use as clustering features.", |
| "items": {"type": "string"}, |
| }, |
| "x_column": { |
| "type": "string", |
| "description": "Column name from query.csv for the x-axis of the scatter plot. Usually one of the feature columns.", |
| }, |
| "y_column": { |
| "type": "string", |
| "description": "Column name from query.csv for the y-axis of the scatter plot. Usually one of the feature columns.", |
| }, |
| "n_clusters": { |
| "type": "integer", |
| "description": "Number of clusters (k). Default 3. Infer from the user's request.", |
| }, |
| "layout": { |
| "type": "array", |
| "description": "Optional. An array containing a single JSON-formatted Plotly layout dictionary.", |
| "items": {"type": "string"}, |
| }, |
| }, |
| "required": ["feature_columns", "x_column", "y_column"], |
| }, |
| }, |
| { |
| "name": "hypothesis_test_func", |
| "description": ( |
| "Performs a statistical hypothesis test on query.csv data and returns a formatted results table " |
| "with test statistic, p-value, and significance at α=0.05. " |
| "Supported tests:\n" |
| "- 't_test_independent': compare means of a numeric column across two groups " |
| "(requires group_column; use group_values if the column has more than 2 unique values).\n" |
| "- 't_test_one_sample': test whether a column's mean equals a hypothesized value (requires pop_mean).\n" |
| "- 'chi_square': test independence between two categorical columns (requires column and column2)." |
| ), |
| "parameters": { |
| "type": "object", |
| "properties": { |
| "test_type": { |
| "type": "string", |
| "description": "Test to run. One of: 't_test_independent', 't_test_one_sample', 'chi_square'.", |
| }, |
| "column": { |
| "type": "string", |
| "description": "Primary column for the test. Numeric for t-tests; first categorical column for chi-square.", |
| }, |
| "column2": { |
| "type": "string", |
| "description": "Second categorical column. Required for 'chi_square'.", |
| }, |
| "group_column": { |
| "type": "string", |
| "description": "Grouping column. Required for 't_test_independent'. Must have exactly 2 unique values, or specify group_values.", |
| }, |
| "group_values": { |
| "type": "array", |
| "description": "Exactly 2 group labels to compare. Use when group_column has more than 2 unique values.", |
| "items": {"type": "string"}, |
| }, |
| "pop_mean": { |
| "type": "number", |
| "description": "Hypothesized population mean (μ₀). Required for 't_test_one_sample'.", |
| }, |
| }, |
| "required": ["test_type", "column"], |
| }, |
| }, |
| { |
| "name": "regression_func", |
| "description": ( |
| "Runs an OLS linear regression on query.csv data. " |
| "Use when the user wants to model the relationship between variables, assess predictors, or run a regression. " |
| "Returns a regression summary (coefficients, R², p-values) and a scatter plot with the fitted line as an HTML iframe." |
| ), |
| "parameters": { |
| "type": "object", |
| "properties": { |
| "independent_variables": { |
| "type": "array", |
| "description": "Column names from query.csv to use as independent (predictor) variables.", |
| "items": {"type": "string"}, |
| }, |
| "dependent_variable": { |
| "type": "string", |
| "description": "Column name from query.csv to use as the dependent (outcome) variable.", |
| }, |
| "category": { |
| "type": "string", |
| "description": "Optional column name used to colour-code points and fit separate regression lines per group.", |
| }, |
| }, |
| "required": ["independent_variables", "dependent_variable"], |
| }, |
| }, |
| ] |
|
|