mirror of
https://github.com/Microsoft/sql-server-samples.git
synced 2025-12-08 14:58:54 +00:00
Initial samples for SQL Server 2019 big data cluster
Demonstrates various functionality in big data cluster.
This commit is contained in:
@@ -0,0 +1,11 @@
|
||||
# Query data in HDFS from SQL Server master
|
||||
|
||||
In SQL Server 2019 big data clusters, the SQL Server engine has gained the ability to natively read HDFS files, such as CSV and parquet files, by using SQL Server instances collocated on each of the HDFS data nodes to filter and aggregate data locally in parallel across all of the HDFS data nodes.
|
||||
|
||||
In this example, you are going to create an external table in the SQL Server Master instance that points to data in HDFS within the SQL Server Big data cluster. Then you will join the data in the external table with high value data in SQL Master instance.
|
||||
|
||||
## Instructions
|
||||
|
||||
1. Connect to SQL Server Master instance.
|
||||
|
||||
1. Execute the [external-table-hdfs.sql](external-table-hdfs.sql).
|
||||
@@ -0,0 +1,52 @@
|
||||
USE sales
|
||||
GO
|
||||
|
||||
-- Create file format for CSV file with appropriate properties.
|
||||
--
|
||||
CREATE EXTERNAL FILE FORMAT csv_file
|
||||
WITH (
|
||||
FORMAT_TYPE = DELIMITEDTEXT,
|
||||
FORMAT_OPTIONS(
|
||||
FIELD_TERMINATOR = ',',
|
||||
STRING_DELIMITER = '"',
|
||||
FIRST_ROW = 2,
|
||||
USE_TYPE_DEFAULT = TRUE)
|
||||
);
|
||||
|
||||
-- Create external table over HDFS data source (SqlStoragePool) in
|
||||
-- SQL Server 2019 big data cluster. The SqlStoragePool data source
|
||||
-- is a special data source that is available in any new database in
|
||||
-- SQL Master instance.
|
||||
--
|
||||
CREATE EXTERNAL TABLE [web_clickstreams_hdfs]
|
||||
("wcs_click_date_sk" BIGINT , "wcs_click_time_sk" BIGINT , "wcs_sales_sk" BIGINT , "wcs_item_sk" BIGINT , "wcs_web_page_sk" BIGINT , "wcs_user_sk" BIGINT)
|
||||
WITH
|
||||
(
|
||||
DATA_SOURCE = SqlStoragePool,
|
||||
LOCATION = '/clickstream_data',
|
||||
FILE_FORMAT = csv_file
|
||||
);
|
||||
GO
|
||||
|
||||
-- Join external table with local tables
|
||||
--
|
||||
SELECT
|
||||
wcs_user_sk,
|
||||
SUM( CASE WHEN i_category = 'Books' THEN 1 ELSE 0 END) AS book_category_clicks,
|
||||
SUM( CASE WHEN i_category_id = 1 THEN 1 ELSE 0 END) AS [Home & Kitchen],
|
||||
SUM( CASE WHEN i_category_id = 2 THEN 1 ELSE 0 END) AS [Music],
|
||||
SUM( CASE WHEN i_category_id = 3 THEN 1 ELSE 0 END) AS [Books],
|
||||
SUM( CASE WHEN i_category_id = 4 THEN 1 ELSE 0 END) AS [Clothing & Accessories],
|
||||
SUM( CASE WHEN i_category_id = 5 THEN 1 ELSE 0 END) AS [Electronics],
|
||||
SUM( CASE WHEN i_category_id = 6 THEN 1 ELSE 0 END) AS [Tools & Home Improvement],
|
||||
SUM( CASE WHEN i_category_id = 7 THEN 1 ELSE 0 END) AS [Toys & Games],
|
||||
SUM( CASE WHEN i_category_id = 8 THEN 1 ELSE 0 END) AS [Movies & TV],
|
||||
SUM( CASE WHEN i_category_id = 9 THEN 1 ELSE 0 END) AS [Sports & Outdoors]
|
||||
FROM [dbo].[web_clickstreams_hdfs]
|
||||
INNER JOIN item it ON (wcs_item_sk = i_item_sk
|
||||
AND wcs_user_sk IS NOT NULL)
|
||||
GROUP BY wcs_user_sk;
|
||||
GO
|
||||
|
||||
DROP EXTERNAL TABLE [dbo].[web_clickstreams_hdfs];
|
||||
GO
|
||||
@@ -0,0 +1,12 @@
|
||||
# Query data in Oracle from SQL Server master
|
||||
|
||||
Create external table over an Oracle database
|
||||
by leveraging SQL Server Polybase technology. SQL Server Big Data clusters can query external data sources without importing the data in SQL Server. SQL Server 2019 introduces new connectors to data sources like Oracle, MongoDB and Teradata. In this example, you are going to create an external table in SQL Server Master instance over the inventory table that sits on an Oracle server.
|
||||
|
||||
**Before you begin**, you need to have an Oracle instance and credentials. Execute the SQL script [inventory-ora.sql](inventory-ora.sql/) in Oracle to create the table and import the "inventory.csv" file created by the bootstrap sample database.
|
||||
|
||||
## Instructions
|
||||
|
||||
1. Connect to SQL Server Master instance.
|
||||
|
||||
1. Execute the SQL [external-table-oracle.sql](external-table-oracle.sql/).
|
||||
@@ -0,0 +1,44 @@
|
||||
USE sales
|
||||
GO
|
||||
|
||||
-- Create database scoped credential to connect to Oracle server
|
||||
-- Provide appropriate credentials to Oracle server in below statement.
|
||||
-- If you are using SQL Server Management Studio then you can replace the parameters using
|
||||
-- the Query menu, and "Specify Values for Template Parameters" option.
|
||||
CREATE DATABASE SCOPED CREDENTIAL [OracleCredential]
|
||||
WITH IDENTITY = '<oracle_user,nvarchar(100),SYSTEM>', SECRET = '<oracle_user_password,nvarchar(100),manager>';
|
||||
|
||||
-- Create external data source that points to Oracle server
|
||||
--
|
||||
CREATE EXTERNAL DATA SOURCE [OracleSalesSrvr]
|
||||
WITH (LOCATION = 'oracle://<oracle_server,nvarchar(100)>',CREDENTIAL = [OracleCredential]);
|
||||
|
||||
-- Create external table over inventory table on Oracle server
|
||||
-- NOTE: Table names and column names will use ANSI SQL quoted identifier while querying against Oracle.
|
||||
-- As a result, the names are case-sensitive so specify the name in the external table definition
|
||||
-- that matches the exact case of the table and column names in the Oracle metadata.
|
||||
CREATE EXTERNAL TABLE [inventory_ora]
|
||||
([inv_date] DECIMAL(10,0) NOT NULL, [inv_item] DECIMAL(10,0) NOT NULL,
|
||||
[inv_warehouse] DECIMAL(10,0) NOT NULL, [inv_quantity_on_hand] DECIMAL(10,0))
|
||||
WITH (DATA_SOURCE=[OracleSalesSrvr],
|
||||
LOCATION='<oracle_service_name,nvarchar(30),xe>.<oracle_schema,nvarchar(128),HR>.<oracle_table,nvarchar(128),INVENTORY>');
|
||||
GO
|
||||
|
||||
-- Join external table with local tables
|
||||
--
|
||||
SELECT TOP(100) w.w_warehouse_name, i.inv_item, SUM(i.inv_quantity_on_hand) as total_quantity
|
||||
FROM [inventory_ora] as i
|
||||
JOIN item as it
|
||||
ON it.i_item_sk = i.inv_item
|
||||
JOIN warehouse as w
|
||||
ON w.w_warehouse_sk = i.inv_warehouse
|
||||
WHERE it.i_category = 'Books' and i.inv_item BETWEEN 1 and 18000 --> get items within specific range
|
||||
GROUP BY w.w_warehouse_name, i.inv_item;
|
||||
GO
|
||||
|
||||
-- Cleanup
|
||||
--
|
||||
DROP EXTERNAL TABLE [inventory_ora];
|
||||
DROP EXTERNAL DATA SOURCE [OracleSalesSrvr] ;
|
||||
DROP DATABASE SCOPED CREDENTIAL [OracleCredential];
|
||||
GO
|
||||
@@ -0,0 +1,10 @@
|
||||
-- Inventory table over which the SQL Server external table will be defined
|
||||
CREATE TABLE "INVENTORY"
|
||||
(
|
||||
"INV_DATE" NUMBER(10,0) NOT NULL,
|
||||
"INV_ITEM" NUMBER(10,0) NOT NULL,
|
||||
"INV_WAREHOUSE" NUMBER(10,0) NOT NULL,
|
||||
"INV_QUANTITY_ON_HAND" NUMBER(10,0)
|
||||
);
|
||||
|
||||
CREATE INDEX INV_ITEM ON HR.INVENTORY(INV_ITEM);
|
||||
Reference in New Issue
Block a user