refactor CHPA 01 and 02 warehouse jobs

This commit is contained in:
2026-08-20 19:39:11 +08:00
parent 8c3f45f178
commit fd1355a4fb
26 changed files with 4391 additions and 44 deletions
@@ -0,0 +1,81 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Build the manufacturer-to-corporation mapping dimension:
-- one row per manufacturer with its own attributes plus the
-- matched corporation master code, abbreviation and name.
-- Source : dwd.dwd_ims_td_manufacturer (self-join on
-- Corporation_Code = Manufacturer_CODE)
-- Target : dws.dws_ext_td_ims_manufacturer_corporation
-- Grain : One row per manufacturer; CORP_MASTER_CODE/CORP_ABBR/CORP_DES
-- are NULL when no corporation master matches.
-- Write mode : Full refresh (INSERT OVERWRITE).
-- Replaces : dwd.dwd_ims_td_manufacturer_corp (legacy output).
-- Consumers : 02_dws/04_dws_ext_td_ims_pack_property,
-- 02_dws/06_dws_ext_td_ims_corporation_cn,
-- 02_dws/07_dws_ext_td_ims_manufacturer_cn -- all must read the
-- DWS target.
-- Notes : The legacy job wrote T1.* but no source DDL/DESCRIBE is available
-- in this repository. This migration therefore defines a deliberate
-- 9-column DWS contract containing every field used by repository
-- consumers, instead of creating unknown columns that would be
-- silently populated with NULL. The commented CTAS infers physical
-- types from source expressions without guessing them. DESCRIBE the
-- source before adding any future consumer field.
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: infer the explicit 9-column contract without copying data.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_td_ims_manufacturer_corporation AS
-- SELECT
-- manufacturer.Manufacturer_ID,
-- manufacturer.Manufacturer_CODE,
-- manufacturer.Manufacturer_Abbr,
-- manufacturer.Manufacturer_Name,
-- manufacturer.ManufacturerType_ID,
-- manufacturer.Corporation_Code,
-- corporation_master.Corporation_Code AS CORP_MASTER_CODE,
-- corporation_master.Manufacturer_Abbr AS CORP_ABBR,
-- corporation_master.Manufacturer_Name AS CORP_DES
-- FROM dwd.dwd_ims_td_manufacturer AS manufacturer
-- LEFT JOIN dwd.dwd_ims_td_manufacturer AS corporation_master
-- ON manufacturer.Corporation_Code = corporation_master.Manufacturer_CODE
-- WHERE 1 = 0;
-- COMMAND ----------
INSERT OVERWRITE TABLE dws.dws_ext_td_ims_manufacturer_corporation (
Manufacturer_ID,
Manufacturer_CODE,
Manufacturer_Abbr,
Manufacturer_Name,
ManufacturerType_ID,
Corporation_Code,
CORP_MASTER_CODE,
CORP_ABBR,
CORP_DES
)
WITH manufacturer AS (
SELECT
manufacturer_id,
manufacturer_code,
manufacturer_abbr,
manufacturer_name,
manufacturertype_id,
corporation_code
FROM dwd.dwd_ims_td_manufacturer
)
SELECT
manufacturer.manufacturer_id AS manufacturer_id,
manufacturer.manufacturer_code AS manufacturer_code,
manufacturer.manufacturer_abbr AS manufacturer_abbr,
manufacturer.manufacturer_name AS manufacturer_name,
manufacturer.manufacturertype_id AS manufacturertype_id,
manufacturer.corporation_code AS corporation_code,
corporation_master.corporation_code AS corp_master_code,
corporation_master.manufacturer_abbr AS corp_abbr,
corporation_master.manufacturer_name AS corp_des
FROM manufacturer
LEFT JOIN dwd.dwd_ims_td_manufacturer AS corporation_master
ON manufacturer.corporation_code = corporation_master.manufacturer_code
;
@@ -0,0 +1,179 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Build the master pack property dimension: one row per pack with
-- pack/product/molecule codes and descriptions, ATC/NFC hierarchy
-- codes, molecule attributes, manufacturer/corporation codes and
-- descriptions, manufacturer type and brand type. Also applies the
-- AZ (A5Z/A5ZD) business-scope overrides exactly as legacy.
-- Source : dwd.dwd_ims_td_pack,
-- dwd.dwd_ims_td_product,
-- dwd.dwd_ims_td_new_form_class,
-- dws.dws_ext_td_ims_nfc_hierarchy (repointed from legacy DWD),
-- dwd.dwd_ims_td_therapeutic_class,
-- dws.dws_ext_td_ims_atc_hierarchy (repointed from legacy DWD),
-- dwd.dwd_ims_td_pack_additional_attribute,
-- dws.dws_ext_td_ims_manufacturer_corporation (repointed from
-- legacy DWD),
-- dwd.dwd_ims_td_manufacturertype,
-- dwd.dwd_gnd_ims_tblbrandtype,
-- dwd.dwd_gnd_tbl_corp_change (AZ override key)
-- Target : dws.dws_ext_td_ims_pack_property
-- Grain : One row per PACK_COD (SELECT DISTINCT; hierarchy/attribute fan-out
-- is masked by DISTINCT, matching legacy).
-- Write mode : Full refresh (INSERT OVERWRITE) + 2 business-override UPDATEs
-- (A5Z/A5ZD), executed in this order.
-- Replaces : dwd.dwd_ims_td_pack_property (legacy output).
-- Consumers : 02_dws/08_dws_ext_td_ims_product_multi_manufacturer,
-- 02_dws/16_dws_ext_td_ims_market, 03 dm_ims_td_pack_property --
-- all must migrate to the DWS target.
-- Notes : Legacy expressions, padding rules, join keys and DISTINCT are
-- preserved verbatim. The two post-write NULL cleanups
-- (CORP_COD='' / STGH_DES='') are folded into COALESCE in the
-- projection (identical final values; the A5Z/A5ZD overrides key
-- on PROD_COD/PACK_COD, so they are unaffected). The BrandType
-- join compares the raw source Pack_Code with the pre-padded
-- BRANDTYPE.PACK_COD -- a known legacy hazard, intentionally not
-- changed. Run order: after 01_dwd/01_standardize_gnd_codes.sql
-- (brandtype padding) and 02_dws/01, 02, 03 (hierarchy and
-- manufacturer-corporation).
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: copy the verified legacy schema without copying data.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_td_ims_pack_property
-- LIKE dwd.dwd_ims_td_pack_property;
-- COMMAND ----------
INSERT OVERWRITE TABLE dws.dws_ext_td_ims_pack_property (
PACK_COD,
PACK_DES,
STGH_DES,
PACK_LCH,
PROD_COD,
CMPS_COD,
CMPS_DES,
ATC1_COD,
ATC2_COD,
ATC3_COD,
ATC4_COD,
APP1_COD,
APP2_COD,
APP3_COD,
BIO_DESC,
GENE_ORIG_DESC,
ETH_OTC_DESC,
NRDL_DESC,
NRDL_Entry_Date,
EDL_DESC,
TCM_DESC,
PAED_DESC,
GQCE_DESC,
VBP_DESC,
MANU_COD,
MANU_DES,
MNFL_COD,
MNFL_DES,
CORP_COD,
CORP_DES,
BrandType
)
SELECT DISTINCT
if(PACK.Pack_Code REGEXP '^[0-9]', right(concat('000000000000', PACK.Pack_Code), 12), PACK.Pack_Code) AS PACK_COD,
PACK.Pack_Description AS PACK_DES,
COALESCE(PACK.STRENGTH, '') AS STGH_DES,
concat('Y', LEFT(PACK.LAUNCHTIME, 4), 'M', RIGHT(PACK.LAUNCHTIME, 2)) AS PACK_LCH,
RIGHT(concat('000000000', PROD.Product_Code), 9) AS PROD_COD,
RIGHT(concat('000000', MOLE.MoleCompCode), 6) AS CMPS_COD,
MOLE.MoleCompDesc AS CMPS_DES,
ATCH.ATC1_CODE AS ATC1_COD,
ATCH.ATC2_CODE AS ATC2_COD,
ATCH.ATC3_CODE AS ATC3_COD,
ATCH.ATC4_CODE AS ATC4_COD,
NFCH.NFC1_CODE AS APP1_COD,
NFCH.NFC2_CODE AS APP2_COD,
NFCH.NFC3_CODE AS APP3_COD,
MOLE.BIO AS BIO_DESC,
MOLE.Gene_Orig AS GENE_ORIG_DESC,
MOLE.Rx_Flag AS ETH_OTC_DESC,
MOLE.NRDL AS NRDL_DESC,
MOLE.NRDL_Entry_Date,
MOLE.EDL AS EDL_DESC,
MOLE.TCMEX AS TCM_DESC,
MOLE.PAED AS PAED_DESC,
MOLE.GQCE AS GQCE_DESC,
MOLE.VBP AS VBP_DESC,
MANU.Manufacturer_Abbr AS MANU_COD,
MANU.Manufacturer_Name AS MANU_DES,
MANUT.ManufacturerType_CODE AS MNFL_COD,
MANUT.ManufacturerType_Name AS MNFL_DES,
COALESCE(MANU.CORP_ABBR, '') AS CORP_COD,
MANU.CORP_DES AS CORP_DES,
BRANDTYPE.Brand_Type AS BrandType
FROM dwd.dwd_ims_td_pack AS PACK
LEFT JOIN dwd.dwd_ims_td_product AS PROD
ON PACK.Product_ID = PROD.Product_ID
LEFT JOIN dwd.dwd_ims_td_new_form_class AS NFC
ON PACK.NewFormClass_ID = NFC.NewFormClass_ID
LEFT JOIN dws.dws_ext_td_ims_nfc_hierarchy AS NFCH
ON NFC.NewFormClass_Code = NFCH.NFC3_CODE
LEFT JOIN dwd.dwd_ims_td_therapeutic_class AS ATC
ON PACK.Therapeutic_ID = ATC.Therapeutic_ID
LEFT JOIN dws.dws_ext_td_ims_atc_hierarchy AS ATCH
ON ATC.Therapeutic_Code = ATCH.ATC4_CODE
LEFT JOIN dwd.dwd_ims_td_pack_additional_attribute AS MOLE
ON PACK.Pack_ID = MOLE.Pack_ID
LEFT JOIN dws.dws_ext_td_ims_manufacturer_corporation AS MANU
ON PROD.Manufacturer_ID = MANU.Manufacturer_ID
LEFT JOIN dwd.dwd_ims_td_manufacturertype AS MANUT
ON MANU.ManufacturerType_ID = MANUT.ManufacturerType_ID
LEFT JOIN dwd.dwd_gnd_ims_tblbrandtype AS BRANDTYPE
ON PACK.Pack_Code = BRANDTYPE.PACK_COD
;
-- COMMAND ----------
-- AZ brand scope override A5Z (legacy business rule, preserved verbatim):
-- products/packs flagged with corp_cod = 'A5Z' in the corporate-change table
-- are displayed as ASTRAZENECA GROUP. Matches on product level or pack level.
UPDATE dws.dws_ext_td_ims_pack_property
SET
CORP_COD = 'A5Z',
CORP_DES = 'ASTRAZENECA GROUP'
WHERE PROD_COD IN (
SELECT RIGHT(concat('0000000000', PROD_COD), 9) AS PROD_COD
FROM dwd.dwd_gnd_tbl_corp_change
WHERE corp_cod = 'A5Z'
)
OR PACK_COD IN (
SELECT if(PACK_COD REGEXP '^[0-9]', RIGHT(concat('000000000000', PACK_COD), 12), PACK_COD) AS PACK_COD
FROM dwd.dwd_gnd_tbl_corp_change
WHERE corp_cod = 'A5Z'
)
;
-- COMMAND ----------
-- AZ brand scope override A5ZD (legacy business rule, preserved verbatim):
-- products/packs flagged with corp_cod = 'A5ZD' are re-labeled as AZDealed on
-- both corporation and manufacturer fields.
UPDATE dws.dws_ext_td_ims_pack_property
SET
CORP_COD = 'A5ZD',
CORP_DES = 'AZDealed',
MANU_COD = 'A5ZD',
MANU_DES = 'AZDealed'
WHERE PROD_COD IN (
SELECT RIGHT(concat('0000000000', PROD_COD), 9) AS PROD_COD
FROM dwd.dwd_gnd_tbl_corp_change
WHERE corp_cod = 'A5ZD'
)
OR PACK_COD IN (
SELECT if(PACK_COD REGEXP '^[0-9]', RIGHT(concat('000000000000', PACK_COD), 12), PACK_COD) AS PACK_COD
FROM dwd.dwd_gnd_tbl_corp_change
WHERE corp_cod = 'A5ZD'
)
;
+56
View File
@@ -0,0 +1,56 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Build the IMS audit/geo (province) dimension with English and
-- Chinese names, audit type and tier attributes. Province data is
-- hard-coded in the staging source; city data is no longer
-- provided (20260122 note).
-- Source : tmp.tmp_province_rawdata (hard-coded province data)
-- Target : dws.dws_ext_td_ims_geo
-- Grain : One row per AUDIT_COD as provided by the source. The grain is
-- not enforced by this job (no DISTINCT, matching legacy).
-- Write mode : Full refresh (INSERT OVERWRITE).
-- Replaces : dws.dws_ims_td_geo (legacy output).
-- Consumers : dm_ims_td_geo (DM build), dm_ims_td_org, dm_ims_td_org_hvh --
-- all must migrate to the DWS target.
-- Notes : Source stays on the tmp staging table (legacy contract; no DWD
-- equivalent exists yet). ETL_INSERT_DT/ETL_UPDATE_DT are copied
-- from the source unchanged, matching legacy. 03 dm_ims_td_geo.sql
-- currently ALSO writes dws.dws_ims_td_geo; that duplicate write
-- must be removed when consumers migrate.
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: copy the verified legacy schema without copying data.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_td_ims_geo
-- LIKE dws.dws_ims_td_geo;
-- COMMAND ----------
INSERT OVERWRITE TABLE dws.dws_ext_td_ims_geo (
AUDIT_COD,
AUDIT_DES,
AUDIT_DES_C,
AUDIT_TYPE,
CITY_TIER,
AZ_CITY_TIER,
PROVINCE,
PROVINCE_C,
REGIONCENTER,
ETL_INSERT_DT,
ETL_UPDATE_DT
)
SELECT
audit_cod,
audit_des,
audit_des_c,
audit_type,
city_tier,
az_city_tier,
province,
province_c,
regioncenter,
etl_insert_dt,
etl_update_dt
FROM tmp.tmp_province_rawdata
;
@@ -0,0 +1,72 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Attach Chinese (CN) corporation names; also forces the
-- hard-coded AZDealed row (CORP_COD = 'A5ZD') exactly as the
-- legacy job did.
-- Source : dws.dws_ext_td_ims_manufacturer_corporation,
-- dwd.dwd_gnd_ims_tblmanucn (CN names on CORP_ABBR = abbrev)
-- Target : dws.dws_ext_td_ims_corporation_cn
-- Grain : One row per distinct CORP_COD (plus the forced A5ZD row).
-- Write mode : Full refresh (INSERT OVERWRITE).
-- Replaces : dws.dws_ims_td_corp_cn (legacy output).
-- Consumers : tmp_ims_td_prod_tmp (join on CORP_COD); dm_ims_td_pack_property
-- (DIM_CORP) -- both must migrate to the DWS target.
-- Notes : Corporation source repointed from dwd.dwd_ims_td_manufacturer_corp
-- to dws.dws_ext_td_ims_manufacturer_corporation (same schema,
-- incl. CORP_ABBR/CORP_DES); that DWS table must be built before
-- this job runs. The legacy unconditional second INSERT of the
-- A5ZD row is folded into a single deterministic overwrite via
-- UNION ALL; the resulting row set is identical to legacy.
-- Legacy writes CORP_DES_C (no English fallback): a corporation
-- without a Chinese name keeps a NULL CORP_DES_C.
-- DIM_CORP currently selects CORP_DES_CN (name mismatch vs the
-- CORP_DES_C contract); align that consumer during migration.
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: copy the verified legacy schema without copying data.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_td_ims_corporation_cn
-- LIKE dws.dws_ims_td_corp_cn;
-- COMMAND ----------
INSERT OVERWRITE TABLE dws.dws_ext_td_ims_corporation_cn (
CORP_COD,
CORP_DES,
CORP_DES_C,
ETL_INSERT_DT,
ETL_UPDATE_DT
)
WITH corporation AS (
SELECT
corp_abbr,
corp_des
FROM dws.dws_ext_td_ims_manufacturer_corporation
WHERE corp_abbr IS NOT NULL
),
corporation_cn AS (
SELECT DISTINCT
abbrev,
namec
FROM dwd.dwd_gnd_ims_tblmanucn
)
SELECT DISTINCT
corporation.corp_abbr AS corp_cod,
corporation.corp_des AS corp_des,
corporation_cn.namec AS corp_des_c,
FROM_UTC_TIMESTAMP(CURRENT_TIMESTAMP(), 'UTC+8') AS etl_insert_dt,
FROM_UTC_TIMESTAMP(CURRENT_TIMESTAMP(), 'UTC+8') AS etl_update_dt
FROM corporation
LEFT JOIN corporation_cn
ON corporation.corp_abbr = corporation_cn.abbrev
UNION ALL
SELECT
'A5ZD' AS corp_cod,
'AZDealed' AS corp_des,
'AZDealed' AS corp_des_c,
FROM_UTC_TIMESTAMP(CURRENT_TIMESTAMP(), 'UTC+8') AS etl_insert_dt,
FROM_UTC_TIMESTAMP(CURRENT_TIMESTAMP(), 'UTC+8') AS etl_update_dt
;
@@ -0,0 +1,72 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Attach Chinese (CN) manufacturer names; also forces the
-- hard-coded AZDealed row (MANU_COD = 'A5ZD') exactly as the
-- legacy job did.
-- Source : dws.dws_ext_td_ims_manufacturer_corporation,
-- dwd.dwd_gnd_ims_tblmanucn (CN names on Manufacturer_Abbr = abbrev)
-- Target : dws.dws_ext_td_ims_manufacturer_cn
-- Grain : One row per distinct MANU_COD (plus the forced A5ZD row).
-- Write mode : Full refresh (INSERT OVERWRITE).
-- Replaces : dws.dws_ims_td_manu_cn (legacy output).
-- Consumers : dm_ims_td_pack_property (DIM_MANU) -- must migrate to the DWS
-- target.
-- Notes : Manufacturer source repointed from dwd.dwd_ims_td_manufacturer_corp
-- to dws.dws_ext_td_ims_manufacturer_corporation (same schema,
-- incl. Manufacturer_Abbr/Manufacturer_Name); that DWS table must
-- be built before this job runs. The legacy unconditional second
-- INSERT of the A5ZD row is folded into a single deterministic
-- overwrite via UNION ALL; the resulting row set is identical to
-- legacy. Legacy writes MANU_DES_C (no English fallback): a
-- manufacturer without a Chinese name keeps a NULL MANU_DES_C.
-- DIM_MANU currently selects MANU_DES_CN (name mismatch vs the
-- MANU_DES_C contract); align that consumer during migration.
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: copy the verified legacy schema without copying data.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_td_ims_manufacturer_cn
-- LIKE dws.dws_ims_td_manu_cn;
-- COMMAND ----------
INSERT OVERWRITE TABLE dws.dws_ext_td_ims_manufacturer_cn (
MANU_COD,
MANU_DES,
MANU_DES_C,
ETL_INSERT_DT,
ETL_UPDATE_DT
)
WITH manufacturer AS (
SELECT
manufacturer_abbr,
manufacturer_name
FROM dws.dws_ext_td_ims_manufacturer_corporation
WHERE manufacturer_abbr IS NOT NULL
),
manufacturer_cn AS (
SELECT DISTINCT
abbrev,
namec
FROM dwd.dwd_gnd_ims_tblmanucn
)
SELECT DISTINCT
manufacturer.manufacturer_abbr AS manu_cod,
manufacturer.manufacturer_name AS manu_des,
manufacturer_cn.namec AS manu_des_c,
FROM_UTC_TIMESTAMP(CURRENT_TIMESTAMP(), 'UTC+8') AS etl_insert_dt,
FROM_UTC_TIMESTAMP(CURRENT_TIMESTAMP(), 'UTC+8') AS etl_update_dt
FROM manufacturer
LEFT JOIN manufacturer_cn
ON manufacturer.manufacturer_abbr = manufacturer_cn.abbrev
UNION ALL
SELECT
'A5ZD' AS manu_cod,
'AZDealed' AS manu_des,
'AZDealed' AS manu_des_c,
FROM_UTC_TIMESTAMP(CURRENT_TIMESTAMP(), 'UTC+8') AS etl_insert_dt,
FROM_UTC_TIMESTAMP(CURRENT_TIMESTAMP(), 'UTC+8') AS etl_update_dt
;
@@ -0,0 +1,62 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Flag products whose Chinese name (PROD_DES_C) is shared by more
-- than 5 distinct corporations (CORP_DES) or more than 5 distinct
-- manufacturers (MANU_DES). These products are demoted in report
-- ordering via RANK_TYPE in dws.dws_ext_td_ims_product_cn.
-- Source : dws.dws_ext_td_ims_pack_property (pack-level CORP_DES, MANU_DES)
-- dwd.dwd_gnd_ims_tblprodcn (prodcode -> PROD_COD, namec -> PROD_DES_C)
-- Target : dws.dws_ext_td_ims_product_multi_manufacturer
-- Grain : One row per PROD_COD (only flagged products are materialized).
-- Write mode : Full refresh (INSERT OVERWRITE).
-- Replaces : tmp.tmp_ims_td_prod_tmp (legacy staging list used by the
-- legacy UPDATE on dws.dws_ims_td_prod_cn).
-- Consumers : dws.dws_ext_td_ims_product_cn (RANK_TYPE 0/1 via join).
-- Notes : The legacy >5 distinct CORP_DES OR >5 distinct MANU_DES rule is
-- preserved, counted at PROD_DES_C level across all packs sharing
-- the Chinese name, then applied to every PROD_COD carrying that
-- name. Legacy dead joins (pack_property, corp_cn) are dropped.
-- PROD_COD is zero-padded to 9 digits exactly as in the legacy
-- product build (RIGHT(CONCAT('0000000000', prodcode), 9)).
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: copy the verified legacy schema without copying data.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_td_ims_product_multi_manufacturer
-- LIKE tmp.tmp_ims_td_prod_tmp;
-- COMMAND ----------
INSERT OVERWRITE TABLE dws.dws_ext_td_ims_product_multi_manufacturer (
PROD_COD
)
WITH product_cn AS (
-- Chinese-name master derived from the raw source (same zero-padding as
-- the legacy product build) so this job does not depend on the product_cn
-- table it feeds.
SELECT DISTINCT
RIGHT(CONCAT('0000000000', prodcode), 9) AS prod_cod,
namec AS prod_des_c
FROM dwd.dwd_gnd_ims_tblprodcn
),
flagged_names AS (
-- Chinese names shared by more than 5 distinct CORPs or MANUs.
SELECT
product_cn.prod_des_c
FROM dws.dws_ext_td_ims_pack_property pack_property
LEFT JOIN product_cn
ON pack_property.prod_cod = product_cn.prod_cod
GROUP BY
product_cn.prod_des_c
HAVING
( COUNT(DISTINCT pack_property.corp_des) > 5
OR COUNT(DISTINCT pack_property.manu_des) > 5
)
AND product_cn.prod_des_c IS NOT NULL
)
SELECT DISTINCT
product_cn.prod_cod
FROM product_cn
WHERE product_cn.prod_des_c IN (SELECT prod_des_c FROM flagged_names)
;
@@ -0,0 +1,61 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Build the product master with Chinese names and a declarative
-- RANK_TYPE. RANK_TYPE = 0 marks products demoted in report
-- ordering (Chinese name shared by >5 CORPs or MANUs, see
-- dws.dws_ext_td_ims_product_multi_manufacturer); otherwise 1.
-- Source : dwd.dwd_gnd_ims_tblprodcn
-- dws.dws_ext_td_ims_product_multi_manufacturer (flag join)
-- Target : dws.dws_ext_td_ims_product_cn
-- Grain : One row per PROD_COD (9-digit zero-padded product code).
-- Write mode : Full refresh (INSERT OVERWRITE).
-- Replaces : dws.dws_ims_td_prod_cn (legacy 7-column contract preserved) and
-- the legacy two-step RANK_TYPE demotion (tmp list + UPDATE) which
-- is folded into a declarative CASE here.
-- Consumers : Report/BI layer via RANK_TYPE (sorting priority). This job must
-- run after dws_ext_td_ims_product_multi_manufacturer (file 08).
-- Notes : Legacy DISTINCT semantics preserved. Column names follow the
-- legacy dws_ims_td_prod_cn contract; ETL timestamps named per
-- the DWS convention (ETL_INSERT_DT/ETL_UPDATE_DT).
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: copy the verified legacy schema without copying data.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_td_ims_product_cn
-- LIKE dws.dws_ims_td_prod_cn;
-- COMMAND ----------
INSERT OVERWRITE TABLE dws.dws_ext_td_ims_product_cn (
PROD_COD,
PROD_DES,
PROD_DES_C,
CMPS_DES_C,
RANK_TYPE,
ETL_INSERT_DT,
ETL_UPDATE_DT
)
WITH product_cn AS (
SELECT DISTINCT
RIGHT(CONCAT('0000000000', prodcode), 9) AS prod_cod,
ename AS prod_des,
namec AS prod_des_c,
gene_name AS cmps_des_c
FROM dwd.dwd_gnd_ims_tblprodcn
)
SELECT DISTINCT
product_cn.prod_cod AS PROD_COD,
product_cn.prod_des AS PROD_DES,
product_cn.prod_des_c AS PROD_DES_C,
product_cn.cmps_des_c AS CMPS_DES_C,
CASE
WHEN multi_mfr.prod_cod IS NOT NULL THEN 0
ELSE 1
END AS RANK_TYPE,
from_utc_timestamp(current_timestamp(), 'UTC+8') AS ETL_INSERT_DT,
from_utc_timestamp(current_timestamp(), 'UTC+8') AS ETL_UPDATE_DT
FROM product_cn
LEFT JOIN dws.dws_ext_td_ims_product_multi_manufacturer multi_mfr
ON product_cn.prod_cod = multi_mfr.prod_cod
;
@@ -0,0 +1,108 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Attach Chinese (CN) descriptions to the flattened IMS ATC
-- level 1-4 hierarchy; the English description is the fallback
-- when no CN name exists for a code.
-- Source : dws.dws_ext_td_ims_atc_hierarchy,
-- dwd.dwd_gnd_ims_tblATC (per-level CN names)
-- Target : dws.dws_ext_td_ims_atc_cn
-- Grain : One row per distinct ATC1..ATC4 hierarchy path; lower levels
-- may be null when a parent code has no matching child.
-- Write mode : Full refresh (INSERT OVERWRITE).
-- Replaces : dws.dws_ims_td_atc_cn (legacy output).
-- Consumers : dm_ims_td_pack_property (DIM_ATC) -- must migrate to the DWS
-- target.
-- Notes : Hierarchy source repointed from dwd.dwd_ims_atc_hierarchy to
-- dws.dws_ext_td_ims_atc_hierarchy; the DWS table carries the same
-- code/name columns, so the joins are unchanged. CN-name joins are
-- case-sensitive string equality, preserved from legacy. Legacy
-- typo'd column names ATC2_CODe/ATC3_CODe/ATC4_CODe are normalized
-- to ATC2_CODE/ATC3_CODE/ATC4_CODE (case-insensitive resolution
-- made them equivalent).
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: copy the verified legacy schema without copying data.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_td_ims_atc_cn
-- LIKE dws.dws_ims_td_atc_cn;
-- COMMAND ----------
INSERT OVERWRITE TABLE dws.dws_ext_td_ims_atc_cn (
ATC1_COD,
ATC1_DES,
ATC1_DES_C,
ATC2_COD,
ATC2_DES,
ATC2_DES_C,
ATC3_COD,
ATC3_DES,
ATC3_DES_C,
ATC4_COD,
ATC4_DES,
ATC4_DES_C,
ETL_INSERT_DT,
ETL_UPDATE_DT
)
WITH hierarchy AS (
SELECT
atc1_code,
atc1_des,
atc2_code,
atc2_des,
atc3_code,
atc3_des,
atc4_code,
atc4_des
FROM dws.dws_ext_td_ims_atc_hierarchy
),
atc1_cn AS (
SELECT DISTINCT
atc1_cod,
atc1_des_c
FROM dwd.dwd_gnd_ims_tblATC
),
atc2_cn AS (
SELECT DISTINCT
atc2_cod,
atc2_des_c
FROM dwd.dwd_gnd_ims_tblATC
),
atc3_cn AS (
SELECT DISTINCT
atc3_cod,
atc3_des_c
FROM dwd.dwd_gnd_ims_tblATC
),
atc4_cn AS (
SELECT DISTINCT
atc4_cod,
atc4_des_c
FROM dwd.dwd_gnd_ims_tblATC
)
SELECT DISTINCT
hierarchy.atc1_code AS atc1_cod,
hierarchy.atc1_des AS atc1_des,
COALESCE(atc1_cn.atc1_des_c, hierarchy.atc1_des) AS atc1_des_c,
hierarchy.atc2_code AS atc2_cod,
hierarchy.atc2_des AS atc2_des,
COALESCE(atc2_cn.atc2_des_c, hierarchy.atc2_des) AS atc2_des_c,
hierarchy.atc3_code AS atc3_cod,
hierarchy.atc3_des AS atc3_des,
COALESCE(atc3_cn.atc3_des_c, hierarchy.atc3_des) AS atc3_des_c,
hierarchy.atc4_code AS atc4_cod,
hierarchy.atc4_des AS atc4_des,
COALESCE(atc4_cn.atc4_des_c, hierarchy.atc4_des) AS atc4_des_c,
FROM_UTC_TIMESTAMP(CURRENT_TIMESTAMP(), 'UTC+8') AS etl_insert_dt,
FROM_UTC_TIMESTAMP(CURRENT_TIMESTAMP(), 'UTC+8') AS etl_update_dt
FROM hierarchy
LEFT JOIN atc1_cn
ON hierarchy.atc1_code = atc1_cn.atc1_cod
LEFT JOIN atc2_cn
ON hierarchy.atc2_code = atc2_cn.atc2_cod
LEFT JOIN atc3_cn
ON hierarchy.atc3_code = atc3_cn.atc3_cod
LEFT JOIN atc4_cn
ON hierarchy.atc4_code = atc4_cn.atc4_cod
;
@@ -0,0 +1,89 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Attach Chinese (CN) descriptions to the flattened IMS NFC
-- level 1-3 hierarchy; the English description is the fallback
-- when no CN name exists for a code.
-- Source : dws.dws_ext_td_ims_nfc_hierarchy,
-- dwd.dwd_gnd_ims_tblAPP (per-level CN names)
-- Target : dws.dws_ext_td_ims_nfc_cn
-- Grain : One row per distinct NFC1..NFC3 hierarchy path; lower levels
-- may be null when a parent code has no matching child.
-- Write mode : Full refresh (INSERT OVERWRITE).
-- Replaces : dws.dws_ims_td_nfc_cn (legacy output).
-- Consumers : dm_ims_td_pack_property (DIM_NFC) -- must migrate to the DWS
-- target.
-- Notes : Hierarchy source repointed from dwd.dwd_ims_nfc_hierarchy to
-- dws.dws_ext_td_ims_nfc_hierarchy; the DWS table carries the same
-- code/name columns, so the joins are unchanged. CN-name joins are
-- case-sensitive string equality, preserved from legacy.
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: copy the verified legacy schema without copying data.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_td_ims_nfc_cn
-- LIKE dws.dws_ims_td_nfc_cn;
-- COMMAND ----------
INSERT OVERWRITE TABLE dws.dws_ext_td_ims_nfc_cn (
APP1_COD,
APP1_DES,
APP1_DES_C,
APP2_COD,
APP2_DES,
APP2_DES_C,
APP3_COD,
APP3_DES,
APP3_DES_C,
ETL_INSERT_DT,
ETL_UPDATE_DT
)
WITH hierarchy AS (
SELECT
nfc1_code,
nfc1_des,
nfc2_code,
nfc2_des,
nfc3_code,
nfc3_des
FROM dws.dws_ext_td_ims_nfc_hierarchy
),
nfc1_cn AS (
SELECT DISTINCT
app1_cod,
app1_des_c
FROM dwd.dwd_gnd_ims_tblAPP
),
nfc2_cn AS (
SELECT DISTINCT
app2_cod,
app2_des_c
FROM dwd.dwd_gnd_ims_tblAPP
),
nfc3_cn AS (
SELECT DISTINCT
app3_cod,
app3_des_c
FROM dwd.dwd_gnd_ims_tblAPP
)
SELECT DISTINCT
hierarchy.nfc1_code AS app1_cod,
hierarchy.nfc1_des AS app1_des,
COALESCE(nfc1_cn.app1_des_c, hierarchy.nfc1_des) AS app1_des_c,
hierarchy.nfc2_code AS app2_cod,
hierarchy.nfc2_des AS app2_des,
COALESCE(nfc2_cn.app2_des_c, hierarchy.nfc2_des) AS app2_des_c,
hierarchy.nfc3_code AS app3_cod,
hierarchy.nfc3_des AS app3_des,
COALESCE(nfc3_cn.app3_des_c, hierarchy.nfc3_des) AS app3_des_c,
FROM_UTC_TIMESTAMP(CURRENT_TIMESTAMP(), 'UTC+8') AS etl_insert_dt,
FROM_UTC_TIMESTAMP(CURRENT_TIMESTAMP(), 'UTC+8') AS etl_update_dt
FROM hierarchy
LEFT JOIN nfc1_cn
ON hierarchy.nfc1_code = nfc1_cn.app1_cod
LEFT JOIN nfc2_cn
ON hierarchy.nfc2_code = nfc2_cn.app2_cod
LEFT JOIN nfc3_cn
ON hierarchy.nfc3_code = nfc3_cn.app3_cod
;
@@ -0,0 +1,38 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Copy the market-to-TA mapping as a DWS dimension.
-- Source : dwd.dwd_gnd_ims_tblmarket_ta_map
-- Target : dws.dws_ext_td_ims_market_ta
-- Grain : One row per source mapping row (grain defined by the source;
-- no DISTINCT, matching legacy).
-- Write mode : Full refresh (INSERT OVERWRITE).
-- Replaces : dws.dws_ims_td_market_ta (legacy output).
-- Consumers : None found in the repo today (03 reads the DWD map directly).
-- Table is kept for lineage parity; wire consumers during
-- migration.
-- Notes : Legacy SELECT * is replaced with an explicit MARKET, TA
-- projection; if the source contains additional columns they are
-- intentionally not carried -- confirm against the source DDL.
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: infer the explicit two-column contract without copying data.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_td_ims_market_ta AS
-- SELECT
-- MARKET,
-- TA
-- FROM dwd.dwd_gnd_ims_tblmarket_ta_map
-- WHERE 1 = 0;
-- COMMAND ----------
INSERT OVERWRITE TABLE dws.dws_ext_td_ims_market_ta (
MARKET,
TA
)
SELECT
market,
ta
FROM dwd.dwd_gnd_ims_tblmarket_ta_map
;
@@ -0,0 +1,68 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Maintain the rolling 5-year (ym, pack_id) -> pack_code snapshot.
-- The recent 5-year window is refreshed with the current pack
-- mapping; rows older than the window are retained unchanged from
-- previous runs (legacy delete+insert behavior).
-- Source : dwd.dwd_ims_tf_fact_sales (ym range)
-- dwd.dwd_ims_td_pack (distinct pack_id, pack_code mapping)
-- Target : dws.dws_ext_td_ims_pack_ym
-- Grain : One row per (ym, pack_id).
-- Write mode : Delete + insert (legacy rolling maintenance preserved; NOT a
-- full refresh -- converting to INSERT OVERWRITE would drop the
-- retained older-than-5-year rows).
-- Replaces : dws.dws_ims_td_pack_ym (legacy table, same maintenance logic).
-- Consumers : dws.dws_ext_tf_ims_chpa_sales (Part 1 national pack_code join).
-- Notes : The legacy temp view "dwd_ims_td_pack" (distinct pack_id,
-- pack_code) is inlined as CTE pack_dim. The inner join between
-- distinct ym and distinct packs is a cross join, exactly as in
-- the legacy script. The first deployment must DEEP CLONE the
-- legacy table so rows outside the rolling refresh window are
-- retained with their original schema and timestamps.
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: this stateful rolling table must seed retained history as well
-- as schema. An empty LIKE table would permanently lose rows outside the
-- five-year refresh window on the first run.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_td_ims_pack_ym
-- DEEP CLONE dws.dws_ims_td_pack_ym;
-- COMMAND ----------
DELETE FROM dws.dws_ext_td_ims_pack_ym
WHERE ym + 500 > (SELECT MAX(year * 100 + month) FROM dwd.dwd_ims_tf_fact_sales);
-- COMMAND ----------
INSERT INTO dws.dws_ext_td_ims_pack_ym (
ym,
pack_id,
pack_code,
ETL_INSERT_DT,
ETL_UPDATE_DT
)
WITH fact_ym AS (
-- Distinct year-months within the recent 5-year window.
SELECT DISTINCT
year * 100 + month AS ym
FROM dwd.dwd_ims_tf_fact_sales
WHERE year * 100 + month + 500 > (SELECT MAX(year * 100 + month) FROM dwd.dwd_ims_tf_fact_sales)
),
pack_dim AS (
-- Current pack_id -> pack_code mapping (legacy temp view inlined).
SELECT DISTINCT
pack_id,
pack_code
FROM dwd.dwd_ims_td_pack
)
SELECT DISTINCT
fact_ym.ym,
pack_dim.pack_id,
pack_dim.pack_code,
from_utc_timestamp(current_timestamp(), 'UTC+8') AS ETL_INSERT_DT,
from_utc_timestamp(current_timestamp(), 'UTC+8') AS ETL_UPDATE_DT
FROM fact_ym
CROSS JOIN pack_dim
;
@@ -0,0 +1,237 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Build the unified CHPA fact sales table: national (CHT) rows
-- from IMS plus province (CHPA) rows from Pharbers, each row
-- carrying current and prior-year (LY) measures in MTH00* columns.
-- Source : dwd.dwd_gnd_pharbers_prov_fact (province fact)
-- dws.dws_ext_td_ims_geo (province dimension, replaces the legacy
-- dm.dm_td_geography normalization -- see Notes)
-- dwd.dwd_gnd_dept_pack_property (IQVIA pack code / counting-unit
-- ratio)
-- dwd.dwd_ims_tf_fact_sales (national fact)
-- dws.dws_ext_td_ims_pack_ym (pack_id -> pack_code by ym)
-- dwd.dwd_ims_td_audit (Audit_Code = 'CHT' filter)
-- Target : dws.dws_ext_tf_ims_chpa_sales
-- Grain : One row per (YM, AUDIT_COD, PACK_COD).
-- Write mode : Full refresh (INSERT OVERWRITE).
-- Replaces : tmp.tmp_ims_tf_fact_sales (legacy staging fact).
-- Consumers : dws.dws_ext_td_ims_date, dm_ims_tf_sales (L2Y view), DM layer.
-- Notes :
-- * Province mapping now joins source PROVINCE_C directly to
-- dws.dws_ext_td_ims_geo.PROVINCE_C and outputs geo.AUDIT_COD. This
-- REPLACES the legacy DIM_PROVINCE normalization over dm.dm_td_geography
-- (CONCAT of 市/自治区/省 suffixes) and removes the DWS -> DM layer
-- violation. Contract: dws_ext_td_ims_geo must be province-grain (one row
-- per province, no city rows) and its PROVINCE_C values must exactly match
-- dwd_gnd_pharbers_prov_fact.PROVINCE_C; unmatched provinces produce NULL
-- AUDIT_COD and collapse into one NULL group, exactly as legacy.
-- * Mixed pack-code domains preserved: Part 1 PACK_COD = IMS pack_code (via
-- pack_ym), Part 2 PACK_COD = IQVIA_PACK_CODE. Do NOT join the two parts
-- on PACK_COD.
-- * Asymmetric time windows preserved: Part 1 filters YM >= 202201
-- (20260320 chenwu CHPA-only rule); Part 2 has no lower bound. Both parts
-- are capped at yearmont_range = MIN(max province ym, max national ym).
-- * LY construction preserved: current rows carry MTH00* and LY-shifted rows
-- (YM+1 year, values in *LY columns) are unioned and summed into the same
-- (YM, AUDIT_COD, PACK_COD) key.
-- * COUNTINGUNIT_RATIO fallback to conversion_ratio preserved.
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: copy the verified legacy schema without copying data.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_tf_ims_chpa_sales
-- LIKE tmp.tmp_ims_tf_fact_sales;
-- COMMAND ----------
INSERT OVERWRITE TABLE dws.dws_ext_tf_ims_chpa_sales (
YM,
AUDIT_COD,
PACK_COD,
MTH00LC,
MTH00LCLY,
MTH00CN,
MTH00CNLY,
MTH00UN,
MTH00UNLY
)
WITH yearmont_range AS (
-- Cap both parts at the smaller of the two sources' max month to avoid a
-- leading-corner mismatch between province and national data.
SELECT
MIN(ym) AS ym
FROM (
SELECT MAX(ym) AS ym FROM dwd.dwd_gnd_pharbers_prov_fact
UNION ALL
SELECT MAX(year * 100 + month) AS ym FROM dwd.dwd_ims_tf_fact_sales
)
),
prov_fact AS (
-- Pharbers province fact with normalized types (mirrors the legacy
-- FACT_CHPA_SALES_TEMP_WITH_PREVIOUS view).
SELECT
CAST(ym AS INT) AS ym,
CAST(year AS INT) AS year,
CAST(REPLACE(ym, year, '') AS INT) AS month,
CAST(value AS DECIMAL(38, 10)) AS value,
CAST(countingunit AS DECIMAL(38, 10)) AS countingunit,
CAST(totalunit AS DECIMAL(38, 10)) AS totalunit,
province_c,
phcd,
conversion_ratio
FROM dwd.dwd_gnd_pharbers_prov_fact
),
geo_province AS (
-- Province dimension from DWS. Replaces the legacy DIM_PROVINCE view that
-- normalized dm.dm_td_geography province names with CONCAT suffixes; the
-- fact's PROVINCE_C now matches geo.PROVINCE_C directly.
SELECT
province_c,
audit_cod
FROM dws.dws_ext_td_ims_geo
),
chpa_pack_info AS (
-- IQVIA pack code and counting-unit ratio (mirrors the legacy
-- DIM_CHPA_PACK_INFO view; MAX() dedup preserved).
SELECT
pack_cod,
MAX(iqvia_pack_code) AS iqvia_pack_code,
MAX(countingunit) AS countingunit_ratio
FROM dwd.dwd_gnd_dept_pack_property
GROUP BY
pack_cod
),
fact_chpa_sales AS (
-- Province rows: current month plus the same month one year earlier
-- (LY), unioned so the final GROUP BY recombines them per (YM, audit,
-- pack). PACK_COD is the IQVIA pack code domain.
SELECT
prov_fact.ym,
prov_fact.year,
REPLACE(prov_fact.ym, prov_fact.year, '') AS month,
chpa_pack_info.iqvia_pack_code AS pack_code,
geo_province.audit_cod,
prov_fact.value AS mth00lc,
0 AS mth00lcly,
CASE
WHEN chpa_pack_info.countingunit_ratio IS NULL
THEN prov_fact.totalunit * prov_fact.conversion_ratio
ELSE prov_fact.totalunit * chpa_pack_info.countingunit_ratio
END AS mth00cn,
0 AS mth00cnly,
prov_fact.totalunit AS mth00un,
0 AS mth00unly
FROM prov_fact
LEFT JOIN geo_province
ON prov_fact.province_c = geo_province.province_c
LEFT JOIN chpa_pack_info
ON prov_fact.phcd = chpa_pack_info.pack_cod
UNION ALL
SELECT
CAST((prov_fact.year + 1) * 100 + REPLACE(prov_fact.ym, prov_fact.year, '') AS INT) AS ym,
prov_fact.year + 1 AS year,
CAST(REPLACE(prov_fact.ym, prov_fact.year, '') AS INT) AS month,
chpa_pack_info.iqvia_pack_code AS pack_code,
geo_province.audit_cod,
0 AS mth00lc,
prov_fact.value AS mth00lcly,
0 AS mth00cn,
CASE
WHEN chpa_pack_info.countingunit_ratio IS NULL
THEN prov_fact.totalunit * prov_fact.conversion_ratio
ELSE prov_fact.totalunit * chpa_pack_info.countingunit_ratio
END AS mth00cnly,
0 AS mth00un,
prov_fact.totalunit AS mth00unly
FROM prov_fact
LEFT JOIN geo_province
ON prov_fact.province_c = geo_province.province_c
LEFT JOIN chpa_pack_info
ON prov_fact.phcd = chpa_pack_info.pack_cod
),
national_sales AS (
-- National rows: current month plus LY-shifted rows; pack_code comes from
-- the rolling pack_ym snapshot (IMS pack_code domain -- mixed with the
-- IQVIA domain in Part 2 by design).
SELECT
year * 100 + month AS ym,
year,
month,
pack_ym.pack_code,
audit_id,
sales_value_lc AS mth00lc,
0 AS mth00lcly,
counting_unit AS mth00cn,
0 AS mth00cnly,
sales_unit AS mth00un,
0 AS mth00unly
FROM dwd.dwd_ims_tf_fact_sales fact_sales
LEFT JOIN dws.dws_ext_td_ims_pack_ym pack_ym
ON fact_sales.pack_id = pack_ym.pack_id
AND fact_sales.year * 100 + fact_sales.month = pack_ym.ym
UNION ALL
SELECT
(year + 1) * 100 + month AS ym,
year + 1 AS year,
month,
pack_ym.pack_code,
audit_id,
0 AS mth00lc,
sales_value_lc AS mth00lcly,
0 AS mth00cn,
counting_unit AS mth00cnly,
0 AS mth00un,
sales_unit AS mth00unly
FROM dwd.dwd_ims_tf_fact_sales fact_sales
LEFT JOIN dws.dws_ext_td_ims_pack_ym pack_ym
ON fact_sales.pack_id = pack_ym.pack_id
AND fact_sales.year * 100 + fact_sales.month = pack_ym.ym
)
-- Part 1: national CHT rows. Asymmetric window preserved: YM >= 202201 and
-- YM <= yearmont_range (the legacy 20260320 CHPA-only lower bound).
SELECT
national_sales.ym AS YM,
audit.audit_code AS AUDIT_COD,
national_sales.pack_code AS PACK_COD,
SUM(national_sales.mth00lc) AS MTH00LC,
SUM(national_sales.mth00lcly) AS MTH00LCLY,
SUM(national_sales.mth00cn) AS MTH00CN,
SUM(national_sales.mth00cnly) AS MTH00CNLY,
SUM(national_sales.mth00un) AS MTH00UN,
SUM(national_sales.mth00unly) AS MTH00UNLY
FROM national_sales
LEFT JOIN dwd.dwd_ims_td_audit audit
ON national_sales.audit_id = audit.audit_id
WHERE national_sales.ym <= (SELECT ym FROM yearmont_range)
AND national_sales.ym >= 202201
AND audit.audit_code = 'CHT'
GROUP BY
national_sales.ym,
audit.audit_code,
national_sales.pack_code
UNION ALL
-- Part 2: province CHPA rows. No lower ym bound (asymmetric with Part 1).
SELECT
fact_chpa_sales.ym AS YM,
fact_chpa_sales.audit_cod AS AUDIT_COD,
fact_chpa_sales.pack_code AS PACK_COD,
SUM(fact_chpa_sales.mth00lc) AS MTH00LC,
SUM(fact_chpa_sales.mth00lcly) AS MTH00LCLY,
SUM(fact_chpa_sales.mth00cn) AS MTH00CN,
SUM(fact_chpa_sales.mth00cnly) AS MTH00CNLY,
SUM(fact_chpa_sales.mth00un) AS MTH00UN,
SUM(fact_chpa_sales.mth00unly) AS MTH00UNLY
FROM fact_chpa_sales
WHERE fact_chpa_sales.ym <= (SELECT ym FROM yearmont_range)
GROUP BY
fact_chpa_sales.ym,
fact_chpa_sales.audit_cod,
fact_chpa_sales.pack_code
;
@@ -0,0 +1,84 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Build the dynamic month dimension from the CHPA fact sales
-- range, with the report month flagged 'R' (max YM present).
-- Source : dws.dws_ext_tf_ims_chpa_sales (YM range; reads the promoted
-- DWS fact instead of the legacy tmp.tmp_ims_tf_fact_sales)
-- Target : dws.dws_ext_td_ims_date
-- Grain : One row per YM present in the fact (approximately nine years by
-- the legacy YYYYMM-minus-900 arithmetic).
-- Write mode : Full refresh (INSERT OVERWRITE).
-- Replaces : dws.dws_ims_td_date (legacy output, same column contract and
-- the same 900 arithmetic: YM > max(YM) - 900).
-- Consumers : DM calendar / pack-property jobs must migrate to this target.
-- Notes : Legacy header comment claims "最近五年" (5 years), but subtracting
-- 900 from a YYYYMM value gives approximately nine years; that
-- arithmetic is preserved exactly. QUARTER/YQ/HALF_YEAR/DATE_FLAG
-- preserved verbatim.
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: copy the verified legacy schema without copying data.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_td_ims_date
-- LIKE dws.dws_ims_td_date;
-- COMMAND ----------
INSERT OVERWRITE TABLE dws.dws_ext_td_ims_date (
YM,
YEAR,
MONTH,
QUARTER,
YQ,
DATE_FLAG,
HALF_YEAR,
ETL_INSERT_DT,
ETL_UPDATE_DT
)
SELECT DISTINCT
fact_sales.ym AS YM,
LEFT(CAST(fact_sales.ym AS STRING), 4) AS YEAR,
RIGHT(CAST(fact_sales.ym AS STRING), 2) AS MONTH,
CONCAT(
'Q',
QUARTER(
DATE(
CONCAT(
LEFT(CAST(fact_sales.ym AS STRING), 4),
'-',
RIGHT(CAST(fact_sales.ym AS STRING), 2),
'-01'
)
)
)
) AS QUARTER,
CONCAT(
LEFT(CAST(fact_sales.ym AS STRING), 4),
'Q',
QUARTER(
DATE(
CONCAT(
LEFT(CAST(fact_sales.ym AS STRING), 4),
'-',
RIGHT(CAST(fact_sales.ym AS STRING), 2),
'-01'
)
)
)
) AS YQ,
CASE
WHEN fact_sales.ym = (SELECT MAX(ym) FROM dws.dws_ext_tf_ims_chpa_sales) THEN 'R'
ELSE RIGHT(CAST(fact_sales.ym AS STRING), 2)
END AS DATE_FLAG,
CASE
WHEN fact_sales.ym % 100 > 6
THEN CONCAT(LEFT(CAST(fact_sales.ym AS STRING), 4), 'H2')
ELSE CONCAT(LEFT(CAST(fact_sales.ym AS STRING), 4), 'H1')
END AS HALF_YEAR,
from_utc_timestamp(current_timestamp(), 'UTC+8') AS ETL_INSERT_DT,
from_utc_timestamp(current_timestamp(), 'UTC+8') AS ETL_UPDATE_DT
FROM dws.dws_ext_tf_ims_chpa_sales fact_sales
WHERE fact_sales.ym > (SELECT MAX(ym) - 900 FROM dws.dws_ext_tf_ims_chpa_sales)
ORDER BY fact_sales.ym DESC
;
@@ -0,0 +1,813 @@
-- Databricks notebook source
-- =============================================================================
-- Purpose : Build the IMS market dictionary: one row per market x pack with
-- pack attributes, business unit, market ratio and key competitor.
-- Source : dws.dws_ext_td_ims_pack_property (31-column pack hub)
-- dwd.dwd_gnd_ims_tblmarket (include / exclude / extend rules)
-- dwd.dwd_gnd_ims_tblkeycompetitor (key-competitor rules)
-- Target : dws.dws_ext_td_ims_market
-- Grain : Distinct market x pack (attributes, bu, market_ratio); at most one
-- Key_Competitor per (market, PACK_COD, PROD_COD). The same
-- (market, PACK_COD) may fan out across bu / market_ratio rows.
-- Write mode : Full refresh (INSERT OVERWRITE).
-- Replaces : dws.dws_ims_td_market (legacy output, kept as validation baseline).
-- Consumers : dm_ims_td_org / dm_ims_td_org_hvh (LEFT JOIN on pack_cod -> market).
-- Depends on : 01_dwd pad-zero updates of dwd.dwd_gnd_ims_tblmarket and
-- dwd.dwd_gnd_ims_tblkeycompetitor, and the pack-property job that
-- lands dws.dws_ext_td_ims_pack_property.
-- =============================================================================
-- Compatibility notes (legacy behavior preserved, NOT fixed):
-- 1. Extend-market ratio: the legacy Market_Ratio CASE for extended rows had no
-- ELSE branch, so a configured Extend_Market_Ratio yielded NULL, which the
-- legacy cell-5 UPDATE then forced to '1'. NET LEGACY BEHAVIOR: every
-- extended-market row ends with Market_Ratio = '1' and the configured
-- Extend_Market_Ratio is silently ignored. Reproduced verbatim below (buggy
-- CASE in market_extended + NULL -> '1' in market_all). Do not "fix" without
-- business sign-off.
-- 2. MERGE multi-match failure modes removed: the legacy MERGE ... DELETE cells
-- could raise DELTA_MULTIPLE_SOURCE_ROW_MATCHING when several source rows
-- matched one target row (config rows whose 14-key tuple is identical except
-- bu / market_ratio). The anti-join equivalents below apply the delete
-- deterministically and match the MERGE result in every non-error case.
-- 3. KC dedup NULL semantics: the anti-join reproduces the MERGE's SQL equality
-- on (14 keys + Key_Competitor + no). NULL never equals NULL, so -- exactly
-- like legacy -- duplicate rows inside one (market, PACK_COD, PROD_COD)
-- partition that carry NO key-competitor match (Key_Competitor / no NULL) are
-- NOT deduped; all of them survive and later become 'Others'. A naive
-- rank = 1 filter would have dropped them; this script does not.
-- 4. no1 window preserved verbatim, including the duplicated NFC2_CODE sort key
-- (legacy lines 152-155). ROW_NUMBER() OVER(ORDER BY ...) has no total order;
-- among exact ties the winner is engine-defined (same as legacy).
-- 5. Pack hub swap: t1 now reads dws.dws_ext_td_ims_pack_property. Equivalence
-- holds only if that table is a logic-identical copy of
-- dwd.dwd_ims_td_pack_property (same 31-column contract incl. zero-padded
-- codes and non-null CORP_COD / STGH_DES).
-- 6. Timestamps: the legacy final INSERT had no column list and two anonymous
-- from_utc_timestamp expressions. They are named ETL_INSERT_DT /
-- ETL_UPDATE_DT here (repo convention); confirm against the LIKE-created DDL
-- via DESCRIBE before the first run.
-- 7. t1.* is replaced by an explicit 31-column projection; if the pack hub ever
-- gains columns this script pins the contract where legacy positional
-- inserts would have silently shifted.
-- 8. DISTINCT is kept at exactly the legacy stages (include, exclude, extend,
-- KC join, final); no cross-stage dedup is added and the base/extended
-- UNION ALL does not dedup between its two branches.
-- 9. Extended rows preserve a legacy positional-write defect: cell 4 selected
-- BU before BrandType into a target whose final fields are BrandType, bu,
-- Market_Ratio. Therefore extended rows expose configured BU as BrandType
-- and the source pack BrandType as bu. This is explicit below so refactoring
-- does not silently change persisted output; correct only with business
-- sign-off in a separate change.
-- =============================================================================
-- COMMAND ----------
-- One-time DDL: copy the verified legacy schema without copying data.
-- CREATE TABLE IF NOT EXISTS dws.dws_ext_td_ims_market
-- LIKE dws.dws_ims_td_market;
-- COMMAND ----------
INSERT OVERWRITE TABLE dws.dws_ext_td_ims_market (
market,
PACK_COD,
PACK_DES,
STGH_DES,
PACK_LCH,
PROD_COD,
CMPS_COD,
CMPS_DES,
ATC1_COD,
ATC2_COD,
ATC3_COD,
ATC4_COD,
APP1_COD,
APP2_COD,
APP3_COD,
BIO_DESC,
GENE_ORIG_DESC,
ETH_OTC_DESC,
NRDL_DESC,
NRDL_Entry_Date,
EDL_DESC,
TCM_DESC,
PAED_DESC,
GQCE_DESC,
VBP_DESC,
MANU_COD,
MANU_DES,
MNFL_COD,
MNFL_DES,
CORP_COD,
CORP_DES,
BrandType,
bu,
Market_Ratio,
Key_Competitor,
ETL_INSERT_DT,
ETL_UPDATE_DT
)
WITH pack_hub AS (
-- 31-column pack property contract (legacy dwd.dwd_ims_td_pack_property).
SELECT
PACK_COD,
PACK_DES,
STGH_DES,
PACK_LCH,
PROD_COD,
CMPS_COD,
CMPS_DES,
ATC1_COD,
ATC2_COD,
ATC3_COD,
ATC4_COD,
APP1_COD,
APP2_COD,
APP3_COD,
BIO_DESC,
GENE_ORIG_DESC,
ETH_OTC_DESC,
NRDL_DESC,
NRDL_Entry_Date,
EDL_DESC,
TCM_DESC,
PAED_DESC,
GQCE_DESC,
VBP_DESC,
MANU_COD,
MANU_DES,
MNFL_COD,
MNFL_DES,
CORP_COD,
CORP_DES,
BrandType
FROM dws.dws_ext_td_ims_pack_property
),
market_config_include AS (
-- Explicitly defined markets (legacy cell 1 filter): Extend_Market IS NULL
-- and NOT_IN_FLAG IS NULL or '1' (despite the name, NOT_IN_FLAG = '1' is an
-- include rule; '0' is the exclude rule handled below).
SELECT
market,
bu,
ATC1_Code AS atc1_code,
ATC2_Code AS atc2_code,
ATC3_Code AS atc3_code,
ATC4_Code AS atc4_code,
NFC1_Code AS nfc1_code,
NFC2_Code AS nfc2_code,
NFC3_Code AS nfc3_code,
corporation_code,
Manufacturer_Code AS manufacturer_code,
Product_Code AS product_code,
Pack_Code AS pack_code,
Strength AS strength,
Molecule_Code AS molecule_code,
extend_market_ratio
FROM dwd.dwd_gnd_ims_tblmarket
WHERE Extend_Market IS NULL
AND (NOT_IN_FLAG IS NULL OR NOT_IN_FLAG = '1')
),
market_config_exclude AS (
-- Anti rules (legacy cell 2 filter): Extend_Market IS NULL and NOT_IN_FLAG = '0'.
SELECT
market,
bu,
ATC1_Code AS atc1_code,
ATC2_Code AS atc2_code,
ATC3_Code AS atc3_code,
ATC4_Code AS atc4_code,
NFC1_Code AS nfc1_code,
NFC2_Code AS nfc2_code,
NFC3_Code AS nfc3_code,
corporation_code,
Manufacturer_Code AS manufacturer_code,
Product_Code AS product_code,
Pack_Code AS pack_code,
Strength AS strength,
Molecule_Code AS molecule_code,
extend_market_ratio
FROM dwd.dwd_gnd_ims_tblmarket
WHERE Extend_Market IS NULL
AND NOT_IN_FLAG = '0'
),
market_config_extend AS (
-- Extend rules (legacy cell 4 filter): Extend_Market IS NOT NULL.
SELECT
Market AS market,
BU AS bu,
Extend_Market AS extend_market,
Extend_Market_Ratio AS extend_market_ratio
FROM dwd.dwd_gnd_ims_tblmarket
WHERE Extend_Market IS NOT NULL
),
market_include_rows AS (
-- Legacy cell 1: wildcard equality -- a config code constrains the match
-- only when non-null (NULL config code = wildcard). Kept as LEFT JOIN with
-- WHERE t2.market IS NOT NULL, exactly like legacy (effectively inner).
SELECT DISTINCT
t2.market AS market,
t1.PACK_COD,
t1.PACK_DES,
t1.STGH_DES,
t1.PACK_LCH,
t1.PROD_COD,
t1.CMPS_COD,
t1.CMPS_DES,
t1.ATC1_COD,
t1.ATC2_COD,
t1.ATC3_COD,
t1.ATC4_COD,
t1.APP1_COD,
t1.APP2_COD,
t1.APP3_COD,
t1.BIO_DESC,
t1.GENE_ORIG_DESC,
t1.ETH_OTC_DESC,
t1.NRDL_DESC,
t1.NRDL_Entry_Date,
t1.EDL_DESC,
t1.TCM_DESC,
t1.PAED_DESC,
t1.GQCE_DESC,
t1.VBP_DESC,
t1.MANU_COD,
t1.MANU_DES,
t1.MNFL_COD,
t1.MNFL_DES,
t1.CORP_COD,
t1.CORP_DES,
t1.BrandType,
t2.bu AS bu,
CASE
WHEN t2.extend_market_ratio IS NULL THEN '1'
ELSE t2.extend_market_ratio
END AS Market_Ratio
FROM pack_hub t1
LEFT JOIN market_config_include t2
ON t1.ATC1_COD = CASE WHEN t2.atc1_code IS NULL THEN t1.ATC1_COD ELSE t2.atc1_code END
AND t1.ATC2_COD = CASE WHEN t2.atc2_code IS NULL THEN t1.ATC2_COD ELSE t2.atc2_code END
AND t1.ATC3_COD = CASE WHEN t2.atc3_code IS NULL THEN t1.ATC3_COD ELSE t2.atc3_code END
AND t1.ATC4_COD = CASE WHEN t2.atc4_code IS NULL THEN t1.ATC4_COD ELSE t2.atc4_code END
AND t1.APP1_COD = CASE WHEN t2.nfc1_code IS NULL THEN t1.APP1_COD ELSE t2.nfc1_code END
AND t1.APP2_COD = CASE WHEN t2.nfc2_code IS NULL THEN t1.APP2_COD ELSE t2.nfc2_code END
AND t1.APP3_COD = CASE WHEN t2.nfc3_code IS NULL THEN t1.APP3_COD ELSE t2.nfc3_code END
AND t1.CORP_COD = CASE WHEN t2.corporation_code IS NULL THEN t1.CORP_COD ELSE t2.corporation_code END
AND t1.MANU_COD = CASE WHEN t2.manufacturer_code IS NULL THEN t1.MANU_COD ELSE t2.manufacturer_code END
AND t1.PROD_COD = CASE WHEN t2.product_code IS NULL THEN t1.PROD_COD ELSE t2.product_code END
AND t1.PACK_COD = CASE WHEN t2.pack_code IS NULL THEN t1.PACK_COD ELSE t2.pack_code END
AND t1.STGH_DES = CASE WHEN t2.strength IS NULL THEN t1.STGH_DES ELSE t2.strength END
AND t1.CMPS_COD = CASE WHEN t2.molecule_code IS NULL THEN t1.CMPS_COD ELSE t2.molecule_code END
WHERE t2.market IS NOT NULL
),
market_exclude_rows AS (
-- Legacy cell 2: same wildcard join against the exclude rules.
SELECT DISTINCT
t2.market AS market,
t1.PACK_COD,
t1.PACK_DES,
t1.STGH_DES,
t1.PACK_LCH,
t1.PROD_COD,
t1.CMPS_COD,
t1.CMPS_DES,
t1.ATC1_COD,
t1.ATC2_COD,
t1.ATC3_COD,
t1.ATC4_COD,
t1.APP1_COD,
t1.APP2_COD,
t1.APP3_COD,
t1.BIO_DESC,
t1.GENE_ORIG_DESC,
t1.ETH_OTC_DESC,
t1.NRDL_DESC,
t1.NRDL_Entry_Date,
t1.EDL_DESC,
t1.TCM_DESC,
t1.PAED_DESC,
t1.GQCE_DESC,
t1.VBP_DESC,
t1.MANU_COD,
t1.MANU_DES,
t1.MNFL_COD,
t1.MNFL_DES,
t1.CORP_COD,
t1.CORP_DES,
t1.BrandType,
t2.bu AS bu,
CASE
WHEN t2.extend_market_ratio IS NULL THEN '1'
ELSE t2.extend_market_ratio
END AS Market_Ratio
FROM pack_hub t1
LEFT JOIN market_config_exclude t2
ON t1.ATC1_COD = CASE WHEN t2.atc1_code IS NULL THEN t1.ATC1_COD ELSE t2.atc1_code END
AND t1.ATC2_COD = CASE WHEN t2.atc2_code IS NULL THEN t1.ATC2_COD ELSE t2.atc2_code END
AND t1.ATC3_COD = CASE WHEN t2.atc3_code IS NULL THEN t1.ATC3_COD ELSE t2.atc3_code END
AND t1.ATC4_COD = CASE WHEN t2.atc4_code IS NULL THEN t1.ATC4_COD ELSE t2.atc4_code END
AND t1.APP1_COD = CASE WHEN t2.nfc1_code IS NULL THEN t1.APP1_COD ELSE t2.nfc1_code END
AND t1.APP2_COD = CASE WHEN t2.nfc2_code IS NULL THEN t1.APP2_COD ELSE t2.nfc2_code END
AND t1.APP3_COD = CASE WHEN t2.nfc3_code IS NULL THEN t1.APP3_COD ELSE t2.nfc3_code END
AND t1.CORP_COD = CASE WHEN t2.corporation_code IS NULL THEN t1.CORP_COD ELSE t2.corporation_code END
AND t1.MANU_COD = CASE WHEN t2.manufacturer_code IS NULL THEN t1.MANU_COD ELSE t2.manufacturer_code END
AND t1.PROD_COD = CASE WHEN t2.product_code IS NULL THEN t1.PROD_COD ELSE t2.product_code END
AND t1.PACK_COD = CASE WHEN t2.pack_code IS NULL THEN t1.PACK_COD ELSE t2.pack_code END
AND t1.STGH_DES = CASE WHEN t2.strength IS NULL THEN t1.STGH_DES ELSE t2.strength END
AND t1.CMPS_COD = CASE WHEN t2.molecule_code IS NULL THEN t1.CMPS_COD ELSE t2.molecule_code END
WHERE t2.market IS NOT NULL
),
market_after_exclude AS (
-- Legacy cell 3 MERGE ... WHEN MATCHED THEN DELETE, expressed as an
-- anti-join on the 14 materialized keys (13 codes + market). NULL never
-- equals NULL, exactly like the MERGE join.
SELECT
t1.market,
t1.PACK_COD,
t1.PACK_DES,
t1.STGH_DES,
t1.PACK_LCH,
t1.PROD_COD,
t1.CMPS_COD,
t1.CMPS_DES,
t1.ATC1_COD,
t1.ATC2_COD,
t1.ATC3_COD,
t1.ATC4_COD,
t1.APP1_COD,
t1.APP2_COD,
t1.APP3_COD,
t1.BIO_DESC,
t1.GENE_ORIG_DESC,
t1.ETH_OTC_DESC,
t1.NRDL_DESC,
t1.NRDL_Entry_Date,
t1.EDL_DESC,
t1.TCM_DESC,
t1.PAED_DESC,
t1.GQCE_DESC,
t1.VBP_DESC,
t1.MANU_COD,
t1.MANU_DES,
t1.MNFL_COD,
t1.MNFL_DES,
t1.CORP_COD,
t1.CORP_DES,
t1.BrandType,
t1.bu,
t1.Market_Ratio
FROM market_include_rows t1
LEFT ANTI JOIN market_exclude_rows t2
ON t1.ATC1_COD = t2.ATC1_COD
AND t1.ATC2_COD = t2.ATC2_COD
AND t1.ATC3_COD = t2.ATC3_COD
AND t1.ATC4_COD = t2.ATC4_COD
AND t1.APP1_COD = t2.APP1_COD
AND t1.APP2_COD = t2.APP2_COD
AND t1.APP3_COD = t2.APP3_COD
AND t1.CORP_COD = t2.CORP_COD
AND t1.MANU_COD = t2.MANU_COD
AND t1.PROD_COD = t2.PROD_COD
AND t1.PACK_COD = t2.PACK_COD
AND t1.STGH_DES = t2.STGH_DES
AND t1.CMPS_COD = t2.CMPS_COD
AND t1.market = t2.market
),
market_extended AS (
-- Legacy cell 4: clone every pack of an existing market into the extend
-- market (append). Market_Ratio keeps the legacy missing-ELSE CASE exactly
-- (a configured extend_market_ratio yields NULL here; see compatibility
-- note 1 -- it is intentionally NOT fixed).
SELECT DISTINCT
t2.market AS market,
t1.PACK_COD,
t1.PACK_DES,
t1.STGH_DES,
t1.PACK_LCH,
t1.PROD_COD,
t1.CMPS_COD,
t1.CMPS_DES,
t1.ATC1_COD,
t1.ATC2_COD,
t1.ATC3_COD,
t1.ATC4_COD,
t1.APP1_COD,
t1.APP2_COD,
t1.APP3_COD,
t1.BIO_DESC,
t1.GENE_ORIG_DESC,
t1.ETH_OTC_DESC,
t1.NRDL_DESC,
t1.NRDL_Entry_Date,
t1.EDL_DESC,
t1.TCM_DESC,
t1.PAED_DESC,
t1.GQCE_DESC,
t1.VBP_DESC,
t1.MANU_COD,
t1.MANU_DES,
t1.MNFL_COD,
t1.MNFL_DES,
t1.CORP_COD,
t1.CORP_DES,
t2.bu AS BrandType,
t1.BrandType AS bu,
CASE
WHEN t2.extend_market_ratio IS NULL THEN '1'
END AS Market_Ratio
FROM market_after_exclude t1
LEFT JOIN market_config_extend t2
ON t1.market = t2.extend_market
WHERE t2.market IS NOT NULL
),
market_all AS (
-- Legacy cells 4 (INSERT INTO append) + 5 (UPDATE Market_Ratio = 1 WHERE
-- Market_Ratio IS NULL): union of base and extend rows, then NULL ratios
-- default to '1'. Together with the missing-ELSE CASE above this makes every
-- extended row end with Market_Ratio = '1' (compatibility note 1).
SELECT
market,
PACK_COD,
PACK_DES,
STGH_DES,
PACK_LCH,
PROD_COD,
CMPS_COD,
CMPS_DES,
ATC1_COD,
ATC2_COD,
ATC3_COD,
ATC4_COD,
APP1_COD,
APP2_COD,
APP3_COD,
BIO_DESC,
GENE_ORIG_DESC,
ETH_OTC_DESC,
NRDL_DESC,
NRDL_Entry_Date,
EDL_DESC,
TCM_DESC,
PAED_DESC,
GQCE_DESC,
VBP_DESC,
MANU_COD,
MANU_DES,
MNFL_COD,
MNFL_DES,
CORP_COD,
CORP_DES,
BrandType,
bu,
CASE
WHEN Market_Ratio IS NULL THEN '1'
ELSE Market_Ratio
END AS Market_Ratio
FROM (
SELECT
market,
PACK_COD,
PACK_DES,
STGH_DES,
PACK_LCH,
PROD_COD,
CMPS_COD,
CMPS_DES,
ATC1_COD,
ATC2_COD,
ATC3_COD,
ATC4_COD,
APP1_COD,
APP2_COD,
APP3_COD,
BIO_DESC,
GENE_ORIG_DESC,
ETH_OTC_DESC,
NRDL_DESC,
NRDL_Entry_Date,
EDL_DESC,
TCM_DESC,
PAED_DESC,
GQCE_DESC,
VBP_DESC,
MANU_COD,
MANU_DES,
MNFL_COD,
MNFL_DES,
CORP_COD,
CORP_DES,
BrandType,
bu,
Market_Ratio
FROM market_after_exclude
UNION ALL
SELECT
market,
PACK_COD,
PACK_DES,
STGH_DES,
PACK_LCH,
PROD_COD,
CMPS_COD,
CMPS_DES,
ATC1_COD,
ATC2_COD,
ATC3_COD,
ATC4_COD,
APP1_COD,
APP2_COD,
APP3_COD,
BIO_DESC,
GENE_ORIG_DESC,
ETH_OTC_DESC,
NRDL_DESC,
NRDL_Entry_Date,
EDL_DESC,
TCM_DESC,
PAED_DESC,
GQCE_DESC,
VBP_DESC,
MANU_COD,
MANU_DES,
MNFL_COD,
MNFL_DES,
CORP_COD,
CORP_DES,
BrandType,
bu,
Market_Ratio
FROM market_extended
) market_union
),
key_competitor_rules AS (
-- Legacy cell 6 source: no1 = specificity rank over all KC rules, computed
-- verbatim -- including the duplicated NFC2_CODE sort key (compatibility
-- note 4; ROW_NUMBER() has no total order, ties are engine-defined).
SELECT
ROW_NUMBER() OVER (
ORDER BY
CASE
WHEN ATC1_Code IS NOT NULL THEN 1
WHEN ATC2_Code IS NOT NULL THEN 2
WHEN ATC3_Code IS NOT NULL THEN 3
WHEN ATC4_Code IS NOT NULL THEN 4
WHEN Molecule_Code IS NOT NULL THEN 5
WHEN Product_Code IS NOT NULL THEN 6
WHEN Pack_Code IS NOT NULL THEN 7
ELSE 999
END,
CASE WHEN NFC1_CODE IS NULL THEN 0 ELSE 1 END,
CASE WHEN NFC2_CODE IS NULL THEN 0 ELSE 1 END,
CASE WHEN NFC2_CODE IS NULL THEN 0 ELSE 1 END,
CASE WHEN NFC3_CODE IS NULL THEN 0 ELSE 1 END
) AS no1,
keycompetitor,
no,
market,
ATC1_Code AS atc1_code,
ATC2_Code AS atc2_code,
ATC3_Code AS atc3_code,
ATC4_Code AS atc4_code,
NFC1_CODE AS nfc1_code,
NFC2_CODE AS nfc2_code,
NFC3_CODE AS nfc3_code,
corporation_code,
Manufacturer_Code AS manufacturer_code,
Product_Code AS product_code,
Pack_Code AS pack_code,
Strength AS strength,
Molecule_Code AS molecule_code
FROM dwd.dwd_gnd_ims_tblkeycompetitor
),
market_kc AS (
-- Legacy cell 6: wildcard LEFT JOIN to key-competitor rules on the 13 codes
-- plus market; unmatched packs keep NULL Key_Competitor / no / no1.
SELECT DISTINCT
t2.keycompetitor AS Key_Competitor,
t2.no AS no,
t2.no1 AS no1,
t1.market,
t1.PACK_COD,
t1.PACK_DES,
t1.STGH_DES,
t1.PACK_LCH,
t1.PROD_COD,
t1.CMPS_COD,
t1.CMPS_DES,
t1.ATC1_COD,
t1.ATC2_COD,
t1.ATC3_COD,
t1.ATC4_COD,
t1.APP1_COD,
t1.APP2_COD,
t1.APP3_COD,
t1.BIO_DESC,
t1.GENE_ORIG_DESC,
t1.ETH_OTC_DESC,
t1.NRDL_DESC,
t1.NRDL_Entry_Date,
t1.EDL_DESC,
t1.TCM_DESC,
t1.PAED_DESC,
t1.GQCE_DESC,
t1.VBP_DESC,
t1.MANU_COD,
t1.MANU_DES,
t1.MNFL_COD,
t1.MNFL_DES,
t1.CORP_COD,
t1.CORP_DES,
t1.BrandType,
t1.bu,
t1.Market_Ratio
FROM market_all t1
LEFT JOIN key_competitor_rules t2
ON t1.ATC1_COD = CASE WHEN t2.atc1_code IS NULL THEN t1.ATC1_COD ELSE t2.atc1_code END
AND t1.ATC2_COD = CASE WHEN t2.atc2_code IS NULL THEN t1.ATC2_COD ELSE t2.atc2_code END
AND t1.ATC3_COD = CASE WHEN t2.atc3_code IS NULL THEN t1.ATC3_COD ELSE t2.atc3_code END
AND t1.ATC4_COD = CASE WHEN t2.atc4_code IS NULL THEN t1.ATC4_COD ELSE t2.atc4_code END
AND t1.APP1_COD = CASE WHEN t2.nfc1_code IS NULL THEN t1.APP1_COD ELSE t2.nfc1_code END
AND t1.APP2_COD = CASE WHEN t2.nfc2_code IS NULL THEN t1.APP2_COD ELSE t2.nfc2_code END
AND t1.APP3_COD = CASE WHEN t2.nfc3_code IS NULL THEN t1.APP3_COD ELSE t2.nfc3_code END
AND t1.CORP_COD = CASE WHEN t2.corporation_code IS NULL THEN t1.CORP_COD ELSE t2.corporation_code END
AND t1.MANU_COD = CASE WHEN t2.manufacturer_code IS NULL THEN t1.MANU_COD ELSE t2.manufacturer_code END
AND t1.PROD_COD = CASE WHEN t2.product_code IS NULL THEN t1.PROD_COD ELSE t2.product_code END
AND t1.PACK_COD = CASE WHEN t2.pack_code IS NULL THEN t1.PACK_COD ELSE t2.pack_code END
AND t1.STGH_DES = CASE WHEN t2.strength IS NULL THEN t1.STGH_DES ELSE t2.strength END
AND t1.CMPS_COD = CASE WHEN t2.molecule_code IS NULL THEN t1.CMPS_COD ELSE t2.molecule_code END
AND t1.market = CASE WHEN t2.market IS NULL THEN t1.market ELSE t2.market END
),
market_kc_ranked AS (
-- Legacy cell 7: rank KC rules per (market, PACK_COD, PROD_COD); the winner
-- is the highest no1, ties broken by the highest no (latest rule). Ordering
-- is preserved verbatim (no1 DESC, no DESC).
SELECT
row_number() OVER (
PARTITION BY market, PACK_COD, PROD_COD
ORDER BY no1 DESC, no DESC
) AS rank_id,
market,
PACK_COD,
PACK_DES,
STGH_DES,
PACK_LCH,
PROD_COD,
CMPS_COD,
CMPS_DES,
ATC1_COD,
ATC2_COD,
ATC3_COD,
ATC4_COD,
APP1_COD,
APP2_COD,
APP3_COD,
BIO_DESC,
GENE_ORIG_DESC,
ETH_OTC_DESC,
NRDL_DESC,
NRDL_Entry_Date,
EDL_DESC,
TCM_DESC,
PAED_DESC,
GQCE_DESC,
VBP_DESC,
MANU_COD,
MANU_DES,
MNFL_COD,
MNFL_DES,
CORP_COD,
CORP_DES,
BrandType,
bu,
Market_Ratio,
Key_Competitor,
no,
no1
FROM market_kc
),
market_kc_losers AS (
-- Legacy cell 7 output (id > 1): every non-winning row, keyed by the 16
-- columns the legacy MERGE used to delete it.
SELECT
ATC1_COD,
ATC2_COD,
ATC3_COD,
ATC4_COD,
APP1_COD,
APP2_COD,
APP3_COD,
CORP_COD,
MANU_COD,
PROD_COD,
PACK_COD,
STGH_DES,
CMPS_COD,
market,
Key_Competitor,
no
FROM market_kc_ranked
WHERE rank_id > 1
),
market_kc_kept AS (
-- Legacy cell 8 MERGE ... WHEN MATCHED THEN DELETE as an anti-join on
-- (14 keys + Key_Competitor + no). SQL equality semantics are preserved:
-- a row is deleted only when every join column is non-null and equal, so
-- KC-less duplicates inside one partition survive exactly as in legacy
-- (compatibility note 3).
SELECT
t1.market,
t1.PACK_COD,
t1.PACK_DES,
t1.STGH_DES,
t1.PACK_LCH,
t1.PROD_COD,
t1.CMPS_COD,
t1.CMPS_DES,
t1.ATC1_COD,
t1.ATC2_COD,
t1.ATC3_COD,
t1.ATC4_COD,
t1.APP1_COD,
t1.APP2_COD,
t1.APP3_COD,
t1.BIO_DESC,
t1.GENE_ORIG_DESC,
t1.ETH_OTC_DESC,
t1.NRDL_DESC,
t1.NRDL_Entry_Date,
t1.EDL_DESC,
t1.TCM_DESC,
t1.PAED_DESC,
t1.GQCE_DESC,
t1.VBP_DESC,
t1.MANU_COD,
t1.MANU_DES,
t1.MNFL_COD,
t1.MNFL_DES,
t1.CORP_COD,
t1.CORP_DES,
t1.BrandType,
t1.bu,
t1.Market_Ratio,
t1.Key_Competitor
FROM market_kc t1
LEFT ANTI JOIN market_kc_losers t2
ON t1.ATC1_COD = t2.ATC1_COD
AND t1.ATC2_COD = t2.ATC2_COD
AND t1.ATC3_COD = t2.ATC3_COD
AND t1.ATC4_COD = t2.ATC4_COD
AND t1.APP1_COD = t2.APP1_COD
AND t1.APP2_COD = t2.APP2_COD
AND t1.APP3_COD = t2.APP3_COD
AND t1.CORP_COD = t2.CORP_COD
AND t1.MANU_COD = t2.MANU_COD
AND t1.PROD_COD = t2.PROD_COD
AND t1.PACK_COD = t2.PACK_COD
AND t1.STGH_DES = t2.STGH_DES
AND t1.CMPS_COD = t2.CMPS_COD
AND t1.market = t2.market
AND t1.Key_Competitor = t2.Key_Competitor
AND t1.no = t2.no
)
-- Legacy cell 9 (Key_Competitor = 'Others' WHERE NULL) is the COALESCE below and
-- legacy cell 10 (drop no / no1) is handled by the explicit projection.
SELECT DISTINCT
market,
PACK_COD,
PACK_DES,
STGH_DES,
PACK_LCH,
PROD_COD,
CMPS_COD,
CMPS_DES,
ATC1_COD,
ATC2_COD,
ATC3_COD,
ATC4_COD,
APP1_COD,
APP2_COD,
APP3_COD,
BIO_DESC,
GENE_ORIG_DESC,
ETH_OTC_DESC,
NRDL_DESC,
NRDL_Entry_Date,
EDL_DESC,
TCM_DESC,
PAED_DESC,
GQCE_DESC,
VBP_DESC,
MANU_COD,
MANU_DES,
MNFL_COD,
MNFL_DES,
CORP_COD,
CORP_DES,
BrandType,
bu,
Market_Ratio,
COALESCE(Key_Competitor, 'Others') AS Key_Competitor,
from_utc_timestamp(current_timestamp(), 'UTC+8') AS ETL_INSERT_DT,
from_utc_timestamp(current_timestamp(), 'UTC+8') AS ETL_UPDATE_DT
FROM market_kc_kept
;