Configure RT 2.0 ID stitching - merge real-time profiles across different identifiers using stitching keys and primary keys
60
70%
Does it follow best practices?
Run evals on this skill
Adds up to 20 points to the overall score
View guide
Passed
No findings from the security scan
Fix and improve this skill with Tessl
tessl review fix ./realtime-skills/rt-config-id-stitching/SKILL.mdConfigure profile merging across multiple identifiers in real-time.
id_stitching:
primary_key: "td_client_id" # Main unique identifier
stitching_keys: # All identifiers for merging
- name: "td_client_id"
exclude_regex: "^test_.*" # Filter invalid IDs
- name: "email"
exclude_regex: ".*@test\\.com$"
- name: "user_id"
ext_lookup_key: "email" # Optional: External lookupThe main unique identifier for profiles:
primary_key: "td_client_id"Requirements:
stitching_keysCommon choices:
td_client_id - TD cookie/device ID (recommended for web)user_id - Your application's user IDcustomer_id - CRM customer IDAll identifiers used to merge profiles:
stitching_keys:
- name: "td_client_id"
exclude_regex: "^test_.*" # Exclude test IDs
- name: "email"
exclude_regex: ".*@(test|example)\\.com$" # Exclude test emails
- name: "user_id"
exclude_regex: "^guest_" # Exclude guest users
- name: "phone_number"
# No exclusion patternKey Selection:
Filter out invalid or test identifiers:
# Test IDs
exclude_regex: "^test_.*" # Starts with "test_"
exclude_regex: ".*_test$" # Ends with "_test"
# Test emails
exclude_regex: ".*@test\\.com$" # @test.com domain
exclude_regex: ".*@example\\.com$" # @example.com domain
exclude_regex: ".*@(test|example|demo)\\.com$" # Multiple test domains
# Guest/anonymous users
exclude_regex: "^guest_" # Starts with "guest_"
exclude_regex: "^anon_" # Starts with "anon_"
exclude_regex: "^(guest|anonymous)_" # Starts with guest_ or anonymous_
# Development IDs
exclude_regex: "^dev_" # Development IDs
exclude_regex: "localhost" # Localhost IDs
# Invalid formats
exclude_regex: "^$" # Empty strings
exclude_regex: "^null$" # String "null"
exclude_regex: "^undefined$" # String "undefined"Optional key for external profile lookups:
ext_lookup_key: "email"Use cases:
Requirements:
stitching_keysemail or user_idid_stitching:
primary_key: "td_client_id"
stitching_keys:
- name: "td_client_id"
exclude_regex: "^test_"
- name: "email"
exclude_regex: ".*@(test|example)\\.com$"
- name: "customer_id"
exclude_regex: "^guest_"
- name: "user_id"
ext_lookup_key: "email"id_stitching:
primary_key: "user_id"
stitching_keys:
- name: "user_id"
exclude_regex: "^test_"
- name: "email"
exclude_regex: ".*@(test|example|localhost)\\.com$"
- name: "td_client_id"
- name: "account_id"
ext_lookup_key: "email"id_stitching:
primary_key: "td_client_id"
stitching_keys:
- name: "td_client_id" # Device ID
- name: "user_id" # App user ID
exclude_regex: "^guest_"
- name: "advertising_id" # IDFA/GAID
- name: "email"
exclude_regex: ".*@test\\.com$"
ext_lookup_key: "user_id"When an event comes in with multiple IDs:
{
"td_client_id": "abc123",
"email": "user@example.com",
"user_id": "user_456"
}Check what ID fields are in your event tables:
# View table schema
tdx table describe <database> <table>
# Check ID field distribution
tdx query "
select
count(distinct td_client_id) as client_ids,
count(distinct email) as emails,
count(distinct user_id) as user_ids,
count(*) as total_events
from <database>.<table>
where td_interval(time, '-7d')
"
# Check ID co-occurrence
tdx query "
select
count(*) as events,
count(case when td_client_id is not null then 1 end) as has_client_id,
count(case when email is not null then 1 end) as has_email,
count(case when user_id is not null then 1 end) as has_user_id
from <database>.<table>
where td_interval(time, '-7d')
"
# Find test patterns to exclude
tdx query "
select
email,
count(*) as event_count
from <database>.<table>
where td_interval(time, '-7d')
and email like '%test%'
group by email
order by event_count desc
limit 20
"# Update key columns
tdx api "/audiences/<parent_segment_id>/realtime_setting" --type cdp --method PATCH --data '{
"key_columns": {
"primary_key": "td_client_id",
"stitching_keys": [
{
"name": "td_client_id",
"exclude_regex": "^test_.*"
},
{
"name": "email",
"exclude_regex": ".*@test\\.com$"
},
{
"name": "user_id"
}
],
"ext_lookup_key": "email"
}
}'# Validate configuration
tdx ps rt validate rt_config.yaml
# Checks:
# - primary_key is in stitching_keys
# - All key names are valid
# - Regex patterns are validCheck profile merge activity using the id_changes table in the parent segment's real-time database (cdp_audience_<parent_segment_id>_rt):
# Count identity events by type
tdx query "
SELECT
profile_change_type,
COUNT(*) AS event_count
FROM cdp_audience_<parent_segment_id>_rt.id_changes
WHERE TD_INTERVAL(time, '-1d/now')
GROUP BY profile_change_type
ORDER BY event_count DESC
"
# Check recent merges (profile_updated_by_stitching with non-empty merged_rids)
tdx query "
SELECT
TD_TIME_FORMAT(time, 'yyyy-MM-dd HH:mm:ss', 'GMT') AS ts,
current_rid,
merged_rids,
current_id_attributes,
key_values,
td_rt_tracking_id
FROM cdp_audience_<parent_segment_id>_rt.id_changes
WHERE TD_INTERVAL(time, '-1h/now')
AND profile_change_type = 'profile_updated_by_stitching'
AND cardinality(CAST(json_parse(merged_rids) AS ARRAY(VARCHAR))) > 0
ORDER BY time DESC
LIMIT 20
"
# Check for 200-association limit evictions
tdx query "
SELECT
TD_TIME_FORMAT(time, 'yyyy-MM-dd HH:mm:ss', 'GMT') AS ts,
current_rid,
evicted_ids,
current_id_attributes,
td_rt_tracking_id
FROM cdp_audience_<parent_segment_id>_rt.id_changes
WHERE TD_INTERVAL(time, '-1d/now')
AND profile_change_type = 'profile_ids_evicted'
ORDER BY time DESC
LIMIT 20
"When you suspect a stitching-key misconfiguration (wrong key name, wrong type, bad values), check the validation_failures table. It logs every key/value pair that failed validation and was excluded from stitching, along with the reason.
# Check for validation failures
tdx query "
SELECT
TD_TIME_FORMAT(time, 'yyyy-MM-dd HH:mm:ss', 'GMT') AS ts,
json_extract_scalar(invalid_key, '$.key') AS key_name,
json_extract_scalar(invalid_key, '$.value') AS key_value,
json_extract_scalar(invalid_key, '$.reason') AS reason,
td_rt_tracking_id
FROM cdp_audience_<parent_segment_id>_rt.validation_failures
CROSS JOIN UNNEST(CAST(invalid_keys AS ARRAY(JSON))) AS t(invalid_key)
WHERE TD_INTERVAL(time, '-1d/now')
ORDER BY time DESC
LIMIT 50
"
# Count failures by reason to identify the most common misconfiguration
tdx query "
SELECT
json_extract_scalar(invalid_key, '$.reason') AS reason,
json_extract_scalar(invalid_key, '$.key') AS key_name,
COUNT(*) AS failure_count
FROM cdp_audience_<parent_segment_id>_rt.validation_failures
CROSS JOIN UNNEST(CAST(invalid_keys AS ARRAY(JSON))) AS t(invalid_key)
WHERE TD_INTERVAL(time, '-1d/now')
GROUP BY 1, 2
ORDER BY failure_count DESC
"A non-empty validation_failures table means something needs fixing. Common causes: key name not in the config, key name case mismatch (e.g. User_ID vs user_id), non-string values, empty values, values failing valid_regexp or matching invalid_texts.
| Error | Solution |
|---|---|
| "Primary key not in stitching_keys" | Add primary_key to stitching_keys list |
| "Invalid regex pattern" | Test regex with online validator |
| "Key column not found" | Verify field exists in event tables |
| "Duplicate key names" | Remove duplicate stitching_keys |
| "ext_lookup_key not in stitching_keys" | Add ext_lookup_key to stitching_keys |
# Test exclude regex patterns
tdx query "
select
td_client_id,
email,
user_id,
case when regexp_like(td_client_id, '^test_') then 'excluded' else 'included' end as client_id_status,
case when regexp_like(email, '.*@test\\.com$') then 'excluded' else 'included' end as email_status
from <database>.<table>
where td_interval(time, '-1h')
limit 20
"After configuring ID stitching:
tdx ps pushid_changes table for merge and eviction activityrt-pz-service skillrt-journey-create skill1a0845f
If you maintain this skill, you can claim it as your own. Once claimed, you can manage eval scenarios, bundle related skills, attach documentation or rules, and ensure cross-agent compatibility.