List Datasets
curl --request GET \
--url https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets \
--header 'Authorization: <authorization>'import requests
url = "https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets"
headers = {"Authorization": "<authorization>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: '<authorization>'}};
fetch('https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: <authorization>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "<authorization>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets")
.header("Authorization", "<authorization>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = '<authorization>'
response = http.request(request)
puts response.read_body{
"items": [
{
"dataSetId": "<string>",
"dataSetName": "<string>",
"dataSetDescription": "<string>",
"businessAreaId": "<string>",
"businessAreaName": "<string>",
"schemaId": "<string>",
"schemaName": "<string>",
"recordCount": 123,
"erroredRecordsCount": 123,
"runningJobsCount": 123,
"version": 123,
"createdBy": "<string>",
"createdByEmail": "<string>",
"createdByName": "<string>",
"updatedBy": "<string>",
"updatedByEmail": "<string>",
"updatedByName": "<string>",
"createdDate": "<string>",
"updatedDate": "<string>",
"deleted": true
}
],
"nextPageKey": "<string>"
}Datasets
List Datasets
Retrieve the list of datasets within a specific schema
GET
/
v1
/
businessareas
/
{businessAreaId}
/
schemas
/
{schemaId}
/
datasets
List Datasets
curl --request GET \
--url https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets \
--header 'Authorization: <authorization>'import requests
url = "https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets"
headers = {"Authorization": "<authorization>"}
response = requests.get(url, headers=headers)
print(response.text)const options = {method: 'GET', headers: {Authorization: '<authorization>'}};
fetch('https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "GET",
CURLOPT_HTTPHEADER => [
"Authorization: <authorization>"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"net/http"
"io"
)
func main() {
url := "https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets"
req, _ := http.NewRequest("GET", url, nil)
req.Header.Add("Authorization", "<authorization>")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.get("https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets")
.header("Authorization", "<authorization>")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Get.new(url)
request["Authorization"] = '<authorization>'
response = http.request(request)
puts response.read_body{
"items": [
{
"dataSetId": "<string>",
"dataSetName": "<string>",
"dataSetDescription": "<string>",
"businessAreaId": "<string>",
"businessAreaName": "<string>",
"schemaId": "<string>",
"schemaName": "<string>",
"recordCount": 123,
"erroredRecordsCount": 123,
"runningJobsCount": 123,
"version": 123,
"createdBy": "<string>",
"createdByEmail": "<string>",
"createdByName": "<string>",
"updatedBy": "<string>",
"updatedByEmail": "<string>",
"updatedByName": "<string>",
"createdDate": "<string>",
"updatedDate": "<string>",
"deleted": true
}
],
"nextPageKey": "<string>"
}The List Datasets endpoint returns all datasets defined within a specific schema. Datasets are collections of data objects that share the same schema structure and represent actual data records in your master data repository.
Prerequisites
- A Pretectum API key (see API Keys)
- Permission to access datasets in your tenant
- Valid business area ID (see List Business Areas)
- Valid schema ID (see List Schemas)
Authentication
Include your API key in theAuthorization header.
Authorization: pre_your_api_key
Request
Path Parameters
string
required
The unique identifier of the business area. You can obtain this from the List Business Areas endpoint.
string
required
The unique identifier of the schema containing the datasets. You can obtain this from the List Schemas endpoint.
Query Parameters
string
A pagination token for retrieving the next page of results. This value is returned in the response as
nextPageKey when more results are available.Headers
string
required
Your Pretectum API key. Create one in the Pretectum app under Configuration → API Keys.
string
default:"application/json"
The response content type. Currently only
application/json is supported.Example Requests
# List all datasets in a schema
curl -X GET "https://api.pretectum.io/v1/businessareas/20240115103000123a1b2c3d4e5f6789012345678901234/schemas/20240115103000456d1e2f3a4b5c6789012345678901234/datasets" \
-H "Authorization: pre_your_api_key" \
-H "Accept: application/json"
# Paginate through datasets
curl -X GET "https://api.pretectum.io/v1/businessareas/20240115103000123a1b2c3d4e5f6789012345678901234/schemas/20240115103000456d1e2f3a4b5c6789012345678901234/datasets?pageKey=eyJMYXN0RXZhbHVhdGVkS2V5Ijp7Li4ufQ" \
-H "Authorization: pre_your_api_key" \
-H "Accept: application/json"
const apiKey = 'pre_your_api_key';
const businessAreaId = '20240115103000123a1b2c3d4e5f6789012345678901234';
const schemaId = '20240115103000456d1e2f3a4b5c6789012345678901234';
async function getDatasets(businessAreaId, schemaId, pageKey = null) {
const url = new URL(
`https://api.pretectum.io/v1/businessareas/${businessAreaId}/schemas/${schemaId}/datasets`
);
if (pageKey) {
url.searchParams.set('pageKey', pageKey);
}
const response = await fetch(url, {
headers: {
'Authorization': apiKey,
'Accept': 'application/json'
}
});
return response.json();
}
const datasets = await getDatasets(businessAreaId, schemaId);
console.log(`Found ${datasets.items.length} datasets`);
datasets.items.forEach(dataset => {
console.log(`- ${dataset.dataSetName}: ${dataset.recordCount} records`);
});
import requests
api_key = 'pre_your_api_key'
business_area_id = '20240115103000123a1b2c3d4e5f6789012345678901234'
schema_id = '20240115103000456d1e2f3a4b5c6789012345678901234'
def get_datasets(business_area_id, schema_id, page_key=None):
params = {}
if page_key:
params['pageKey'] = page_key
response = requests.get(
f'https://api.pretectum.io/v1/businessareas/{business_area_id}/schemas/{schema_id}/datasets',
params=params,
headers={
'Authorization': api_key,
'Accept': 'application/json'
}
)
response.raise_for_status()
return response.json()
datasets = get_datasets(business_area_id, schema_id)
print(f"Found {len(datasets['items'])} datasets")
for dataset in datasets['items']:
print(f"- {dataset['dataSetName']}: {dataset['recordCount']} records")
Response
A successful request returns an object containing an array of datasets and pagination information.array
required
An array of dataset objects within the schema.
Show Dataset object properties
Show Dataset object properties
string
The unique identifier for the dataset. Use this ID when filtering data object searches.
string
The display name of the dataset. This is the human-readable name you can use in the
dataSet filter parameter when searching data objects.string
A description of the dataset explaining its purpose and the data it contains.
string
The ID of the business area this dataset belongs to.
string
The name of the business area this dataset belongs to.
string
The ID of the schema this dataset uses.
string
The name of the schema this dataset uses.
integer
The total number of data objects (records) in this dataset.
integer
The number of records that have validation errors.
integer
The number of background jobs currently running on this dataset (e.g., imports, exports).
integer
The version number of the dataset. This increments each time the dataset is modified.
string
The identifier of the user who created the dataset.
string
The email address of the user who created the dataset.
string
The full name of the user who created the dataset.
string
The identifier of the user who last modified the dataset.
string
The email address of the user who last modified the dataset.
string
The full name of the user who last modified the dataset.
string
The ISO 8601 timestamp when the dataset was created.
string
The ISO 8601 timestamp when the dataset was last modified.
boolean
Indicates whether the dataset has been marked as deleted.
string
A pagination token for retrieving the next page of results. If this field is present, more datasets are available. Pass this value as the
pageKey query parameter in your next request.Example Response
{
"items": [
{
"dataSetId": "20240925152201042a1b2c3d4e5f6789012345678901234",
"dataSetName": "US Customers",
"dataSetDescription": "Customer records for United States region",
"businessAreaId": "20240115103000123a1b2c3d4e5f6789012345678901234",
"businessAreaName": "Customer",
"schemaId": "20240115103000456d1e2f3a4b5c6789012345678901234",
"schemaName": "Individual Customer",
"recordCount": 15420,
"erroredRecordsCount": 12,
"runningJobsCount": 0,
"version": 5,
"createdBy": "9ae5f422-bb62-4c9d-b277-594ddcda6d8d",
"createdByEmail": "admin@example.com",
"createdByName": "John Admin",
"updatedBy": "9ae5f422-bb62-4c9d-b277-594ddcda6d8d",
"updatedByEmail": "admin@example.com",
"updatedByName": "John Admin",
"createdDate": "2024-09-25T15:22:01.042Z",
"updatedDate": "2024-12-15T10:30:00.000Z",
"deleted": false
},
{
"dataSetId": "20240926090000123b2c3d4e5f6a7890123456789012345",
"dataSetName": "European Customers",
"dataSetDescription": "Customer records for European region",
"businessAreaId": "20240115103000123a1b2c3d4e5f6789012345678901234",
"businessAreaName": "Customer",
"schemaId": "20240115103000456d1e2f3a4b5c6789012345678901234",
"schemaName": "Individual Customer",
"recordCount": 8750,
"erroredRecordsCount": 3,
"runningJobsCount": 0,
"version": 2,
"createdBy": "b5f6g733-cc73-5d0e-c388-605eeda7e9e",
"createdByEmail": "data_manager@example.com",
"createdByName": "Jane Manager",
"updatedBy": "b5f6g733-cc73-5d0e-c388-605eeda7e9e",
"updatedByEmail": "data_manager@example.com",
"updatedByName": "Jane Manager",
"createdDate": "2024-09-26T09:00:00.123Z",
"updatedDate": "2024-11-20T14:45:00.000Z",
"deleted": false
}
],
"nextPageKey": "eyJMYXN0RXZhbHVhdGVkS2V5Ijp7ImRhdGFTZXRJZCI6IjIwMjQwOTI2MDkwMDAw..."
}
Response Without Pagination
When all datasets fit in a single response, nonextPageKey is returned:
{
"items": [
{
"dataSetId": "20240925152201042a1b2c3d4e5f6789012345678901234",
"dataSetName": "US Customers",
"dataSetDescription": "Customer records for United States region",
"businessAreaId": "20240115103000123a1b2c3d4e5f6789012345678901234",
"businessAreaName": "Customer",
"schemaId": "20240115103000456d1e2f3a4b5c6789012345678901234",
"schemaName": "Individual Customer",
"recordCount": 15420,
"erroredRecordsCount": 0,
"runningJobsCount": 0,
"version": 5,
"createdBy": "9ae5f422-bb62-4c9d-b277-594ddcda6d8d",
"createdByEmail": "admin@example.com",
"createdByName": "John Admin",
"updatedBy": "9ae5f422-bb62-4c9d-b277-594ddcda6d8d",
"updatedByEmail": "admin@example.com",
"updatedByName": "John Admin",
"createdDate": "2024-09-25T15:22:01.042Z",
"updatedDate": "2024-12-15T10:30:00.000Z",
"deleted": false
}
]
}
Empty Response
If the schema has no datasets defined, the response will contain an empty items array:{
"items": []
}
Error Responses
| Status Code | Description |
|---|---|
401 Unauthorized | The API key is missing, malformed, unknown, inactive, expired or deleted. Check the key in Configuration → API Keys. |
403 Forbidden | Your application client does not have permission to access datasets. Contact your tenant administrator. |
404 Not Found | The specified business area or schema does not exist, or you do not have access to it. |
500 Internal Server Error | An unexpected error occurred on the server. Try again later or contact support. |
Pagination
When a schema contains many datasets, results are paginated. Use thenextPageKey from the response to fetch subsequent pages:
async function getAllDatasets(businessAreaId, schemaId) {
const allDatasets = [];
let pageKey = null;
do {
const response = await getDatasets(businessAreaId, schemaId, pageKey);
allDatasets.push(...response.items);
pageKey = response.nextPageKey;
} while (pageKey);
return allDatasets;
}
const allDatasets = await getAllDatasets(businessAreaId, schemaId);
console.log(`Total datasets: ${allDatasets.length}`);
def get_all_datasets(business_area_id, schema_id):
all_datasets = []
page_key = None
while True:
response = get_datasets(business_area_id, schema_id, page_key)
all_datasets.extend(response['items'])
page_key = response.get('nextPageKey')
if not page_key:
break
return all_datasets
all_datasets = get_all_datasets(business_area_id, schema_id)
print(f"Total datasets: {len(all_datasets)}")
Use Cases
Filtering Search Results by Dataset
Use the dataset names returned by this endpoint to filter your data object searches:# First, get the list of datasets
curl -X GET "https://api.pretectum.io/v1/businessareas/{businessAreaId}/schemas/{schemaId}/datasets" \
-H "Authorization: pre_your_api_key"
# Then search within a specific dataset
curl -X GET "https://api.pretectum.io/v1/dataobjects/search?query=John&dataSet=US%20Customers" \
-H "Authorization: pre_your_api_key"
Monitoring Data Quality
Check theerroredRecordsCount to identify datasets with data quality issues:
const datasets = await getDatasets(businessAreaId, schemaId);
const datasetsWithErrors = datasets.items.filter(ds => ds.erroredRecordsCount > 0);
datasetsWithErrors.forEach(ds => {
console.log(`${ds.dataSetName}: ${ds.erroredRecordsCount} errors out of ${ds.recordCount} records`);
});
Tracking Record Counts
Monitor the size of your datasets:datasets = get_datasets(business_area_id, schema_id)
total_records = sum(ds['recordCount'] for ds in datasets['items'])
print(f"Total records across all datasets: {total_records}")
for ds in sorted(datasets['items'], key=lambda x: x['recordCount'], reverse=True):
print(f" {ds['dataSetName']}: {ds['recordCount']:,} records")
Best Practices
- Cache dataset lists: Dataset metadata changes less frequently than record data. Cache the response and refresh periodically.
- Filter by active datasets: Exclude datasets where
deleted: truein user-facing interfaces. - Use names for search filters: When filtering searches with the
dataSetparameter, use thedataSetNamefield value. - Handle pagination: Always check for
nextPageKeyin responses and fetch all pages if needed. - Monitor error counts: Regularly check
erroredRecordsCountto identify data quality issues early.
Related Endpoints
List Schemas
Get schema IDs for dataset queries
List Business Areas
Get business area IDs
Search Data Objects
Search within specific datasets
API Keys
Obtain authentication token
