import crypto from 'node:crypto'; import fs from 'node:fs'; import path from 'node:path'; import { TextDecoder } from 'node:util'; const LOCK_FILE = 'qkforce.test-data.lock.json'; const LOCK_SCHEMA_VERSION = 1; const MAX_DATASETS = 128; const CONTENT_TYPE = 'text/csv; charset=utf-8'; const DATA_PATH_PATTERN = /^test-data\/[A-Za-z0-9][A-Za-z0-9._-]{0,239}\.csv$/; const ID_PATTERN = /^[A-Za-z0-9_-]{1,128}$/; const SHA256_PATTERN = /^[a-f0-9]{64}$/; const BLOCKED_HEADERS = new Set(['__proto__', 'prototype', 'constructor']); const DATASET_KEYS = ['datasetId','versionId','versionNumber','logicalName','path','sha256','rowCount','contentType']; const fail = message => { throw new Error(`Invalid QKForce test data: ${message}`); }; const isObject = value => value !== null && typeof value === 'object' && !Array.isArray(value); const rejectUnknownKeys = (value, allowed, location) => { for (const key of Object.keys(value)) if (!allowed.includes(key)) fail(`${location}.${key} is not supported`); }; const sha256 = value => crypto.createHash('sha256').update(value).digest('hex'); const datasetLabel = metadata => `${metadata.logicalName} (datasetId=${metadata.datasetId}, versionId=${metadata.versionId})`; export function validateQkforceTestDataLock(candidate) { if (!isObject(candidate)) fail('the lock document must be a JSON object'); rejectUnknownKeys(candidate, ['schemaVersion','datasets'], 'lock'); if (candidate.schemaVersion !== LOCK_SCHEMA_VERSION) fail(`schemaVersion must be ${LOCK_SCHEMA_VERSION}`); if (!Array.isArray(candidate.datasets)) fail('datasets must be an array'); if (candidate.datasets.length > MAX_DATASETS) fail(`datasets must contain at most ${MAX_DATASETS} entries`); const seenDatasetIds = new Set(), seenVersionIds = new Set(), seenLogicalNames = new Set(), seenPaths = new Set(); const datasets = candidate.datasets.map((entry, index) => { const location = `datasets[${index}]`; if (!isObject(entry)) fail(`${location} must be an object`); rejectUnknownKeys(entry, DATASET_KEYS, location); for (const key of DATASET_KEYS) if (!(key in entry)) fail(`${location}.${key} is required`); if (typeof entry.datasetId !== 'string' || !ID_PATTERN.test(entry.datasetId)) fail(`${location}.datasetId is invalid`); if (typeof entry.versionId !== 'string' || !ID_PATTERN.test(entry.versionId)) fail(`${location}.versionId is invalid`); if (!Number.isInteger(entry.versionNumber) || entry.versionNumber < 1) fail(`${location}.versionNumber must be an integer greater than zero`); if (typeof entry.logicalName !== 'string' || entry.logicalName.length < 1 || entry.logicalName.length > 128) fail(`${location}.logicalName is invalid`); if (typeof entry.path !== 'string' || !DATA_PATH_PATTERN.test(entry.path)) fail(`${location}.path must be a safe package-local CSV path below test-data/`); if (typeof entry.sha256 !== 'string' || !SHA256_PATTERN.test(entry.sha256)) fail(`${location}.sha256 must be a lowercase SHA-256 digest`); if (!Number.isInteger(entry.rowCount) || entry.rowCount < 0) fail(`${location}.rowCount must be a non-negative integer`); if (entry.contentType !== CONTENT_TYPE) fail(`${location}.contentType must be ${CONTENT_TYPE}`); if (seenDatasetIds.has(entry.datasetId)) fail(`${location}.datasetId is duplicated`); if (seenVersionIds.has(entry.versionId)) fail(`${location}.versionId is duplicated`); if (seenLogicalNames.has(entry.logicalName)) fail(`${location}.logicalName is duplicated`); if (seenPaths.has(entry.path)) fail(`${location}.path is duplicated`); seenDatasetIds.add(entry.datasetId); seenVersionIds.add(entry.versionId); seenLogicalNames.add(entry.logicalName); seenPaths.add(entry.path); return Object.freeze({ ...entry }); }); return Object.freeze({ schemaVersion: LOCK_SCHEMA_VERSION, datasets: Object.freeze(datasets) }); } export function parseRfc4180Csv(text, identity = 'dataset') { if (typeof text !== 'string') fail(`${identity}: CSV content must be UTF-8 text`); if (text.charCodeAt(0) === 0xFEFF) text = text.slice(1); const parsed = []; let row = []; let field = ''; let inQuotes = false; let afterQuote = false; let atFieldStart = true; for (let index = 0; index < text.length; index += 1) { const char = text[index]; if (inQuotes) { if (char === '"') { if (text[index + 1] === '"') { field += '"'; index += 1; } else { inQuotes = false; afterQuote = true; } } else if (char === '\r' && text[index + 1] === '\n') { field += '\r\n'; index += 1; } else field += char; continue; } if (afterQuote) { if (char === ',') { row.push(field); field = ''; afterQuote = false; atFieldStart = true; continue; } if (char === '\n' || char === '\r') { if (char === '\r' && text[index + 1] === '\n') index += 1; row.push(field); parsed.push(row); row = []; field = ''; afterQuote = false; atFieldStart = true; continue; } fail(`${identity}: malformed CSV near row ${parsed.length + 1}; characters after a closing quote are not allowed`); } if (char === '"') { if (!atFieldStart || field.length > 0) fail(`${identity}: malformed CSV near row ${parsed.length + 1}; quote inside an unquoted field`); inQuotes = true; atFieldStart = false; continue; } if (char === ',') { row.push(field); field = ''; atFieldStart = true; continue; } if (char === '\n' || char === '\r') { if (char === '\r' && text[index + 1] === '\n') index += 1; row.push(field); parsed.push(row); row = []; field = ''; atFieldStart = true; continue; } field += char; atFieldStart = false; } if (inQuotes) fail(`${identity}: malformed CSV; quoted field is not terminated`); if (row.length > 0 || field.length > 0 || afterQuote || !atFieldStart) { row.push(field); parsed.push(row); } if (parsed.length === 0) fail(`${identity}: CSV must contain a header row`); const headers = parsed[0]; const seenHeaders = new Set(); headers.forEach((header, index) => { if (header.length === 0 || header.trim().length === 0 || header !== header.trim()) fail(`${identity}: header ${index + 1} is empty or malformed`); if (/[\u0000-\u001F\u007F]/u.test(header)) fail(`${identity}: header ${index + 1} contains a control character`); if (BLOCKED_HEADERS.has(header)) fail(`${identity}: header ${index + 1} is reserved`); if (seenHeaders.has(header)) fail(`${identity}: duplicate header "${header}"`); seenHeaders.add(header); }); const rows = parsed.slice(1).map((values, index) => { if (values.length !== headers.length) fail(`${identity}: row ${index + 2} has ${values.length} columns; expected ${headers.length}`); const record = {}; headers.forEach((header, column) => { record[header] = values[column]; }); return Object.freeze(record); }); return Object.freeze({ headers: Object.freeze([...headers]), rows: Object.freeze(rows) }); } const TYPE_NAMES = new Set(['string','integer','number','boolean','date']); const validateIsoDate = value => { if (!/^\d{4}-\d{2}-\d{2}$/.test(value)) return false; const [y,m,d] = value.split('-').map(Number); const date = new Date(Date.UTC(y,m-1,d)); return date.getUTCFullYear()===y && date.getUTCMonth()===m-1 && date.getUTCDate()===d; }; const normalizeRule = (rule, column, identity) => { if (typeof rule === 'string') { if (!TYPE_NAMES.has(rule)) fail(`${identity}: schema for column "${column}" uses unsupported type ${rule}`); return { type: rule, nullable: false }; } if (!isObject(rule)) fail(`${identity}: schema for column "${column}" must be a type or descriptor`); rejectUnknownKeys(rule, ['type','nullable'], `schema.${column}`); if (!TYPE_NAMES.has(rule.type)) fail(`${identity}: schema for column "${column}" uses an unsupported type`); if ('nullable' in rule && typeof rule.nullable !== 'boolean') fail(`${identity}: schema nullable flag for column "${column}" must be boolean`); return { type: rule.type, nullable: rule.nullable === true }; }; const convertCell = (value, rule, column, identity, rowNumber) => { if (value === '' && rule.nullable) return null; const location = `${identity} row ${rowNumber} column "${column}"`; switch (rule.type) { case 'string': return value; case 'integer': { if (!/^[+-]?\d+$/.test(value)) fail(`${location} is not a valid integer`); const converted = Number(value); if (!Number.isSafeInteger(converted)) fail(`${location} is outside the safe integer range`); return converted; } case 'number': { if (!/^[+-]?(?:\d+(?:\.\d*)?|\.\d+)(?:[eE][+-]?\d+)?$/.test(value)) fail(`${location} is not a valid number`); const converted = Number(value); if (!Number.isFinite(converted)) fail(`${location} is not a finite number`); return converted; } case 'boolean': if (value === 'true') return true; if (value === 'false') return false; fail(`${location} is not a valid boolean`); break; case 'date': if (!validateIsoDate(value)) fail(`${location} is not a valid ISO date`); return value; default: fail(`${location} uses an unsupported type`); } }; class DatasetView { #metadata; #headers; #rows; constructor(metadata, parsed) { this.#metadata = metadata; this.#headers = parsed.headers; this.#rows = parsed.rows; Object.freeze(this); } get metadata() { return this.#metadata; } get headers() { return this.#headers; } get rowCount() { return this.#rows.length; } rows() { return this.#rows; } row(index) { if (!Number.isInteger(index) || index < 0 || index >= this.#rows.length) fail(`${datasetLabel(this.#metadata)}: row index ${index} is out of range`); return this.#rows[index]; } where(criteria) { if (!isObject(criteria) || Object.keys(criteria).length === 0) fail(`${datasetLabel(this.#metadata)}: filter criteria must be a non-empty object`); for (const [column,value] of Object.entries(criteria)) { if (!this.#headers.includes(column)) fail(`${datasetLabel(this.#metadata)}: filter column "${column}" is not declared`); if (typeof value !== 'string') fail(`${datasetLabel(this.#metadata)}: filter values must be strings`); } return Object.freeze(this.#rows.filter(row => Object.entries(criteria).every(([column,value]) => row[column] === value))); } first(criteria) { const matches = this.where(criteria); if (matches.length === 0) fail(`${datasetLabel(this.#metadata)}: filter matched no rows`); return matches[0]; } typedRow(index, schema) { const row = this.row(index); if (!isObject(schema) || Object.keys(schema).length === 0) fail(`${datasetLabel(this.#metadata)}: typed schema must be a non-empty object`); const typed = { ...row }; for (const [column,rawRule] of Object.entries(schema)) { if (!this.#headers.includes(column)) fail(`${datasetLabel(this.#metadata)}: schema column "${column}" is not declared`); typed[column] = convertCell(row[column], normalizeRule(rawRule,column,datasetLabel(this.#metadata)), column, datasetLabel(this.#metadata), index + 1); } return Object.freeze(typed); } } class QkforceTestDataProvider { #datasets; #byLogicalName; #byDatasetId; #lockSha256; constructor(datasets, lockSha256) { this.#datasets = Object.freeze(datasets); this.#byLogicalName = new Map(datasets.map(d => [d.metadata.logicalName,d])); this.#byDatasetId = new Map(datasets.map(d => [d.metadata.datasetId,d])); this.#lockSha256 = lockSha256; Object.freeze(this); } get size() { return this.#datasets.length; } datasets() { return Object.freeze(this.#datasets.map(d => d.metadata)); } has(nameOrId) { return this.#byLogicalName.has(nameOrId) || this.#byDatasetId.has(nameOrId); } dataset(nameOrId) { const dataset = this.#byLogicalName.get(nameOrId) ?? this.#byDatasetId.get(nameOrId); if (!dataset) fail(`dataset "${nameOrId}" is not declared by ${LOCK_FILE}`); return dataset; } provenance() { if (this.#lockSha256 === undefined) return Object.freeze({}); return Object.freeze({ testDataLockSha256: this.#lockSha256, testData: Object.freeze(this.#datasets.map(d => Object.freeze({ datasetId:d.metadata.datasetId, versionId:d.metadata.versionId, sha256:d.metadata.sha256 }))) }); } } function readLock(rootDirectory, required) { const lockPath = path.resolve(rootDirectory, LOCK_FILE); let bytes; try { bytes = fs.readFileSync(lockPath); } catch (error) { if (error && error.code === 'ENOENT') { if (required) fail(`${LOCK_FILE} is missing`); return undefined; } fail(`${LOCK_FILE} cannot be read`); } let parsed; try { parsed = JSON.parse(new TextDecoder('utf-8',{fatal:true}).decode(bytes)); } catch { fail(`${LOCK_FILE} is not valid UTF-8 JSON`); } return { parsed, digest: sha256(bytes) }; } function loadDataset(rootDirectory, metadata) { const dataRoot = path.resolve(rootDirectory,'test-data'); const absolutePath = path.resolve(rootDirectory,metadata.path); if (!absolutePath.startsWith(`${dataRoot}${path.sep}`)) fail(`${datasetLabel(metadata)}: path escapes test-data/`); let stat; try { stat = fs.lstatSync(absolutePath); } catch (error) { if (error && error.code === 'ENOENT') fail(`${datasetLabel(metadata)}: declared CSV file is missing`); fail(`${datasetLabel(metadata)}: declared CSV file cannot be inspected`); } if (stat.isSymbolicLink()) fail(`${datasetLabel(metadata)}: symbolic links are not allowed for CSV inputs`); if (!stat.isFile()) fail(`${datasetLabel(metadata)}: declared CSV path is not a regular file`); let bytes; try { bytes = fs.readFileSync(absolutePath); } catch { fail(`${datasetLabel(metadata)}: declared CSV file cannot be read`); } if (sha256(bytes) !== metadata.sha256) fail(`${datasetLabel(metadata)}: checksum mismatch`); let text; try { text = new TextDecoder('utf-8',{fatal:true}).decode(bytes); } catch { fail(`${datasetLabel(metadata)}: CSV file is not valid UTF-8`); } const parsed = parseRfc4180Csv(text,datasetLabel(metadata)); if (parsed.rows.length !== metadata.rowCount) fail(`${datasetLabel(metadata)}: row count mismatch; lock declares ${metadata.rowCount} rows but CSV contains ${parsed.rows.length}`); return new DatasetView(metadata,parsed); } export function loadQkforceTestData(rootDirectory = process.cwd(), options = {}) { if (!isObject(options)) fail('loader options must be an object'); rejectUnknownKeys(options,['required'],'options'); if ('required' in options && typeof options.required !== 'boolean') fail('options.required must be boolean'); const lock = readLock(rootDirectory, options.required === true); if (!lock) return new QkforceTestDataProvider([],undefined); const validated = validateQkforceTestDataLock(lock.parsed); return new QkforceTestDataProvider(validated.datasets.map(metadata => loadDataset(rootDirectory,metadata)),lock.digest); } export { CONTENT_TYPE, DATA_PATH_PATTERN, LOCK_FILE, LOCK_SCHEMA_VERSION, MAX_DATASETS };