', ['First second.'], [], 'Normalize nonbreaking spaces to ordinary spaces in prose.'],
['
Visible.
Hidden draft
', ['Visible.'], ['Hidden draft'], 'Exclude a hidden attribute and its descendants.'],
['
Visible.
Decorative copy
', ['Visible.'], ['Decorative copy'], 'Explicitly exclude aria-hidden=true for this contract; this is not a CSS visibility inference.'],
['
Visible.
', ['Visible.'], ['Script text', 'Style text'], 'Never include script or style contents as article prose.'],
['
Public paragraph.
', ['Public paragraph.'], ['Internal note'], 'HTML comments are not text output.'],
['
Heading
Section
Explanation.
', ['Heading', 'Section', 'Explanation.'], [], 'Keep heading text as ordered blocks; output does not preserve heading-level metadata.'],
['
First item
Second item
', ['First item', 'Second item'], [], 'Each list item is a block; synthetic bullets are not required.'],
['
Plan
Units
Example
12
', ['Plan | Units', 'Example | 12'], [], 'A table row is one block with cell boundaries represented by a literal vertical bar.'],
['
Quoted sentence.
Comment.
', ['Quoted sentence.', 'Comment.'], [], 'A blockquote does not duplicate its paragraph.'],
['
line 1\n line 2
', ['line 1\n line 2'], [], 'Preserve line breaks and indentation inside preformatted text.'],
['Caption text.', ['Caption text.'], ['Alternative label'], 'Caption is a block; image alt is outside this text-output contract.'],
['Question?
Authored answer.
', ['Question?', 'Authored answer.'], [], 'Static extraction includes details descendants regardless of initial open state; not a visible-pixels policy.'],
['
Алматы: қала. Café.
', ['Алматы: қала. Café.'], [], 'Preserve Unicode letters and punctuation without translation.'],
['', [], [], 'An existing empty region is an empty result, not a successful article extraction.'],
['
Loading article…
', [], ['Loading article…'], 'The authored render-required sentinel requests an unresolved result; it is not a generic loading-text detector.'],
];
return cases.map(([html, blocks, excluded, rationale], i) => row('EXTRACT', i + 1,
{ html: document(html), selector: 'main', policy: 'static-article-text-v1' },
{ status: i === 23 ? 'render_required' : blocks.length ? 'ok' : 'empty', blocks, text: blocks.join('\n'), must_not_include: excluded },
['E-SCOPE', 'E-BLOCKS', 'E-EXCLUSIONS', ...(i === 23 ? ['E-RENDER'] : [])], rationale));
}
const contact = (patch = {}) => ({ source_id: '', name: '', email: '', email_verified: false, phone_test_token: '', company: '', role_mailbox: false, ...patch });
export function crmFixtures() {
const c = contact;
const cases = [
[c({ source_id: 'F001', name: 'Fiction A' }), c({ source_id: 'F001', name: 'Fiction A' }), 'link_candidate', 'Same nonempty ID in the same fixture namespace; no conflicting populated identifier.'],
[c({ source_id: 'F001', email: 'a@example.test' }), c({ source_id: 'F001', email: 'b@example.test' }), 'review', 'Same ID with conflicting populated emails must not silently merge.'],
[c({ source_id: 'F001' }), c({ source_id: 'F002' }), 'keep_separate', 'Different IDs and no corroborating attributes.'],
[c({ email: 'a@example.test', email_verified: true }), c({ email: 'a@example.test', email_verified: true }), 'link_candidate', 'Exact verified non-role email is a candidate signal, not proven person identity.'],
[c({ email: ' a@example.test ', email_verified: true }), c({ email: 'a@example.test', email_verified: true }), 'link_candidate', 'Trim surrounding email whitespace under the explicit policy.'],
[c({ email: 'a@EXAMPLE.TEST', email_verified: true }), c({ email: 'a@example.test', email_verified: true }), 'link_candidate', 'Case-normalize only the domain for this policy.'],
[c({ email: 'A@example.test', email_verified: true }), c({ email: 'a@example.test', email_verified: true }), 'review', 'Local-part case difference is not automatically collapsed.'],
[c({ email: 'a+news@example.test', email_verified: true }), c({ email: 'a@example.test', email_verified: true }), 'review', 'Do not assume plus addressing denotes one mailbox for every domain.'],
[c({ email: 'a.b@example.test', email_verified: true }), c({ email: 'ab@example.test', email_verified: true }), 'review', 'Do not apply a vendor-specific dot rule to arbitrary domains.'],
[c({ email: 'a@example.test' }), c({ email: 'a@example.test' }), 'review', 'Unverified identical email is insufficient for candidate linking here.'],
[c({ email: 'a@example.test', email_verified: true }), c({ email: 'a@example.test' }), 'review', 'Verification is required on both fixture records.'],
[c({ email: 'sales@example.test', email_verified: true, role_mailbox: true }), c({ email: 'sales@example.test', email_verified: true, role_mailbox: true }), 'review', 'A shared role address is not evidence of one person.'],
[c({ name: 'Fiction A', company: 'Example North' }), c({ name: 'Fiction A', company: 'Example North' }), 'review', 'Name plus company is a weak match signal.'],
[c({ name: 'Fiction A', company: 'Example North' }), c({ name: 'Fiction A', company: 'Example South' }), 'review', 'A company difference does not disprove identity or establish a job change.'],
[c({ name: 'Fiction A' }), c({ name: 'Fiction B' }), 'keep_separate', 'Different names without a stronger shared identifier.'],
[c(), c(), 'keep_separate', 'Two empty records are not duplicates merely because every field equals blank.'],
[c({ phone_test_token: '+999000000001' }), c({ phone_test_token: '+999000000001' }), 'review', 'Synthetic phone equality alone can represent a shared or reassigned number.'],
[c({ phone_test_token: '+999000000001', email: 'a@example.test', email_verified: true }), c({ phone_test_token: '+999000000002', email: 'a@example.test', email_verified: true }), 'review', 'Conflicting populated phone identifiers require review even when email matches.'],
[c({ source_id: 'F001', email: 'a@example.test', email_verified: true }), c({ source_id: 'F002', email: 'a@example.test', email_verified: true }), 'review', 'Different IDs in one namespace conflict with the email candidate signal.'],
[c({ name: 'Fiction Café' }), c({ name: 'Fiction Cafe\u0301' }), 'review', 'NFC and NFD code units differ but normalize to the same name; name equivalence is review evidence, not a unique identity.'],
[c({ name: 'Fiction Cafe' }), c({ name: 'Fiction Café' }), 'review', 'An absent versus present accent remains different after NFC normalization; accent stripping is only a weak review signal.'],
[c({ email: 'a@example.test', email_verified: true }), c({ email: 'a@example.invalid', email_verified: true }), 'keep_separate', 'Different domains remain different addresses.'],
[c({ source_id: 'F001', name: 'Fiction A' }), c({ source_id: 'F001', name: 'Fiction A updated' }), 'link_candidate', 'Same ID permits a candidate despite a changed display name; actual merge still needs a survivorship rule.'],
[c({ email: 'fiction-at-example.test', email_verified: true }), c({ email: 'fiction-at-example.test', email_verified: true }), 'review', 'A verification flag does not repair malformed email syntax.'],
];
return cases.map(([left, right, decision, rationale], i) => row('CRM', i + 1,
{ left, right, namespace: `fictional-case-${i + 1}`, policy: 'conservative-link-review-v1' },
{ decision, automatic_merge_allowed: false }, ['C-CANDIDATE', 'C-CONFLICT', 'C-NO-DESTRUCTIVE-MERGE'], rationale));
}
export function changeFixtures() {
const h = text => `
', 'Ready', 'Ready', 'unchanged', 'Wrapper changes do not alter selected text.'],
[h('Ready'), '
Ready
', 'Ready', 'Ready', 'unchanged', 'Class-only changes are excluded from text mode.'],
[h('Ready'), '
Ready
', 'Ready', 'Ready', 'unchanged', 'Comments are excluded.'],
['
Ready
', '
Ready
', 'Ready', 'Ready', 'unchanged', 'Only explicitly marked data-ignore descendants are suppressed.'],
[h('Price: 20'), h('Price: 25'), 'Price: 20', 'Price: 25', 'changed', 'An authored amount change is retained; amounts are not vendor prices.'],
[h('Available'), h('Unavailable'), 'Available', 'Unavailable', 'changed', 'Status text changes.'],
[h('Shipping included'), h('Shipping not included'), 'Shipping included', 'Shipping not included', 'changed', 'A negation must not disappear during normalization.'],
[h('Ready'), h('READY'), 'Ready', 'READY', 'changed', 'Case is significant under this contract.'],
[h('Ready.'), h('Ready!'), 'Ready.', 'Ready!', 'changed', 'Punctuation is significant.'],
[h('Café'), h('Cafe'), 'Café', 'Cafe', 'changed', 'Accents are preserved.'],
[h('First\nSecond'), h('Second\nFirst'), 'First Second', 'Second First', 'changed', 'Order is significant after whitespace normalization.'],
['Reference', 'Reference', 'Reference', 'Reference', 'unchanged', 'Text mode intentionally cannot detect a destination-only change.'],
['
Ready
Old draft', '
Ready
New draft', 'Ready', 'Ready', 'unchanged', 'Hidden descendants are excluded.'],
[h('Ready'), '
Ready
', 'Ready', null, 'unavailable', 'Missing selector is unknown, not unchanged.'],
[h('Ready'), h('Ready') + h('Second region'), 'Ready', null, 'ambiguous', 'Require querySelectorAll(selector).length === 1 before selection; querySelector alone silently returns the first of these two matches.'],
[h('Ready'), 'Loading', 'Ready', null, 'unavailable', 'Explicit fixture sentinel needs a later rendered observation.'],
[h('Ready'), '
Access denied
', 'Ready', null, 'unavailable', 'Simulated HTTP 403 cannot be accepted as a content change.'],
[h('Ready'), '', 'Ready', '', 'changed', 'Successful uniquely selected empty content is distinct from a failed observation.'],
];
return cases.map(([before, after, beforeText, afterText, decision, rationale], i) => row('CHANGE', i + 1,
{ before_html: document(before), after_html: document(after), before_status: 200, after_status: i === 22 ? 403 : 200, selector: '#watch', policy: 'selected-static-text-v1' },
{ decision, before_text: beforeText, after_text: afterText }, ['W-SCOPE', 'W-TEXT', 'W-UNKNOWN'], rationale));
}
const RULES = {
extraction: {
'E-SCOPE': 'Select the single main element. No computed CSS is evaluated and no assets are fetched; inert image src values may exist in the HTML. Static source text is the scope, not a rendered screenshot.',
'E-BLOCKS': 'Ordered blocks: headings, paragraphs, list items, table rows (cells joined by " | "), pre, figcaption, summary. Do not duplicate text through parent blocks. Decode entities; normalize Unicode whitespace except inside pre; preserve accents, case and punctuation. Join blocks by newline.',
'E-EXCLUSIONS': 'Exclude nav, aside, footer, script, style, comments, hidden and aria-hidden=true subtrees. Exclude image alt and link destinations from this output. These are authored choices, not a general readability standard.',
'E-RENDER': 'data-render-required=true is an explicit synthetic sentinel, not an automatic test for whether arbitrary pages need JavaScript.',
},
crm: {
'C-CANDIDATE': 'Each pair is an independent fictional namespace; reused tokens across different cases must not be joined. Within that pair, a shared nonempty ID or exact syntactically valid verified non-role email on both records can propose linking, subject to conflicts. Email normalization trims surrounding space and lowercases domain only.',
'C-CONFLICT': 'Conflicting populated source IDs, emails or phone tokens require review when another identifier proposes a match. Explicit weak-review triggers are identical email with either verification flag false; email local-part case difference; plus-suffix difference; dot-placement difference; identical role mailbox; identical phone token alone; identical name, including a company difference; NFC/NFD-equivalent names; accent-only name difference; and identical malformed email despite verification flags. These signals propose review, not identity. Different names or email domains without another listed signal, different IDs alone, and empty pairs remain keep_separate. keep_separate means no linking evidence under this fixture contract, not proven distinct people.',
'C-NO-DESTRUCTIVE-MERGE': 'All outcomes prohibit automatic destructive merge. Candidate linking still needs field survivorship, suppression, provenance and reversible review. Missing values never constitute shared identifiers. All records are fictional; +999 numbers are intentionally invalid test tokens, not dialable E.164 examples.',
},
change: {
'W-SCOPE': 'Compare one #watch region in each authored static document. Ignore outside content, comments, hidden subtrees and explicit data-ignore descendants. No JavaScript execution or computed CSS.',
'W-TEXT': 'Decode entities and collapse Unicode whitespace. Preserve case, punctuation, accents and text order. Ignore HTML wrappers and attributes, including href: this is a text-change contract, not every possible website change.',
'W-UNKNOWN': 'Both simulated responses must be HTTP 200 with querySelectorAll(selector).length === 1 before selecting and no explicit data-render-required sentinel. querySelector alone must not silently choose the first match. Missing/unready/403 is unavailable; multiple matches ambiguous; neither is unchanged. Successful empty selected text is an actual empty string, not null.',
},
};
const NESTED_DICTIONARIES = {
extraction: {
input: { html: 'Complete authored static HTML document, never fetched from a real site.', selector: 'Single region selected in the static document.', policy: 'Versioned authored extraction contract.' },
expected: { status: 'ok, empty, or render_required; empty and unresolved are distinct.', blocks: 'Ordered array of extracted text blocks.', text: 'Expected blocks joined by newline.', must_not_include: 'Forbidden contamination strings; oracle-only constraints, not fields required in adapter output.' },
},
crm: {
input: { left: 'First fictional record; see contact dictionary.', right: 'Second fictional record; see contact dictionary.', namespace: 'Independent namespace for this pair only; never join IDs between cases.', policy: 'Versioned authored matching contract.' },
contact: { source_id: 'Fictional source identifier, meaningful only in this case namespace.', name: 'Fictional display name, not a unique identifier.', email: 'Authored test-domain address or intentionally malformed value.', email_verified: 'Synthetic input flag, not an actual mailbox verification.', phone_test_token: 'Intentionally invalid +999 token for equality/conflict tests; not a dialable phone.', company: 'Fictional company label.', role_mailbox: 'Authored flag indicating a shared-role address.' },
expected: { decision: 'link_candidate, review, or keep_separate under this contract.', automatic_merge_allowed: 'Always false; field survivorship and destructive operations are outside scope.' },
},
change: {
input: { before_html: 'Authored static document before the comparison.', after_html: 'Authored static document after the comparison.', before_status: 'Simulated HTTP status, not a measured response.', after_status: 'Simulated HTTP status, not a measured response.', selector: 'Region selected in both documents.', policy: 'Versioned authored text-change contract.' },
expected: { decision: 'changed, unchanged, unavailable, or ambiguous.', before_text: 'Normalized selected text; null means unavailable, empty string means successful empty extraction.', after_text: 'Normalized selected text; null means unavailable, empty string means successful empty extraction.' },
},
};
function dataset(slug, name, kind, rows) {
const dictionary = {
fixture_id: 'Stable authored case ID, not a customer or observation identifier.',
observation_type: 'Always authored_test_fixture: synthetic inputs and author-defined expected labels.',
input: 'Nested test input. JSON object serialized as JSON text in the CSV cell; parse it before use.',
expected: 'Author-defined acceptance oracle, not output observed from a vendor. Nested object serialized as JSON in CSV.',
rule_ids: 'Ordered IDs from dataset.rules explaining the applicable authored contract; JSON array in CSV.',
rationale: 'Reason for the case and its limits; not an empirical performance claim.',
};
return {
'@context': 'https://schema.org', '@type': 'Dataset', name,
description: `${rows.length} original synthetic acceptance fixtures with explicit expected labels. No third-party product has been tested or ranked using this release.`,
url: `${SITE}/blog/${slug}/`, slug, version: VERSION, dateModified: DATE,
creator: { '@type': 'Person', name: 'Tugelbay Konabayev', url: `${SITE}/about/` },
isAccessibleForFree: true,
measurementTechnique: 'Manually authored synthetic fixtures and acceptance oracles; deterministic serialization, not an empirical benchmark.',
variableMeasured: Object.entries(dictionary).map(([propertyID, description]) => ({ '@type': 'PropertyValue', propertyID, name: propertyID, description })),
distribution: ['csv', 'json', 'jsonl'].map(format => ({ '@type': 'DataDownload', encodingFormat: { csv: 'text/csv', json: 'application/json', jsonl: 'application/x-ndjson' }[format], contentUrl: `${SITE}/data/${slug}.${format}` })),
row_count: rows.length, observation_type: 'authored_test_fixture', product_performance_measured: false,
usage_scope: 'Download and use locally for your own acceptance tests, and cite the dataset and version. No open redistribution license is asserted for this fixture release.',
rules: RULES[kind], data_dictionary: dictionary, nested_dictionaries: NESTED_DICTIONARIES[kind],
source_provenance: [{ type: 'original_authorship', authored_date: DATE, scope: 'All fixture inputs, expected outputs and decision rules are original synthetic constructions; methodological citations in the article do not supply these labels.' }],
reproduction: { release_ref: 'acceptance-fixtures-v1.0.0', code_url: CODE, source_sha256: SOURCE_SHA256, instructions: 'Download the version 1.0.0 source from code_url and save it as generator.mjs. Verify its SHA-256 against source_sha256 and retain the source with your outputs; no repository access or package installation is required.', command: 'node ./generator.mjs --out ./fixture-data', check_command: 'node ./generator.mjs --out ./fixture-data --check', csv_objects: 'input, expected and rule_ids are JSON strings inside properly quoted CSV cells.', note: 'The generator reproduces datasets; it does not implement an extractor, CRM matcher or website monitor. compareResults emits pass, mismatch or missing, not automatic failure-type diagnoses.' },
data: rows,
};
}
export function datasets() {
return [
dataset('article-extraction-test-dataset', 'Article extraction acceptance fixtures', 'extraction', extractionFixtures()),
dataset('crm-deduplication-test-dataset', 'CRM deduplication review fixtures', 'crm', crmFixtures()),
dataset('website-change-detection-test-dataset', 'Website change detection acceptance fixtures', 'change', changeFixtures()),
];
}
/** Compare adapter-produced results with this contract only; never a product ranking.
* results: [{fixture_id, actual: