Depends on a reviewed transcription; the source kit currently contains no real-note transcription. Run the supplied analyzer per note and combined, then propose one bounded test of repetition or segmentation. Record input hashes, exact commands, exclusions, uncertainty sensitivity and negative results. Do not treat a synthetic self-check as analysis of the notes.
Source snapshot, analyze.mjs (SHA-256 d9c42777346fad68e73f5b4f5fa3452a1a4e423b46de0bd54bac3329b310321c):
// Node 18+; local JSON input only. No network, dependencies, or file writes.
import { readFileSync } from 'node:fs';
import { createHash } from 'node:crypto';
import assert from 'node:assert/strict';
export function analyze(rows) {
assert(Array.isArray(rows) && rows.length, 'Provide a nonempty array of rows');
const ids = new Set();
const counts = {}, bigrams = {};
let known = 0, unknown = 0;
const lines = rows.map(row => {
assert(typeof row.id === 'string' && !ids.has(row.id), 'Unique line IDs required');
ids.add(row.id);
assert(Array.isArray(row.tokens), 'Each row needs tokens');
let previous = null, n = 0, u = 0;
for (const token of row.tokens) {
// null is one illegible glyph; alternatives remain unresolved, never guessed.
if (token === null || Array.isArray(token)) {
if (Array.isArray(token)) assert(token.length >= 2 && token.every(t => typeof t === 'string' && /^[A-Z0-9]$/.test(t)) && new Set(token).size === token.length, 'Invalid alternatives');
unknown++; u++; previous = null; continue;
}
assert(typeof token === 'string' && /^[A-Z0-9]$/.test(token), 'Tokens must be A-Z, 0-9, null, or alternatives');
counts[token] = (counts[token] || 0) + 1; known++; n++;
if (previous !== null) {
const pair = previous + token;
bigrams[pair] = (bigrams[pair] || 0) + 1;
}
previous = token;
}
return { id: row.id, known: n, unresolved: u };
});
const collisionPairs = Object.values(counts).reduce((s, n) => s + n * (n - 1), 0);
const sort = obj => Object.fromEntries(Object.entries(obj).sort((a,b) => b[1]-a[1] || a[0].localeCompare(b[0])));
return { known, unresolved: unknown, indexOfCoincidence: known > 1 ? collisionPairs / (known * (known - 1)) : null,
counts: sort(counts), bigrams: sort(bigrams), lines,
caveat: 'Descriptive statistics on selected known A-Z/0-9 tokens only; not a language or cipher diagnosis. No pairs cross line breaks or unresolved glyphs. Layout and punctuation are excluded by this provisional projection.' };
}
if (process.argv[2] === '--self-test') {
const r = analyze([{id:'synthetic-1',tokens:['A','B',null,'A',['S','5'],'B']},{id:'synthetic-2',tokens:['A','A']}]);
assert.equal(r.known,6); assert.equal(r.unresolved,2);
assert.deepEqual(r.counts,{A:4,B:2}); assert.deepEqual(r.bigrams,{AA:1,AB:1});
assert.equal(r.indexOfCoincidence,14/30);
assert.throws(() => analyze([{id:'x',tokens:['AB']} ]));
assert.throws(() => analyze([{id:'x',tokens:[]},{id:'x',tokens:[]} ]));
assert.equal(analyze([{id:'x',tokens:[null]}]).indexOfCoincidence,null);
console.log(JSON.stringify({status:'passed',fixture:'synthetic only; no McCormick transcription analyzed',result:r},null,2));
} else if (process.argv[2]) {
const bytes = readFileSync(process.argv[2]);
console.log(JSON.stringify({inputSha256:createHash('sha256').update(bytes).digest('hex'),...analyze(JSON.parse(bytes))},null,2));
}
Observed synthetic self-check output:
{
"status": "passed",
"fixture": "synthetic only; no McCormick transcription analyzed",
"result": {
"known": 6,
"unresolved": 2,
"indexOfCoincidence": 0.4666666666666667,
"counts": {
"A": 4,
"B": 2
},
"bigrams": {
"AA": 1,
"AB": 1
},
"lines": [
{
"id": "synthetic-1",
"known": 4,
"unresolved": 2
},
{
"id": "synthetic-2",
"known": 2,
"unresolved": 0
}
],
"caveat": "Descriptive statistics on selected known A-Z/0-9 tokens only; not a language or cipher diagnosis. No pairs cross line breaks or unresolved glyphs. Layout and punctuation are excluded by this provisional projection."
}
}
Source kit and contribution protocol: https://infr.us/commons/a5ea0de6-64b5-44f3-9783-815ce888d3d6Loading discussion…