from pathlib import Path
import json,re
root=Path(__file__).resolve().parents[1]
def read(n):return json.loads((root/'model'/f'{n}.json').read_text())
# Journal entries retain their original wording and hashes. These display-only
# substitutions repair compressed drafting in the readable continuation.
PROSE_WORDS = {
'cumulativehead': 'cumulative head', 'separatecross': 'separate cross',
'roundedlifespans': 'rounded lifespans', 'theaccepted': 'the accepted',
'anyfixed': 'any fixed', 'theirprefixes': 'their prefixes',
'inverseprefix': 'inverse prefix', 'lowerbranch': 'lower branch',
'fullrefinement': 'full refinement', 'primaryheldhead': 'primary held head',
'theirhead': 'their head', 'scalarforms': 'scalar forms',
'primaryspines': 'primary spines', 'connectedplacement': 'connected placement',
'biographycorrespondence': 'biography correspondence',
'localpartition': 'local partition', 'translatededges': 'translated edges',
'alternatecuts': 'alternate cuts', 'affineconversion': 'affine conversion',
'historicalredaction': 'historical redaction',
'statisticalrarity': 'statistical rarity', 'canonicalsource': 'canonical source',
'mainexplanation': 'main explanation', 'onversion': 'on version',
'existingidentity': 'existing identity', 'Sourcefiles': 'Source files',
'sourcefiles': 'source files', 'sourcehash': 'source hash',
'sequentialjournal': 'sequential journal', 'softwarecheck': 'software check',
'thethree': 'the three', 'upperprefix': 'upper prefix',
'sourceprefix': 'source prefix', 'theprefix': 'the prefix',
'andknown': 'and known', 'theknown': 'the known', 'secondarybranch': 'secondary branch',
'thefixed': 'the fixed', 'theidentical': 'the identical',
'thedeclarednative': 'the declared native', 'fornative': 'for native',
'itshead': 'its head', 'allgive': 'all give',
'notneedanynew': 'not need any new', 'theall': 'the all',
'itslocal': 'its local', 'lowerlegs': 'lower legs',
'requiredreversal': 'required reversal', 'localdefects': 'local defects',
'thedigit': 'the digit', 'particularadditive': 'particular additive',
'thelower': 'the lower', 'Everyplace': 'Every place',
'withoutcarry': 'without carry', 'andthethree': 'and the three',
'theoriginal': 'the original', 'theexplicit': 'the explicit',
'declarednative': 'declared native', 'andunchangedlifespans': 'and unchanged lifespans',
'declaredYear': 'declared Year', 'fullcontrollers': 'full controllers',
'mainreader': 'main reader', 'thecurrentreader': 'the current reader',
'sourcebindings': 'source bindings', 'independentnumerical': 'independent numerical',
'headpairs': 'head pairs', 'finalresearch': 'final research',
'theconnected': 'the connected', 'sourcecontrolled': 'source controlled',
'unchangedlifespans': 'unchanged lifespans', 'andunchanged': 'and unchanged',
'lowerleg': 'lower leg', 'upperleg': 'upper leg',
'headclass': 'head class', 'fullword': 'full word',
'inversehead': 'inverse head', 'Executeaccepted': 'Execute accepted',
'Fallhead': 'Fall head', 'itslocalsegment': 'its local segment',
'thefull': 'the full', 'Noahrefinement': 'Noah refinement',
'Thenewtables': 'The new tables', 'andformulas': 'and formulas',
'nextsubheading': 'next subheading', 'thefoot': 'the foot',
'layoutfix': 'layout fix', 'finalvisualreview': 'final visual review',
'onpage': 'on page', 'bothfollowing': 'both following',
'pagebreak': 'page break', 'establishedoperations': 'established operations', 'wholepaths': 'whole paths', 'locatedsourceinputs': 'located source inputs', 'common-grammar': 'common grammar', 'globalminimality': 'global minimality', 'post-Flood910': 'post-Flood 910',
}
def _plain_prose(s):
# Split ordinary compressed CamelCase, including acronym-to-word boundaries.
s=re.sub(r'(?<=[a-z])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])',' ',s)
for old,new in sorted(PROSE_WORDS.items(), key=lambda item: -len(item[0])):
s=re.sub(r'(?<![A-Za-z])'+re.escape(old)+r'(?![A-Za-z])',new,s)
s=re.sub(r'\bpre Noah\b','pre-Noah',s)
s=re.sub(r'\b(MT|LXX|SP)(?=F(?:[/\s]|$))',r'\1 ',s)
s=re.sub(r'(?<=\S)(?=§)',' ',s)
# Protect identifiers, section numbers, dates and literal file paths before
# separating prose from adjacent numbers. In particular, C1483, File52a,
# SHA256, python3 and §4A.3 must retain their identities.
protected=[]
def hold(match):
protected.append(match.group())
return chr(0xE000+len(protected)-1)
token_pattern=(r'§\d+[A-Za-z]?(?:\.\d+[A-Za-z]?)*'
r'|\bC\d+\b|\bFile_?\d+[a-z]?\b'
r'|\bSHA(?:1|256|512)\b|\bpython\d+\b'
r'|\b\d{4}-\d{2}-\d{2}\b'
r'|\b(?:[A-Za-z0-9_.-]+/)*[A-Za-z0-9_.-]+\.(?:json|md|py|pdf|zip)\b')
s=re.sub(token_pattern,hold,s)
s=re.sub(r'(?<=[A-Za-z])(?=\d)|(?<=\d)(?=[A-Za-z])',' ',s)
s=re.sub(r'([,;:])(?=[A-Za-z0-9+−])',r'\1 ',s)
s=re.sub(r'(?<=[A-Za-z])(?=[+−]\d)',' ',s)
s=re.sub(r'(?<=[A-Za-z0-9])(?=[\ue000-\uf8ff])',' ',s)
s=re.sub(r'(?<=[\ue000-\uf8ff])(?=[A-Za-z])',' ',s)
for i,token in enumerate(protected):
s=s.replace(chr(0xE000+i),token)
return s
def prose(s):
# Inline code and inline/display math are opaque: never rewrite commands,
# identifiers or equations while polishing neighboring human prose.
chunks=re.split(r'(`+[^`]*`+|\\\([^\n]*?\\\)|\\\[[^\n]*?\\\])',s)
return ''.join(chunk if i%2 else _plain_prose(chunk)
for i,chunk in enumerate(chunks))
Evidence
Journal entries retain their original wording and hashes. These display-only
Linked sources and evidence
Edition and provenance
prose.py
SHA-256 5cff9cc062d6c704d8aeb451a777347de61e33c16652d5c65cdc884ee30b4d5f
C480–C1634/Research_Cycles/C1585_C1634/prep/prose.py