From 4204980acd7c80ad9a5f2ae5b57e6311c9925468 Mon Sep 17 00:00:00 2001 From: olabusayoT <50379531+olabusayoT@users.noreply.github.com> Date: Mon, 28 Sep 2026 14:42:41 -0400 Subject: [PATCH 1/6] Add a build/write-prefetch unparse path driven by a dedicated Builder tree Adds an off-by-default (useBuildWritePrefetch) two-pass unparse path: build races ahead of write over the same infoset tree via a coroutine handoff (BuildCoroutine/WriteCoroutine, UnparseSharedContext), resolving OVCs directly against the tree instead of via Suspensions where possible. Build is driven by a new Builder tree paralleling Unparser, wired through the grammar combinators, with unparseBeginForBuild/ unparseEndForBuild entry points on ElementUnparserBase that only the Builder tree calls. Write's coroutine thread spawns only if build's lead or pending-suspension count crosses a threshold; otherwise write runs inline on build's own thread. hasAnyPrefetchBeneficialOVC is scoped per compiling root, gated on the tunable, and requires a non-constant OVC expression via the new isPrefetchBeneficial. DAFFODIL-3065 --- .../daffodil/core/grammar/Grammar.scala | 25 + .../daffodil/core/grammar/Production.scala | 15 + .../grammar/primitives/ChoiceCombinator.scala | 31 + .../DelimiterAndEscapeRelated.scala | 19 + .../primitives/ElementCombinator.scala | 60 + .../primitives/HiddenGroupCombinator.scala | 13 + .../grammar/primitives/LayeredSequence.scala | 19 + .../primitives/NilEmptyCombinators.scala | 16 + .../grammar/primitives/SequenceChild.scala | 18 + .../primitives/SequenceCombinator.scala | 18 + .../grammar/primitives/SpecifiedLength.scala | 6 + .../runtime1/ElementBaseRuntime1Mixin.scala | 16 + .../core/runtime1/GramRuntime1Mixin.scala | 10 + .../runtime1/SchemaSetRuntime1Mixin.scala | 25 +- .../apache/daffodil/core/util/TestUtils.scala | 150 +- .../apache/daffodil/lib/util/Coroutines.scala | 3 + .../org/apache/daffodil/lib/util/MStack.scala | 83 +- .../dpath/SuspendableExpression.scala | 19 + .../runtime1/infoset/InfosetImpl.scala | 28 +- .../runtime1/processors/DataProcessor.scala | 359 +++- .../processors/SchemaSetRuntimeData.scala | 10 +- .../runtime1/processors/Suspension.scala | 25 +- .../processors/SuspensionTracker.scala | 74 +- .../processors/unparsers/BuildState.scala | 207 +++ .../unparsers/BuildWriteCoroutines.scala | 100 ++ .../processors/unparsers/Builder.scala | 263 +++ .../processors/unparsers/UState.scala | 153 +- .../unparsers/UnparseSharedContext.scala | 223 +++ .../processors/unparsers/Unparser.scala | 34 +- .../ChoiceAndOtherVariousUnparsers.scala | 226 ++- .../unparsers/runtime1/ElementUnparser.scala | 226 ++- .../HiddenGroupCombinatorUnparser.scala | 17 +- .../runtime1/LayeredSequenceUnparser.scala | 49 +- .../NilEmptyCombinatorUnparsers.scala | 17 +- .../runtime1/SeparatedSequenceUnparsers.scala | 440 ++++- .../runtime1/SpecifiedLengthUnparsers.scala | 60 +- .../UnseparatedSequenceUnparsers.scala | 147 +- .../unparsers/runtime1/WriteUnparser.scala | 79 + ...OutputValueCalcSuspensionRegressions.scala | 230 +++ .../apache/daffodil/lib/util/TestMStack.scala | 106 ++ .../TestBuildWritePrefetchDataProcessor.scala | 1580 +++++++++++++++++ .../unparsers/TestBoundedPrefetch.scala | 122 ++ .../processors/unparsers/TestBuildState.scala | 89 + .../unparsers/TestBuildWriteArrayChoice.scala | 211 +++ .../unparsers/TestLeadCounter.scala | 115 ++ .../UnparseSharedContextTestFixture.scala | 70 + .../org/apache/daffodil/xsd/dafext.xsd | 50 +- .../daffodil/propGen/TunableGenerator.scala | 3 +- 48 files changed, 5599 insertions(+), 260 deletions(-) create mode 100644 daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildState.scala create mode 100644 daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildWriteCoroutines.scala create mode 100644 daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Builder.scala create mode 100644 daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContext.scala create mode 100644 daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala create mode 100644 daffodil-core/src/test/scala/org/apache/daffodil/core/outputValueCalc/TestOutputValueCalcSuspensionRegressions.scala create mode 100644 daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/TestBuildWritePrefetchDataProcessor.scala create mode 100644 daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBoundedPrefetch.scala create mode 100644 daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildState.scala create mode 100644 daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildWriteArrayChoice.scala create mode 100644 daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestLeadCounter.scala create mode 100644 daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContextTestFixture.scala diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Grammar.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Grammar.scala index 47f1f3740f..d9bdd77ac9 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Grammar.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Grammar.scala @@ -21,9 +21,14 @@ import org.apache.daffodil.core.compiler.ForParser import org.apache.daffodil.core.compiler.ForUnparser import org.apache.daffodil.core.dsom.* import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One import org.apache.daffodil.runtime1.processors.parsers.AssertExpressionEvaluationParser import org.apache.daffodil.runtime1.processors.parsers.NadaParser import org.apache.daffodil.runtime1.processors.parsers.SeqCompParser +import org.apache.daffodil.runtime1.processors.unparsers.Builder +import org.apache.daffodil.runtime1.processors.unparsers.SeqCompBuilder import org.apache.daffodil.runtime1.processors.unparsers.SeqCompUnparser import org.apache.daffodil.unparsers.runtime1.NadaUnparser @@ -124,6 +129,26 @@ class SeqComp private (context: SchemaComponent, children: Seq[Gram]) else if (unparserChildren.length == 1) unparserChildren.head else new SeqCompUnparser(context.runtimeData, unparserChildren.toArray) } + + // Only the (usually at most one) child(ren) that actually build infoset + // content contribute here; write-only children (delimiters, padding, etc.) + // simply have no builder to collect. + lazy val builderChildren: Array[Builder] = { + children + .filter(x => !x.isEmpty && (x.forWhat != ForParser)) + .flatMap { x => x.builder.toList } + .toArray + } + + final override lazy val builder: Maybe[Builder] = { + if (builderChildren.isEmpty) { + Nope + } else if (builderChildren.length == 1) { + One(builderChildren.head) + } else { + One(new SeqCompBuilder(builderChildren)) + } + } } object EmptyGram extends Gram(null) { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Production.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Production.scala index 833b1f71f2..ca526f4317 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Production.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Production.scala @@ -22,7 +22,10 @@ import org.apache.daffodil.core.compiler.ForUnparser import org.apache.daffodil.core.compiler.ParserOrUnparser import org.apache.daffodil.core.dsom.SchemaComponent import org.apache.daffodil.lib.util.Logger +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope import org.apache.daffodil.runtime1.processors.parsers.NadaParser +import org.apache.daffodil.runtime1.processors.unparsers.Builder import org.apache.daffodil.unparsers.runtime1.NadaUnparser /** @@ -107,4 +110,16 @@ final class Prod( else unp } + + final override lazy val builder: Maybe[Builder] = { + if (gram.isEmpty) { + Nope + } else { + (forWhat, gram.forWhat) match { + case (ForParser, _) => Nope + case (_, ForParser) => Nope + case _ => gram.builder + } + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ChoiceCombinator.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ChoiceCombinator.scala index b91428e4ca..d2fd1b0b0a 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ChoiceCombinator.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ChoiceCombinator.scala @@ -29,10 +29,14 @@ import org.apache.daffodil.lib.cookers.ChoiceBranchKeyCooker import org.apache.daffodil.lib.cookers.IntRangeCooker import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.schema.annotation.props.gen.ChoiceLengthKind +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One import org.apache.daffodil.lib.util.MaybeInt import org.apache.daffodil.lib.util.ProperlySerializableMap.* import org.apache.daffodil.runtime1.infoset.ChoiceBranchEvent import org.apache.daffodil.runtime1.processors.RangeBound +import org.apache.daffodil.runtime1.processors.TermRuntimeData import org.apache.daffodil.runtime1.processors.parsers.* import org.apache.daffodil.runtime1.processors.unparsers.* import org.apache.daffodil.unparsers.runtime1.* @@ -332,4 +336,31 @@ case class ChoiceCombinator(ch: ChoiceTermBase, alternatives: Seq[Gram]) new ChoiceCombinatorUnparser(ch.modelGroupRuntimeData, cbm, choiceLengthInBits) } } + + override lazy val builder: Maybe[Builder] = { + val (eventRDMap, optDefaultBranch) = ch.choiceBranchMap + + def builderFor(term: Term): (TermRuntimeData, Builder) = { + val cb = term.termContentBody.builder + val b = if (cb.isDefined) { + cb.get + } else { + EmptyBuilder + } + (term.termRuntimeData, b) + } + + val branchMap: Map[ChoiceBranchEvent, (TermRuntimeData, Builder)] = + eventRDMap.map { case (cbe, branchTerm) => (cbe, builderFor(branchTerm)) } + val defaultBranch: Maybe[(TermRuntimeData, Builder)] = optDefaultBranch match { + case Some(term) => One(builderFor(term)) + case None => Nope + } + + if (branchMap.isEmpty && defaultBranch.isEmpty) { + Nope + } else { + One(new ChoiceBuilder(ch.modelGroupRuntimeData, branchMap, defaultBranch)) + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/DelimiterAndEscapeRelated.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/DelimiterAndEscapeRelated.scala index 6bf946a91d..c49afb7a81 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/DelimiterAndEscapeRelated.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/DelimiterAndEscapeRelated.scala @@ -21,10 +21,12 @@ import org.apache.daffodil.core.dsom.* import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.core.grammar.Terminal import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.lib.util.Misc import org.apache.daffodil.lib.xml.XMLUtils import org.apache.daffodil.runtime1.processors.parsers.{ Parser as DaffodilParser, * } +import org.apache.daffodil.runtime1.processors.unparsers.Builder import org.apache.daffodil.runtime1.processors.unparsers.Unparser as DaffodilUnparser import org.apache.daffodil.unparsers.runtime1.* @@ -55,6 +57,10 @@ case class DelimiterStackCombinatorSequence(sq: SequenceTermBase, body: Gram) override lazy val unparser: DaffodilUnparser = new DelimiterStackUnparser(uInit, uSep, uTerm, sq.termRuntimeData, body.unparser) + + // Delimiters are write-only content; the builder tree skips straight to + // whatever this sequence's body itself builds, if anything. + override lazy val builder: Maybe[Builder] = body.builder } case class DelimiterStackCombinatorChoice(ch: ChoiceTermBase, body: Gram) @@ -77,6 +83,10 @@ case class DelimiterStackCombinatorChoice(ch: ChoiceTermBase, body: Gram) override lazy val unparser: DaffodilUnparser = new DelimiterStackUnparser(uInit, None, uTerm, ch.termRuntimeData, body.unparser) + + // Delimiters are write-only content; the builder tree skips straight to + // whatever this choice's body itself builds, if anything. + override lazy val builder: Maybe[Builder] = body.builder } case class DelimiterStackCombinatorElement(e: ElementBase, body: Gram) @@ -111,6 +121,10 @@ case class DelimiterStackCombinatorElement(e: ElementBase, body: Gram) if (u.isEmpty) u else new DelimiterStackUnparser(uInit, None, uTerm, e.termRuntimeData, u) } + + // Delimiters are write-only content; the builder tree skips straight to + // whatever this element's body itself builds, if anything. + override lazy val builder: Maybe[Builder] = body.builder } case class DynamicEscapeSchemeCombinatorElement(e: ElementBase, body: Gram) @@ -135,4 +149,9 @@ case class DynamicEscapeSchemeCombinatorElement(e: ElementBase, body: Gram) if (u.isEmpty || schemeUnparseIsConstant) u else new DynamicEscapeSchemeUnparser(schemeUnparseOpt.get, e.termRuntimeData, u) } + + // The escape scheme only governs delimiter matching in written bytes; the + // builder tree skips straight to whatever this element's body itself + // builds, if anything. + override lazy val builder: Maybe[Builder] = body.builder } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ElementCombinator.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ElementCombinator.scala index 0dcc7ad5cd..a2e27e860b 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ElementCombinator.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ElementCombinator.scala @@ -28,6 +28,8 @@ import org.apache.daffodil.lib.schema.annotation.props.gen.LengthKind import org.apache.daffodil.lib.schema.annotation.props.gen.Representation import org.apache.daffodil.lib.schema.annotation.props.gen.TestKind import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One import org.apache.daffodil.runtime1.processors.parsers.CaptureEndOfContentLengthParser import org.apache.daffodil.runtime1.processors.parsers.CaptureEndOfValueLengthParser import org.apache.daffodil.runtime1.processors.parsers.CaptureStartOfContentLengthParser @@ -36,6 +38,8 @@ import org.apache.daffodil.runtime1.processors.parsers.ElementParser import org.apache.daffodil.runtime1.processors.parsers.ElementParserInputValueCalc import org.apache.daffodil.runtime1.processors.parsers.NadaParser import org.apache.daffodil.runtime1.processors.parsers.Parser +import org.apache.daffodil.runtime1.processors.unparsers.Builder +import org.apache.daffodil.runtime1.processors.unparsers.ElementBuilder import org.apache.daffodil.runtime1.processors.unparsers.Unparser import org.apache.daffodil.unparsers.runtime1.CaptureEndOfContentLengthUnparser import org.apache.daffodil.unparsers.runtime1.CaptureEndOfValueLengthUnparser @@ -44,6 +48,7 @@ import org.apache.daffodil.unparsers.runtime1.CaptureStartOfValueLengthUnparser import org.apache.daffodil.unparsers.runtime1.ElementOVCSpecifiedLengthUnparser import org.apache.daffodil.unparsers.runtime1.ElementOVCUnspecifiedLengthUnparser import org.apache.daffodil.unparsers.runtime1.ElementSpecifiedLengthUnparser +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase import org.apache.daffodil.unparsers.runtime1.ElementUnparserInputValueCalc import org.apache.daffodil.unparsers.runtime1.ElementUnspecifiedLengthUnparser import org.apache.daffodil.unparsers.runtime1.ElementUnusedUnparser @@ -143,6 +148,37 @@ class ElementCombinator( } } + private lazy val eBuilder: Maybe[Builder] = { + if (eValue.isEmpty) { + Nope + } else { + eValue.builder + } + } + private lazy val eReptypeBuilder: Maybe[Builder] = repTypeElementGram.builder + + // Reuses the same already-memoized instance above for unparseBegin/ + // unparseEnd, so build() and write() see identical node-creation + // behavior; the third branch's builder is whatever subComb builds. + override lazy val builder: Maybe[Builder] = unparser match { + case eu @ (_: ElementOVCSpecifiedLengthUnparser | _: ElementSpecifiedLengthUnparser) => { + val eub = eu.asInstanceOf[ElementUnparserBase] + val contentBuilder = if (eReptypeBuilder.isDefined) { + eReptypeBuilder + } else { + eBuilder + } + One( + new ElementBuilder( + context.erd, + eub.unparseBeginForBuild, + eub.unparseEndForBuild, + contentBuilder + ) + ) + } + case _ => subComb.builder + } } case class ElementUnused(ctxt: ElementBase) @@ -374,6 +410,26 @@ class ElementParseAndUnspecifiedLength( new ElementUnparserInputValueCalc(context.erd, uSetVar) } } + + // Reuses the same already-memoized ElementUnparserBase instance above for + // unparseBegin/unparseEnd, so build() and write() see identical + // nilled/OVC/IVC node-creation behavior. + override lazy val builder: Maybe[Builder] = { + val eu = unparser.asInstanceOf[ElementUnparserBase] + val contentBuilder = if (eRepTypeBuilder.isDefined) { + eRepTypeBuilder + } else { + eBuilder + } + One( + new ElementBuilder( + context.erd, + eu.unparseBeginForBuild, + eu.unparseEndForBuild, + contentBuilder + ) + ) + } } abstract class ElementCombinatorBase( @@ -449,4 +505,8 @@ abstract class ElementCombinatorBase( def unparser: Unparser + lazy val eBuilder: Maybe[Builder] = eGram.builder + + lazy val eRepTypeBuilder: Maybe[Builder] = repTypeElementGram.builder + } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/HiddenGroupCombinator.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/HiddenGroupCombinator.scala index a993c98cfe..1de1cc37c9 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/HiddenGroupCombinator.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/HiddenGroupCombinator.scala @@ -20,8 +20,13 @@ package org.apache.daffodil.core.grammar.primitives import org.apache.daffodil.core.dsom.ModelGroup import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.core.grammar.Terminal +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One import org.apache.daffodil.runtime1.processors.parsers.HiddenGroupCombinatorParser import org.apache.daffodil.runtime1.processors.parsers.Parser +import org.apache.daffodil.runtime1.processors.unparsers.Builder +import org.apache.daffodil.runtime1.processors.unparsers.HiddenGroupBuilder import org.apache.daffodil.runtime1.processors.unparsers.Unparser import org.apache.daffodil.unparsers.runtime1.HiddenGroupCombinatorUnparser @@ -34,4 +39,12 @@ final class HiddenGroupCombinator(ctxt: ModelGroup, body: Gram) override lazy val unparser: Unparser = new HiddenGroupCombinatorUnparser(ctxt.modelGroupRuntimeData, body.unparser) + override lazy val builder: Maybe[Builder] = { + val bb = body.builder + if (bb.isEmpty) { + Nope + } else { + One(new HiddenGroupBuilder(bb.get)) + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/LayeredSequence.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/LayeredSequence.scala index 2b75da3f4b..f86a08f97f 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/LayeredSequence.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/LayeredSequence.scala @@ -20,9 +20,14 @@ package org.apache.daffodil.core.grammar.primitives import org.apache.daffodil.core.dsom.* import org.apache.daffodil.core.grammar.Terminal import org.apache.daffodil.core.layers.LayerSchemaCompiler +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One import org.apache.daffodil.lib.util.Misc import org.apache.daffodil.runtime1.processors.parsers.LayeredSequenceParser import org.apache.daffodil.runtime1.processors.parsers.Parser as DaffodilParser +import org.apache.daffodil.runtime1.processors.unparsers.Builder +import org.apache.daffodil.runtime1.processors.unparsers.SequenceBuilder import org.apache.daffodil.runtime1.processors.unparsers.Unparser as DaffodilUnparser import org.apache.daffodil.unparsers.runtime1.LayeredSequenceUnparser @@ -47,4 +52,18 @@ case class LayeredSequence(sq: SequenceGroupTermBase, bodyTerm: SequenceChild) override lazy val unparser: DaffodilUnparser = new LayeredSequenceUnparser(srd, bodyUnparser) + + // The layer transform itself is a write-only, byte-level concern, but + // (unlike delimiters/escape schemes) this is still a genuine one-child + // sequence position: it must push/pop bodyTerm's TRD and advance the + // group index the same way SequenceBuilder does for any other sequence + // child, or next-element resolution on the shared InfosetInputter breaks. + override lazy val builder: Maybe[Builder] = { + val info = bodyTerm.optSequenceChildBuildInfo + if (info.isEmpty) { + Nope + } else { + One(new SequenceBuilder(IndexedSeq(info.get))) + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/NilEmptyCombinators.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/NilEmptyCombinators.scala index 1ed7a027bc..59d784f16a 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/NilEmptyCombinators.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/NilEmptyCombinators.scala @@ -21,8 +21,13 @@ import org.apache.daffodil.core.dsom.ElementBase import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.core.grammar.Terminal import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One import org.apache.daffodil.runtime1.processors.parsers.ComplexNilOrContentParser import org.apache.daffodil.runtime1.processors.parsers.SimpleNilOrValueParser +import org.apache.daffodil.runtime1.processors.unparsers.Builder +import org.apache.daffodil.runtime1.processors.unparsers.NilOrContentBuilder import org.apache.daffodil.unparsers.runtime1.ComplexNilOrContentUnparser import org.apache.daffodil.unparsers.runtime1.SimpleNilOrValueUnparser @@ -59,4 +64,15 @@ case class ComplexNilOrContent(ctxt: ElementBase, nilGram: Gram, contentGram: Gr override lazy val unparser = ComplexNilOrContentUnparser(ctxt.erd, nilUnparser, contentUnparser) + // A nilled complex element has no children to build; nilled-ness is only + // known once the node exists, so this needs a real runtime check, not a + // static pass-through. + override lazy val builder: Maybe[Builder] = { + val cb = contentGram.builder + if (cb.isEmpty) { + Nope + } else { + One(new NilOrContentBuilder(cb.get)) + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceChild.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceChild.scala index 725eff5363..6aaf594e05 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceChild.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceChild.scala @@ -24,8 +24,12 @@ import org.apache.daffodil.lib.schema.annotation.props.SeparatorSuppressionPolic import org.apache.daffodil.lib.schema.annotation.props.gen.LengthKind import org.apache.daffodil.lib.schema.annotation.props.gen.OccursCountKind import org.apache.daffodil.lib.schema.annotation.props.gen.Representation +import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.runtime1.dpath.NodeInfo import org.apache.daffodil.runtime1.processors.parsers.* +import org.apache.daffodil.runtime1.processors.unparsers.Builder +import org.apache.daffodil.runtime1.processors.unparsers.EmptyBuilder +import org.apache.daffodil.runtime1.processors.unparsers.SequenceChildBuildInfo import org.apache.daffodil.unparsers.runtime1.* /** @@ -69,6 +73,7 @@ abstract class SequenceChild(protected val sq: SequenceTermBase, child: Term, gr protected lazy val childParser = child.termContentBody.parser protected lazy val childUnparser = child.termContentBody.unparser + protected lazy val childBuilder: Maybe[Builder] = child.termContentBody.builder final override lazy val parser = sequenceChildParser final override lazy val unparser = sequenceChildUnparser @@ -82,6 +87,19 @@ abstract class SequenceChild(protected val sq: SequenceTermBase, child: Term, gr final lazy val optSequenceChildUnparser: Option[SequenceChildUnparser] = if (childUnparser.isEmpty) None else Some(unparser) + final lazy val optSequenceChildBuildInfo: Option[SequenceChildBuildInfo] = { + if (childUnparser.isEmpty) { + None + } else { + val cb = if (childBuilder.isDefined) { + childBuilder.get + } else { + EmptyBuilder + } + Some(SequenceChildBuildInfo(unparser, cb)) + } + } + /** * There's only parse result helpers here, so let's abbreviate */ diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceCombinator.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceCombinator.scala index aab310aa2c..7d64103def 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceCombinator.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceCombinator.scala @@ -124,6 +124,15 @@ class OrderedSequence(sq: SequenceTermBase, sequenceChildrenArg: Seq[SequenceChi } } } + + override lazy val builder: Maybe[Builder] = { + val childBuildInfos = sequenceChildren.flatMap { _.optSequenceChildBuildInfo } + if (childBuildInfos.isEmpty) { + Maybe.Nope + } else { + Maybe.One(new SequenceBuilder(childBuildInfos.toIndexedSeq)) + } + } } class UnorderedSequence( @@ -220,4 +229,13 @@ class UnorderedSequence( } } } + + override lazy val builder: Maybe[Builder] = { + val childBuildInfos = sequenceChildren.flatMap { _.optSequenceChildBuildInfo } + if (childBuildInfos.isEmpty) { + Maybe.Nope + } else { + Maybe.One(new SequenceBuilder(childBuildInfos.toIndexedSeq)) + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SpecifiedLength.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SpecifiedLength.scala index 0ae3e6a4d9..7ed2b65a52 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SpecifiedLength.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SpecifiedLength.scala @@ -24,6 +24,7 @@ import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.core.grammar.Terminal import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.schema.annotation.props.gen.LengthUnits +import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.runtime1.dpath.NodeInfo.PrimType import org.apache.daffodil.runtime1.processors.parsers.* import org.apache.daffodil.runtime1.processors.unparsers.* @@ -44,6 +45,11 @@ abstract class SpecifiedLengthCombinatorBase(val e: ElementBase, eGramArg: => Gr u } + // None of the length-kind wrapping below (explicit/implicit/prefixed + // lengths, pattern matching) creates infoset nodes; the builder tree skips + // straight to whatever the wrapped element content itself builds. + override lazy val builder: Maybe[Builder] = eGram.builder + def kind: String def toBriefXML(depthLimit: Int = -1): String = { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ElementBaseRuntime1Mixin.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ElementBaseRuntime1Mixin.scala index b17c2b4d15..26c29fb488 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ElementBaseRuntime1Mixin.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ElementBaseRuntime1Mixin.scala @@ -23,10 +23,12 @@ import org.apache.daffodil.core.dsom.PrefixLengthQuasiElementDecl import org.apache.daffodil.core.dsom.PrimitiveType import org.apache.daffodil.core.dsom.Root import org.apache.daffodil.core.dsom.SimpleTypeDefBase +import org.apache.daffodil.core.dsom.TransitiveClosureSchemaComponents import org.apache.daffodil.lib.schema.annotation.props.gen.LengthKind import org.apache.daffodil.lib.schema.annotation.props.gen.Representation.Text import org.apache.daffodil.lib.util.Delay import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.runtime1.dpath.SuspendableExpression import org.apache.daffodil.runtime1.dsom.DPathElementCompileInfo import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.RuntimeData @@ -134,6 +136,20 @@ trait ElementBaseRuntime1Mixin { self: ElementBase => isReferenced || mightHaveSuspensions } + /** + * True if a component reachable from this element has a + * dfdl:outputValueCalc resolvable without writing; gates + * useBuildWritePrefetch. Scoped to a closure seeded from this + * element, not schemaSet.allSchemaComponents, which is shared and + * schema-set-wide and would leak another root's OVC into this one. + */ + final lazy val hasAnyPrefetchBeneficialOVC: Boolean = + TransitiveClosureSchemaComponents(Seq(this)).exists { + case e: ElementBase if e.isOutputValueCalc => + SuspendableExpression.isPrefetchBeneficial(e.ovcCompiledExpression) + case _ => false + } + final override lazy val dpathCompileInfo = dpathElementCompileInfo /** diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/GramRuntime1Mixin.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/GramRuntime1Mixin.scala index 54b45c8148..5d685f28b6 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/GramRuntime1Mixin.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/GramRuntime1Mixin.scala @@ -21,6 +21,7 @@ import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.runtime1.processors.parsers.Parser +import org.apache.daffodil.runtime1.processors.unparsers.Builder import org.apache.daffodil.runtime1.processors.unparsers.Unparser trait GramRuntime1Mixin { self: Gram => @@ -59,4 +60,13 @@ trait GramRuntime1Mixin { self: Gram => else Maybe(u) } } + + /** + * Provides this Gram's node in the much smaller, dedicated Builder tree + * that parallels the full Unparser tree (see Builder's own doc). Most + * Grams contribute nothing to the infoset's structure and simply inherit + * this Nope default; only Grams that create or select infoset content + * override it. + */ + def builder: Maybe[Builder] = Maybe.Nope } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala index 3e9c5634fe..1f2ae97867 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala @@ -21,6 +21,8 @@ import org.apache.daffodil.core.dsom.SchemaSet import org.apache.daffodil.core.dsom.SequenceTermBase import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Logger +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope import org.apache.daffodil.runtime1.iapi.DFDL import org.apache.daffodil.runtime1.layers.LayerRuntimeCompiler import org.apache.daffodil.runtime1.layers.LayerRuntimeData @@ -29,6 +31,7 @@ import org.apache.daffodil.runtime1.processors.Processor import org.apache.daffodil.runtime1.processors.SchemaSetRuntimeData import org.apache.daffodil.runtime1.processors.VariableMap import org.apache.daffodil.runtime1.processors.parsers.NotParsableParser +import org.apache.daffodil.runtime1.processors.unparsers.Builder import org.apache.daffodil.runtime1.processors.unparsers.NotUnparsableUnparser trait SchemaSetRuntime1Mixin { @@ -61,6 +64,20 @@ trait SchemaSetRuntime1Mixin { unp }.value + // Unlike parser/unparser, not forced eagerly: onPath below only + // references this when tunable.useBuildWritePrefetch is on, since + // there's no public API to reuse this SchemaSet's ssrd under a + // different tunable (tunable is fixed at compile time, see + // Compiler().withTunables), so schemas that never enable it never + // pay to construct the Builder tree. + lazy val builder: Maybe[Builder] = { + if (generateUnparser) { + root.document.builder + } else { + Nope + } + } + private lazy val layerRuntimeCompiler = new LayerRuntimeCompiler private lazy val allLayers: Seq[LayerRuntimeData] = LV(Symbol("allLayers")) { @@ -84,14 +101,20 @@ trait SchemaSetRuntime1Mixin { // null parser/unparser, and that it's impossible for a DataProcessor // to have an error Assert.invariant(!root.isError) + // Only reference builder/hasAnyPrefetchBeneficialOVC (both real + // work: a full parallel Builder tree, a schema component scan) when + // the tunable that would actually use them is on; otherwise every + // schema compile would pay for them regardless. val ssrd = new SchemaSetRuntimeData( parser, unparser, + if (tunable.useBuildWritePrefetch) builder else Nope, root.elementRuntimeData, variableMap, allLayers, - layerRuntimeCompiler + layerRuntimeCompiler, + tunable.useBuildWritePrefetch && root.hasAnyPrefetchBeneficialOVC ) if (root.numComponents > root.numUniqueComponents) Logger.log.debug( diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala index bc70863fe3..63dc5e759b 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala @@ -18,6 +18,7 @@ package org.apache.daffodil.core.util import java.io.ByteArrayInputStream +import java.io.ByteArrayOutputStream import java.io.InputStream import java.net.URL import java.nio.channels.Channels @@ -43,10 +44,21 @@ import org.apache.daffodil.lib.iapi.* import org.apache.daffodil.lib.util.* import org.apache.daffodil.lib.xml.* import org.apache.daffodil.runtime1.iapi.DFDL +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.runtime1.infoset.DIComplex +import org.apache.daffodil.runtime1.infoset.DIDocument +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.infoset.InfosetInputter import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetOutputter import org.apache.daffodil.runtime1.processors.DataProcessor +import org.apache.daffodil.runtime1.processors.SuspensionTracker import org.apache.daffodil.runtime1.processors.VariableMap +import org.apache.daffodil.runtime1.processors.unparsers.BuildFinished +import org.apache.daffodil.runtime1.processors.unparsers.UState +import org.apache.daffodil.runtime1.processors.unparsers.UStateMain +import org.apache.daffodil.runtime1.processors.unparsers.UnparseSharedContext +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase import org.apache.commons.io.FileUtils /* @@ -118,9 +130,11 @@ object TestUtils { testSchema: scala.xml.Elem, infosetXML: Node, unparseTo: String, - areTracing: Boolean = false + areTracing: Boolean = false, + tunables: Map[String, String] = Map.empty ): java.util.List[api.Diagnostic] = { - val compiler = Compiler().withTunable("allowExternalPathExpressions", "true") + val compiler = + Compiler().withTunable("allowExternalPathExpressions", "true").withTunables(tunables) val pf = compiler.compileNode(testSchema) if (pf.isError) throwDiagnostics(pf.getDiagnostics) var u = saveAndReload(pf.onPath("/").asInstanceOf[DataProcessor]) @@ -198,6 +212,138 @@ object TestUtils { p } + /** + * Compiles testSchema with the given tunables and returns the resulting + * DataProcessor, with no saveAndReload round-trip (unlike compileSchema) + * since some callers build test-only state directly off the live object. + */ + def compileForUnparse( + testSchema: Node, + tunables: Map[String, String] = Map.empty + ): DataProcessor = { + val pf = Compiler().withTunables(tunables).compileNode(testSchema) + if (pf.isError) throwDiagnostics(pf.getDiagnostics) + val dp = pf.onPath("/").asInstanceOf[DataProcessor] + if (dp.isError) throwDiagnostics(dp.getDiagnostics) + dp + } + + /** + * Builds a fresh InfosetInputter walking infosetXML against dp, already + * initialized (root TRD pushed) the same way a real unparse would. + */ + def newInitializedInputter(infosetXML: Node, dp: DataProcessor): InfosetInputter = { + val inputter = new InfosetInputter(new ScalaXMLInfosetInputter(infosetXML)) + inputter.initialize(dp.ssrd.elementRuntimeData, dp.tunables) + inputter + } + + /** + * Unparses infosetXML against dp, throwing if the result is an error, + * and returns the raw unparsed bytes. + */ + def unparseToBytes(dp: DataProcessor, infosetXML: Node): Array[Byte] = { + val out = new ByteArrayOutputStream() + val res = dp.unparse(new ScalaXMLInfosetInputter(infosetXML), out) + if (res.isError) throwDiagnostics(res.getDiagnostics) + out.toByteArray + } + + /** + * Compiles testSchema twice - once with default tunables, once with + * useBuildWritePrefetch (plus any extraTunables) - and unparses + * infosetXML both ways. Returns (singlePassBytes, prefetchBytes) for + * the caller to assert equality (and any expected-string checks) on. + */ + def getSinglePassAndPrefetchBytes( + testSchema: Node, + infosetXML: Node, + extraTunables: Map[String, String] = Map.empty + ): (Array[Byte], Array[Byte]) = { + val singlePassDp = compileForUnparse(testSchema) + val singlePassBytes = unparseToBytes(singlePassDp, infosetXML) + + val prefetchDp = + compileForUnparse(testSchema, extraTunables + ("useBuildWritePrefetch" -> "true")) + val prefetchBytes = unparseToBytes(prefetchDp, infosetXML) + + (singlePassBytes, prefetchBytes) + } + + /** + * Single-pass check for write's writeContent path: unparses infosetXML + * via dp's normal single-pass unparse (dp must have releaseUnneededInfoset + * disabled, so the built tree survives) to get single-pass bytes and an + * intact tree, then re-walks that same tree through a fresh write-only + * UState's writeContent, draining suspensions and finalizing the same way + * DataProcessor's write side does. Returns (singlePassBytes, walkerBytes) + * for the caller to assert equality on. + */ + def getSinglePassAndWriteContentBytes( + dp: DataProcessor, + infosetXML: Node, + prefetchLimit: Long = 1000 + ): (Array[Byte], Array[Byte]) = { + val singlePassOut = new ByteArrayOutputStream() + val singlePassResult = dp.unparse(new ScalaXMLInfosetInputter(infosetXML), singlePassOut) + if (singlePassResult.isError) throwDiagnostics(singlePassResult.getDiagnostics) + val singlePassBytes = singlePassOut.toByteArray + + val ustate = singlePassResult.resultState.asInstanceOf[UStateMain] + val builtTree: DIDocument = ustate.documentElement + + val walkerOut = new ByteArrayOutputStream() + val writeInputter = newInitializedInputter(infosetXML, dp) + val writeState = UState.createInitialUState(walkerOut, dp, writeInputter, false) + writeState.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + val sharedCtx = new UnparseSharedContext( + new SuspensionTracker( + dp.tunables.unparseSuspensionWaitYoung, + dp.tunables.unparseSuspensionWaitOld + ), + dp, + dp.tunables, + prefetchLimit + ) + sharedCtx.observeBuildSignal(BuildFinished) + writeState.setSharedContext(sharedCtx) + + primeLeadCounter(sharedCtx, builtTree.child(0)) + + val rootUnparser = dp.ssrd.unparser.asInstanceOf[ElementUnparserBase] + val rootNode = sharedCtx.awaitChild(builtTree, 0) + rootUnparser.writeContent(rootNode, writeState) + writeState.evalSuspensions(isFinal = true) + writeState.getDataOutputStream.setFinished(writeState) + + (singlePassBytes, walkerOut.toByteArray) + } + + /** + * Pre-increments UnparseSharedContext's lead counter once per element in + * node's subtree, for a tree built outside BuildState (writeContent's + * decrementLead call requires the counter already be symmetric). + */ + private def primeLeadCounter(sharedCtx: UnparseSharedContext, node: DINode): Unit = + node match { + case complex: DIComplex => + sharedCtx.incrementLead() + var i = 0 + while (i < complex.numChildren) { + primeLeadCounter(sharedCtx, complex.child(i)) + i += 1 + } + case array: DIArray => + var i = 0 + while (i < array.numChildren) { + primeLeadCounter(sharedCtx, array.child(i)) + i += 1 + } + case _ => + sharedCtx.incrementLead() + } + private def runSchemaOnRBC( testSchema: Node, data: ReadableByteChannel, diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/Coroutines.scala b/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/Coroutines.scala index 74d7d3fdb8..ddd4175227 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/Coroutines.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/Coroutines.scala @@ -91,6 +91,9 @@ trait Coroutine[T] { private var thread_ : Option[Future[Unit]] = None + /** True once this coroutine's own thread has actually been created. */ + final def isStarted: Boolean = thread_.isDefined + private final def init(): Unit = { if (!isMain && thread_.isEmpty) { val thr = Future { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/MStack.scala b/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/MStack.scala index 6c25a867bb..f2823f0561 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/MStack.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/MStack.scala @@ -24,6 +24,12 @@ import Maybe.* object MStack { final case class Mark(val v: Int) extends AnyVal val nullMark = Mark(0) + + /** + * Off by default: growing past initialSize isn't itself wrong, so paying + * this bookkeeping cost on every push isn't worth it normally. + */ + final val trackMaxSizeReached: Boolean = false } /** @@ -33,35 +39,25 @@ object MStack { * catches improper initialization. These were not initializing properly, * so the idiom evolved to use the scala initializers. */ -final class MStackOfBoolean private () - extends MStack[Boolean]((n: Int) => new Array[Boolean](n), false) +final class MStackOfBoolean private (initialSize: Int) + extends MStack[Boolean]((n: Int) => new Array[Boolean](n), false, initialSize) object MStackOfBoolean { - def apply() = { - val stk = new MStackOfBoolean() - stk.init() - stk - } + def apply(initialSize: Int = 32) = new MStackOfBoolean(initialSize) } -final class MStackOfInt extends MStack[Int]((n: Int) => new Array[Int](n), 0) +final class MStackOfInt(initialSize: Int) + extends MStack[Int]((n: Int) => new Array[Int](n), 0, initialSize) object MStackOfInt { - def apply() = { - val stk = new MStackOfInt() - stk.init() - stk - } + def apply(initialSize: Int = 32) = new MStackOfInt(initialSize) } -final class MStackOfLong extends MStack[Long]((n: Int) => new Array[Long](n), 0L) +final class MStackOfLong(initialSize: Int) + extends MStack[Long]((n: Int) => new Array[Long](n), 0L, initialSize) object MStackOfLong { - def apply() = { - val stk = new MStackOfLong() - stk.init() - stk - } + def apply(initialSize: Int = 32) = new MStackOfLong(initialSize) } /** @@ -75,11 +71,11 @@ object MStackOfLong { * So we use an Array[AnyRef] as the representation here, and we * convert null to Nope, and an actual object reference to One(x) */ -final class MStackOfMaybe[T <: AnyRef] { +final class MStackOfMaybe[T <: AnyRef](initialSize: Int = 32) { override def toString = delegate.toString - private val delegate = new MStackOf[T] + private val delegate = new MStackOf[T](initialSize) private val nullT = null.asInstanceOf[T] def copyFrom(other: MStackOfMaybe[T]) = delegate.copyFrom(other.delegate) @@ -120,6 +116,7 @@ final class MStackOfMaybe[T <: AnyRef] { def toListMaybe = delegate.toList.map { (x: AnyRef) => Maybe(x) // Scala compiler bug without this cast } + def maxSizeReached = delegate.maxSizeReached } /** @@ -135,7 +132,7 @@ final class MStackOfMaybe[T <: AnyRef] { * an object reference or null, and call Maybe(thing) explicitly outside the * iteration. Maybe(null) is Nope, and Maybe(thing) is One(thing) if thing is not null. */ -final class MStackOf[T <: AnyRef] extends Serializable { +final class MStackOf[T <: AnyRef](initialSize: Int = 32) extends Serializable { override def toString = delegate.toString @@ -143,7 +140,7 @@ final class MStackOf[T <: AnyRef] extends Serializable { @inline final def length = delegate.length - private val delegate = MStackOfAnyRef() + private val delegate = MStackOfAnyRef(initialSize) @inline final def mark = delegate.mark @inline final def reset(m: MStack.Mark) = delegate.reset(m) @@ -156,6 +153,7 @@ final class MStackOf[T <: AnyRef] extends Serializable { @inline final def isEmpty = delegate.isEmpty def clear() = delegate.clear() def toList = delegate.toList + def maxSizeReached = delegate.maxSizeReached def iterator = delegate.iterator.asInstanceOf[ResettableIterator[T]] @@ -163,15 +161,15 @@ final class MStackOf[T <: AnyRef] extends Serializable { } -private[util] final class MStackOfAnyRef private () - extends MStack[AnyRef]((n: Int) => new Array[AnyRef](n), null.asInstanceOf[AnyRef]) +private[util] final class MStackOfAnyRef private (initialSize: Int) + extends MStack[AnyRef]( + (n: Int) => new Array[AnyRef](n), + null.asInstanceOf[AnyRef], + initialSize + ) object MStackOfAnyRef { - def apply() = { - val stk = new MStackOfAnyRef() - stk.init() - stk - } + def apply(initialSize: Int = 32) = new MStackOfAnyRef(initialSize) } /** @@ -184,16 +182,23 @@ object MStackOfAnyRef { */ protected abstract class MStack[@specialized T] private[util] ( arrayAllocator: (Int) => Array[T], - nullValue: T + nullValue: T, + initialSize: Int = 32 ) { private var index = 0 - private var table: Array[T] = null + private var table: Array[T] = arrayAllocator(initialSize) - def init(): Unit = { - index = 0 - table = arrayAllocator(32) - } + private var maxSizeReached_ = 0 + + /** + * The largest this stack's length has ever grown to, across its + * whole lifetime (not just its current length; pops don't reduce + * this). Diagnostic only: useful for profiling to check whether a + * particular use's initialSize is well-chosen, not for any runtime + * decision. Always 0 unless MStack.trackMaxSizeReached is enabled. + */ + final def maxSizeReached: Int = maxSizeReached_ def copyFrom(other: MStack[T]): Unit = { this.index = other.index @@ -212,6 +217,9 @@ protected abstract class MStack[@specialized T] private[util] ( } } + if (MStack.trackMaxSizeReached && other.maxSizeReached_ > this.maxSizeReached_) { + this.maxSizeReached_ = other.maxSizeReached_ + } } // private var currentIteratorIndex = -1 @@ -254,6 +262,9 @@ protected abstract class MStack[@specialized T] private[util] ( if (index == table.length) table = growArray(table) table(index) = x index += 1 + if (MStack.trackMaxSizeReached && index > maxSizeReached_) { + maxSizeReached_ = index + } } /** diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/SuspendableExpression.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/SuspendableExpression.scala index 0d9decbd98..b599aa1215 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/SuspendableExpression.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/SuspendableExpression.scala @@ -32,12 +32,31 @@ import org.apache.daffodil.runtime1.processors.unparsers.UState * dfdl:setVariable expressions (which variables are in-turn used by * dfdl:outputValueCalc. */ +object SuspendableExpression { + + /** True unless expr calls valueLength/contentLength (which needs a + * written DOS position, not merely a known value); other reads + * resolve once the value is known. A static property of expr alone; + * it doesn't know whether the referenced position is already written. */ + def canResolveWithoutWriting(expr: CompiledExpression[AnyRef]): Boolean = + expr.valueReferencedElementInfos.isEmpty && expr.contentReferencedElementInfos.isEmpty + + /** True only for OVCs that can actually create a suspension worth + * prefetching: a constant expression never suspends in the first place, + * so it derives no benefit from racing build ahead of write. */ + def isPrefetchBeneficial(expr: CompiledExpression[AnyRef]): Boolean = + !expr.isConstant && canResolveWithoutWriting(expr) +} + trait SuspendableExpression extends Suspension { override val isReadOnly = true protected def expr: CompiledExpression[AnyRef] + override def canResolveWithoutWriting: Boolean = + SuspendableExpression.canResolveWithoutWriting(expr) + override def toString = "SuspendableExpression(" + rd.diagnosticDebugName + ", expr=" + expr.prettyExpr + ")" diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetImpl.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetImpl.scala index 5edb5196e9..7c83efe866 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetImpl.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetImpl.scala @@ -176,13 +176,13 @@ sealed trait DINode { private var _isFinal: Boolean = false /** - * Use to mark a node as final, indicating that its value will not change or have - * any children added to it. Setting an element as final does not preclude it from - * being discarded by backtracking, i.e. it is only locally final, but might still - * be inside an enclosing PoU. - * - * This cannot be called if an element is already marked as final to help ensure - * correct use. + * Use to mark a node as final, indicating that its value will not change or have + * any children added to it. Setting an element as final does not preclude it from + * being discarded by backtracking, i.e. it is only locally final, but might still + * be inside an enclosing PoU. + * + * This cannot be called if an element is already marked as final to help ensure + * correct use. */ def setFinal(): Unit = { Assert.invariant(!_isFinal) @@ -1334,6 +1334,13 @@ final class DIArray( final def freeChildIfNoLongerNeeded(index: Int, doFree: Boolean): Unit = { val node = _contents(index) + // A null slot here means write's coroutine already freed it before + // build's own (always doFree=false, see UState.releaseUnneededInfoset) + // redundant call arrived; a doFree=true call hitting null is a real bug. + if (node == null) { + Assert.invariant(!doFree) + return + } if (!node.erd.dpathElementCompileInfo.isReferencedByExpressions) { if (doFree) { // set to null so that the garbage collector can free this node @@ -1825,6 +1832,13 @@ sealed class DIComplex(override val erd: ElementRuntimeData) def freeChildIfNoLongerNeeded(index: Int, doFree: Boolean): Unit = { val node = child(index) + // A null slot here means write's coroutine already freed it before + // build's own (always doFree=false, see UState.releaseUnneededInfoset) + // redundant call arrived; a doFree=true call hitting null is a real bug. + if (node == null) { + Assert.invariant(!doFree) + return + } if (!node.erd.dpathElementCompileInfo.isReferencedByExpressions) { if (doFree) { // set to null so that the garbage collector can free this node diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala index 3b16d3680e..15ec0db9b9 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala @@ -68,8 +68,20 @@ import org.apache.daffodil.runtime1.infoset.XMLTextInfosetOutputter import org.apache.daffodil.runtime1.processors.parsers.PState import org.apache.daffodil.runtime1.processors.parsers.ParseError import org.apache.daffodil.runtime1.processors.parsers.Parser +import org.apache.daffodil.runtime1.processors.unparsers.AwaitChildStalledException +import org.apache.daffodil.runtime1.processors.unparsers.BuildCoroutine +import org.apache.daffodil.runtime1.processors.unparsers.BuildFinished +import org.apache.daffodil.runtime1.processors.unparsers.BuildSignal +import org.apache.daffodil.runtime1.processors.unparsers.BuildState +import org.apache.daffodil.runtime1.processors.unparsers.NotUnparsableUnparser +import org.apache.daffodil.runtime1.processors.unparsers.SuspensionCapableUState import org.apache.daffodil.runtime1.processors.unparsers.UState import org.apache.daffodil.runtime1.processors.unparsers.UnparseError +import org.apache.daffodil.runtime1.processors.unparsers.UnparseSharedContext +import org.apache.daffodil.runtime1.processors.unparsers.WriteCoroutine +import org.apache.daffodil.runtime1.processors.unparsers.WriteDone +import org.apache.daffodil.runtime1.processors.unparsers.WriteSignal +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase /** * Implementation mixin - provides simple helper methods @@ -458,6 +470,311 @@ class DataProcessor( } def unparse(actualInputter: api.infoset.InfosetInputter, outStream: java.io.OutputStream) = { + // hasAnyPrefetchBeneficialOVC is false only when every OVC is + // content-length-dependent, which prefetch could never resolve early + // regardless of how far build races ahead; fall back to single-pass + // automatically in that case, regardless of the tunable. + if (tunables.useBuildWritePrefetch && ssrd.hasAnyPrefetchBeneficialOVC) { + unparseViaBuildThenWrite(actualInputter, outStream) + } else { + unparseSinglePass(actualInputter, outStream) + } + } + + /** + * Shared by unparseViaBuildThenWrite and unparseSinglePass's top-level + * catch blocks: maps an exception caught during unparsing to a failed + * `state` plus its `unparseResult`, or rethrows if it's not one of the + * known unparse-error categories. + */ + private def unparseErrorResult(state: UState, t: Throwable): UnparseResult = t match { + case ue: UnparseError => { + state.addUnparseError(ue) + state.unparseResult + } + case procErr: ProcessingError => { + state.setFailed(procErr.toUnparseError) + state.unparseResult + } + case sde: SchemaDefinitionError => { + // A SDE was detected at runtime (perhaps due to a runtime-valued property like byteOrder or encoding) + // These are fatal, and there's no notion of backtracking them, so they propagate to top level + // here. + state.setFailed(sde) + state.unparseResult + } + case sdefw: SchemaDefinitionErrorFromWarning => { + state.setFailed(sdefw) + state.unparseResult + } + case e: ErrorAlreadyHandled => { + state.setFailed(e.th) + state.unparseResult + } + case e: TunableLimitExceededError => { + state.setFailed(e) + state.unparseResult + } + case se: org.xml.sax.SAXException => { + state.setFailed(new UnparseError(None, None, se)) + state.unparseResult + } + case e: scala.xml.parsing.FatalError => { + state.setFailed(new UnparseError(None, None, e)) + state.unparseResult + } + case ie: InfosetException => { + state.setFailed(new UnparseError(None, None, ie)) + state.unparseResult + } + case th: Throwable => throw th + } + + /** + * Build/write-prefetch unparse path (gated on `useBuildWritePrefetch`). + * `BuildState` builds the tree via the ordinary Unparser recursion. + * Only spawns write's own thread, recursing via `writeContent` and + * blocking on `awaitChild` via `Coroutine[T]`, if build's lead ever + * crosses the prefetch limit; otherwise write runs directly on this + * same thread afterward. + */ + private def unparseViaBuildThenWrite( + actualInputter: api.infoset.InfosetInputter, + outStream: java.io.OutputStream + ): UnparseResult = { + // A NotUnparsableUnparser (dfdl:parseUnparsePolicy="parseOnly") can't be + // cast to ElementUnparserBase or driven through the build/write split; + // unparseSinglePass already runs it via unparse1 and gets the correct + // diagnostic, so reuse that instead of a ClassCastException here. + ssrd.unparser match { + case _: NotUnparsableUnparser => return unparseSinglePass(actualInputter, outStream) + case _ => // fall through to the actual build/write-prefetch path below + } + val inputter = new InfosetInputter(actualInputter) + val rootUnparser = ssrd.unparser + var buildState: BuildState = null + // Cleaned up on write's OWN thread (inside runWriteThread below), not + // this method's outer finally block: it's only ever touched from + // write's thread (io.ThreadCheckMixin's affinity invariant), so + // cleaning it up elsewhere would violate that. + var writeState: UState with SuspensionCapableUState = null + // Null until buildState/writeState exists; the catch block below + // constructs a fallback UState to report through only if setup threw + // before either one was assigned here. + var activeState: UState = null + // Hoisted so the outer catch below can reach it (to abort write's + // thread if it's still parked) even when the exception came from + // partway through setup, before the try block's own local val would + // otherwise have gone out of scope. + var sharedCtx: UnparseSharedContext = null + val buildCoroutine = new BuildCoroutine() + + // Constructs write's UState and wires it into sharedCtx. Must run on + // write's own coroutine thread: writeState's DataOutputStream exists + // 1-to-1 with threads (io.ThreadCheckMixin), so constructing it + // eagerly on build's thread would bind it to the wrong one. + def constructWriteSide(): Unit = { + writeState = UState.createInitialUState(outStream, this, inputter, areDebugging) + writeState.setSharedContext(sharedCtx) + if (areDebugging) writeState.notifyDebugging(true) + init(writeState, rootUnparser) + // Forces evaluation of non-constant defineVariable defaults, same + // as single-pass's doUnparse; write is the side that actually + // reads/writes variables here, so it needs this on its own copy. + writeState.initializeVariables() + writeState.getDataOutputStream.setPriorBitOrder(ssrd.elementRuntimeData.defaultBitOrder) + } + + // evalSuspensions(isFinal = true) runs BEFORE the stack-depth + // invariants below: one tripping first could mask the real + // SuspensionDeadlockException diagnostic this ordering exists to + // surface. + def finishWriteSide(): Unit = { + writeState.setProcessor(rootUnparser) + + // Routed via sharedCtx.suspensionTracker, the SAME tracker + // BuildState registered into, so both build-side and write-side + // suspensions get resolved here. + writeState.evalSuspensions(isFinal = true) + + Assert.invariant(writeState.arrayIterationIndexStack.length == 1) + Assert.invariant(writeState.occursIndexStack.length == 1) + Assert.invariant(writeState.groupIndexStack.length == 1) + Assert.invariant(writeState.childIndexStack.length == 1) + Assert.invariant(writeState.currentInfosetNodeMaybe.isEmpty) + Assert.invariant(writeState.escapeSchemeEVCache.isEmpty) + Assert.invariant(writeState.maybeTopTRD().isEmpty) + Assert.invariant(!writeState.withinHiddenNest) + + Assert.invariant(!writeState.getDataOutputStream.isFinished) + try { + writeState.getDataOutputStream.setFinished(writeState) + } catch { + case boc: BitOrderChangeException => + writeState.SDE(boc) + case fio: FileIOException => + writeState.SDE(fio) + } + } + + // The write side's actual work for one unparse call: constructs its + // own UState, walks the tree build has (or will have) produced via + // writeContent, then finalizes. Returns the failure (if any) rather + // than throwing, so both runWriteThread below and the + // no-coroutine-needed inline call further down can report it the + // same way. + def doWriteSide(firstSignal: BuildSignal): Option[Throwable] = { + try { + // firstSignal may already be BuildFinished (build never crossed + // the prefetch-lead threshold) or BuildAborted (build failed + // early); observeBuildSignal handles either before any real work. + sharedCtx.observeBuildSignal(firstSignal) + constructWriteSide() + val rootElemUnp = rootUnparser.asInstanceOf[ElementUnparserBase] + try { + val rootNode = sharedCtx.awaitChild(inputter.documentElement, 0) + rootElemUnp.writeContent(rootNode, writeState) + } catch { + // A genuine deadlock (if any) surfaces via finishWriteSide's + // own evalSuspensions(isFinal = true) call below, which runs + // regardless of how this try block exits. + case _: AwaitChildStalledException => + } + finishWriteSide() + None + } catch { + // Includes BuildAbortedException, when firstSignal is + // BuildAborted: build's own thread has already unwound via its + // own catch below by the time that fires, so this particular + // result is never actually read by anyone (see runWriteThread). + case t: Throwable => Some(t) + } finally { + if (writeState != null) writeState.getDataOutputStream.cleanUp() + } + } + + // Write's entire coroutine-thread body. Reports failure back through + // the coroutine handoff rather than throwing here, since resumeFinal + // must still run. + def runWriteThread(wc: WriteCoroutine, firstSignal: BuildSignal): Unit = { + // Computed before resumeFinal is called below (never in an outer + // finally): resumeFinal requires returning from run() immediately, + // so build's thread mustn't wake until write's cleanup has + // actually finished, or both threads would run at once. doWriteSide's + // own finally block already covers that cleanup before returning. + val result = WriteDone(doWriteSide(firstSignal)) + wc.resumeFinal(buildCoroutine, result) + } + + // Runs build's own structural recursion to completion, asserting its + // stacks end up balanced and the inputter has nothing left unconsumed. + def runBuildPhase(): Unit = { + buildState = new BuildState(inputter, sharedCtx, Nil, areDebugging) + activeState = buildState + if (areDebugging) { + Assert.invariant(optDebugger.isDefined) + addEventHandler(debugger) + buildState.notifyDebugging(true) + } + init(buildState, rootUnparser) + + // The root element always has a builder: it is exactly the case that + // gets ElementBuilder wrapped around it, regardless of schema content. + ssrd.builder.get.build(buildState) + buildState.popTRD(rootUnparser.context.asInstanceOf[TermRuntimeData]) + buildState.setProcessor(rootUnparser) + + Assert.invariant(buildState.arrayIterationIndexStack.length == 1) + Assert.invariant(buildState.occursIndexStack.length == 1) + Assert.invariant(buildState.groupIndexStack.length == 1) + Assert.invariant(buildState.childIndexStack.length == 1) + Assert.invariant(buildState.currentInfosetNodeMaybe.isEmpty) + Assert.invariant(buildState.maybeTopTRD().isEmpty) + + val remainingEvent = buildState.advanceMaybe + if (remainingEvent.isDefined) { + UnparseError( + Nope, + One(buildState.currentLocation), + "Expected no remaining events, but received %s.", + remainingEvent.get + ) + } + } + + try { + inputter.initialize(ssrd.elementRuntimeData, tunables) + + sharedCtx = new UnparseSharedContext( + // Shared by build and write, which together tick this tracker at + // roughly twice the per-node rate a single traversal would; + // doubling both thresholds restores the intended sweep density. + new SuspensionTracker( + tunables.unparseSuspensionWaitYoung * 2, + tunables.unparseSuspensionWaitOld * 2 + ), + this, + tunables, + prefetchLimit = tunables.unparsePrefetchWindowNodes + ) + + val writeCoroutine = new WriteCoroutine(runWriteThread) + sharedCtx.setCoroutines(buildCoroutine, writeCoroutine) + + runBuildPhase() + + // The tree is now entirely built, but some suspensions may still be + // pending. writeCoroutine.isStarted is false whenever build's lead + // never crossed the prefetch threshold, meaning write was never + // needed concurrently at all; run it directly on this thread rather + // than paying to spawn and hand off to a thread purely to give it + // one, first, and only signal. Either way the result is recorded + // into sharedCtx so a later abortWrite/resumeWrite sees write as + // finished, matching what resumeWrite itself already does. + val finalSignal: WriteSignal = + if (writeCoroutine.isStarted) { + sharedCtx.resumeWrite(BuildFinished) + } else { + val result = WriteDone(doWriteSide(BuildFinished)) + sharedCtx.recordWriteDone(result) + result + } + // Only reassign once writeState is known-good: a null here means + // constructWriteSide() itself failed (reported via finalSignal + // below), so keep reporting through buildState instead. + if (writeState != null) activeState = writeState + finalSignal match { + case WriteDone(Some(t)) => throw t + case WriteDone(None) => // writeState.unparseResult below + case other => + Assert.invariantFailed( + s"write coroutine yielded $other instead of signaling completion" + ) + } + + writeState.unparseResult + } catch { + case t: Throwable => + // If write's thread is still parked (this exception came from + // build's own recursion, before the normal resumeWrite(BuildFinished) + // handoff ran), wake it so it can clean up instead of leaking its + // thread and DOS temp files forever. + if (sharedCtx != null) sharedCtx.abortWrite() + // Safe before initialize(): it only copies variableMap, never + // touching inputter.documentElement. + val reportState = + if (activeState != null) activeState + else UState.createInitialUState(outStream, this, inputter, areDebugging) + unparseErrorResult(reportState, t) + } finally { + if (buildState != null) buildState.getDataOutputStream.cleanUp() + } + } + + private def unparseSinglePass( + actualInputter: api.infoset.InfosetInputter, + outStream: java.io.OutputStream + ) = { val inputter = new InfosetInputter(actualInputter) val unparserState = UState.createInitialUState(outStream, this, inputter, areDebugging) @@ -477,47 +794,7 @@ class DataProcessor( unparserState.evalSuspensions(isFinal = true) unparserState.unparseResult } catch { - case ue: UnparseError => { - unparserState.addUnparseError(ue) - unparserState.unparseResult - } - case procErr: ProcessingError => { - val x = procErr - unparserState.setFailed(x.toUnparseError) - unparserState.unparseResult - } - case sde: SchemaDefinitionError => { - // A SDE was detected at runtime (perhaps due to a runtime-valued property like byteOrder or encoding) - // These are fatal, and there's no notion of backtracking them, so they propagate to top level - // here. - unparserState.setFailed(sde) - unparserState.unparseResult - } - case sdefw: SchemaDefinitionErrorFromWarning => { - unparserState.setFailed(sdefw) - unparserState.unparseResult - } - case e: ErrorAlreadyHandled => { - unparserState.setFailed(e.th) - unparserState.unparseResult - } - case e: TunableLimitExceededError => { - unparserState.setFailed(e) - unparserState.unparseResult - } - case se: org.xml.sax.SAXException => { - unparserState.setFailed(new UnparseError(None, None, se)) - unparserState.unparseResult - } - case e: scala.xml.parsing.FatalError => { - unparserState.setFailed(new UnparseError(None, None, e)) - unparserState.unparseResult - } - case ie: InfosetException => { - unparserState.setFailed(new UnparseError(None, None, ie)) - unparserState.unparseResult - } - case th: Throwable => throw th + case t: Throwable => unparseErrorResult(unparserState, t) } finally { unparserState.getDataOutputStream.cleanUp() } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala index e830aead7b..8eb0a6f6c2 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala @@ -18,22 +18,30 @@ package org.apache.daffodil.runtime1.processors import org.apache.daffodil.lib.exceptions.ThrowsSDE +import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.runtime1.layers.LayerRuntimeCompiler import org.apache.daffodil.runtime1.layers.LayerRuntimeData import org.apache.daffodil.runtime1.layers.LayerVarsRuntime import org.apache.daffodil.runtime1.processors.parsers.Parser +import org.apache.daffodil.runtime1.processors.unparsers.Builder import org.apache.daffodil.runtime1.processors.unparsers.Unparser final class SchemaSetRuntimeData( val parser: Parser, val unparser: Unparser, + val builder: Maybe[Builder], val elementRuntimeData: ElementRuntimeData, /* * The original variables determined by the schema compiler. */ variables: VariableMap, allLayers: Seq[LayerRuntimeData], - @transient layerRuntimeCompilerArg: LayerRuntimeCompiler + @transient layerRuntimeCompilerArg: LayerRuntimeCompiler, + /** True if this schema has an outputValueCalc element whose value could + * resolve without writing (see hasAnyPrefetchBeneficialOVC); baked in at + * compile time like the rest of this class. Gates useBuildWritePrefetch - + * zero benefit means automatic fallback to single-pass. */ + val hasAnyPrefetchBeneficialOVC: Boolean ) extends Serializable with ThrowsSDE { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/Suspension.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/Suspension.scala index d41b3accfa..98016cd91d 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/Suspension.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/Suspension.scala @@ -25,8 +25,8 @@ import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.lib.util.MaybeInt import org.apache.daffodil.lib.util.MaybeULong +import org.apache.daffodil.runtime1.processors.unparsers.SuspensionCapableUState import org.apache.daffodil.runtime1.processors.unparsers.UState -import org.apache.daffodil.runtime1.processors.unparsers.UStateMain import org.apache.daffodil.runtime1.processors.unparsers.UnparseError /** @@ -51,6 +51,17 @@ trait Suspension extends Serializable { */ val isReadOnly = false + /** + * True if this suspension might resolve without any bytes written yet + * (e.g. value/variable read); false if it needs a real DOS bit position + * (e.g. valueLength/contentLength, padding/alignment); distinct from + * isReadOnly. A static, direction-blind heuristic (can't tell a + * backward reference to an already-resolved value from a forward + * one), used by build's retries to skip an attempt that's usually, + * but not always, doomed to block. + */ + def canResolveWithoutWriting: Boolean = false + def UE(ustate: UState, s: String, args: Any*) = { UnparseError(One(rd.schemaFileLocation), One(ustate.currentLocation), s, args*) } @@ -111,7 +122,7 @@ trait Suspension extends Serializable { // // As written, we have a bunch of suspensions that occur, but have // specifically known length of zero bits. So nothing being written out. - // In that case, why do we need to split at all? + // TODO: In that case, why do we need to split at all? // val original = ustate.getDataOutputStream if (mkl.isEmpty || (mkl.isDefined && mkl.get > 0)) { @@ -175,18 +186,20 @@ trait Suspension extends Serializable { // // clone the ustate for use when evaluating the expression // - // TODO: Performance - copying this whole state, just for OVC is painful. - // Some sort of copy-on-write scheme would be better. + // This is a targeted partial clone (shallow VariableMap copy, stack + // tops only), not a full deep copy, but still copies the full + // escapeSchemeEVCache/delimiterStack contents unconditionally. + // TODO: Performance: a copy-on-write scheme could avoid that copy. // val didSplit = (ustate.getDataOutputStream ne original) - val cloneUState = ustate.asInstanceOf[UStateMain].cloneForSuspension(original) + val cloneUState = ustate.asInstanceOf[SuspensionCapableUState].cloneForSuspension(original) if (isReadOnly && didSplit) { Assert.invariantFailed("Shouldn't have split. read-only case") } savedUstate_ = cloneUState - ustate.asInstanceOf[UStateMain].addSuspension(this) + ustate.asInstanceOf[SuspensionCapableUState].addSuspension(this) } final def explain(): Unit = { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SuspensionTracker.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SuspensionTracker.scala index b38ba53909..cf7d99f4a8 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SuspensionTracker.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SuspensionTracker.scala @@ -30,6 +30,13 @@ class SuspensionTracker(suspensionWaitYoung: Int, suspensionWaitOld: Int) { def suspensions: Seq[Suspension] = suspensionsYoung.toSeq ++ suspensionsOld.toSeq + /** + * Total not-yet-done suspensions, without allocating (unlike + * `suspensions`, not safe to call once per node). A throttle signal for + * pacing a no-op-sink traversal against the pending backlog. + */ + def pendingCount: Int = suspensionsYoung.length + suspensionsOld.length + private var count: Int = 0 private var suspensionStatTracked: Int = 0 @@ -47,12 +54,24 @@ class SuspensionTracker(suspensionWaitYoung: Int, suspensionWaitOld: Int) { * suspensions, we attempt to evaluate them first, with the hope that their * resolution might make the young suspensions more likely to evaluate. */ - def evalSuspensions(): Unit = { + def evalSuspensions(): Unit = evalSuspensionsThrottled(buildResolvableOnly = false) + + /** + * A no-op-sink sweep variant: same cadence as evalSuspensions, but with + * buildResolvableOnly=true; a suspension whose canResolveWithoutWriting + * is false is skipped-and-requeued instead of genuinely attempted, since + * this heuristic marks it as usually unresolvable here; it stays pending + * for a later unfiltered sweep either way. + */ + def evalBuildResolvableSuspensions(): Unit = + evalSuspensionsThrottled(buildResolvableOnly = true) + + private def evalSuspensionsThrottled(buildResolvableOnly: Boolean): Unit = { if (count % suspensionWaitOld == 0) { - evalSuspensionQueue(suspensionsOld) + evalSuspensionQueue(suspensionsOld, buildResolvableOnly = buildResolvableOnly) } if (count % suspensionWaitYoung == 0) { - evalSuspensionQueue(suspensionsYoung) + evalSuspensionQueue(suspensionsYoung, buildResolvableOnly = buildResolvableOnly) while (suspensionsYoung.nonEmpty) { suspensionsOld.enqueue(suspensionsYoung.dequeue()) } @@ -65,6 +84,20 @@ class SuspensionTracker(suspensionWaitYoung: Int, suspensionWaitOld: Int) { } } + /** + * Attempts every currently-tracked suspension once, bypassing throttling, + * without treating remaining blocks as an error. Intended for a caller + * whose own traversal has fully finished and needs an already-resolvable + * suspension to actually resolve before it can continue. + */ + def evalSuspensionsUnthrottled(): Unit = { + evalSuspensionQueue(suspensionsOld) + evalSuspensionQueue(suspensionsYoung) + while (suspensionsYoung.nonEmpty) { + suspensionsOld.enqueue(suspensionsYoung.dequeue()) + } + } + /** * Evaluates all suspensions until either they are all evaluated or a * deadlock is detected. This moves all young suspensions to the old queue, @@ -94,23 +127,34 @@ class SuspensionTracker(suspensionWaitYoung: Int, suspensionWaitOld: Int) { } /** - * Attempt to evaluate suspensions on the provie queue. Keep repeating the - * evaluates as long as some progress is being made. Suspensions that - * evaluate sucessfully are removed from the queue. Once suspensions make no - * further progress and are all blocked, we return. Blocked suspensions put - * back on the same queue. + * Attempt to evaluate suspensions on the provided queue. Keep repeating + * the evaluates as long as some progress is being made. Suspensions that + * evaluate successfully are removed from the queue. Once suspensions make + * no further progress and are all blocked, we return. Blocked suspensions + * are put back on the same queue. buildResolvableOnly diverts a + * canResolveWithoutWriting=false suspension into skip-and-requeue instead + * of genuinely attempting it, since this heuristic marks it as usually + * unresolvable here. */ - private def evalSuspensionQueue(queue: Queue[Suspension]): Unit = { + private def evalSuspensionQueue( + queue: Queue[Suspension], + buildResolvableOnly: Boolean = false + ): Unit = { var countOfNotMakingProgress = 0 while (!queue.isEmpty && countOfNotMakingProgress < queue.length) { val s = queue.dequeue() - suspensionStatRuns += 1 - s.runSuspension() - if (!s.isDone) queue.enqueue(s) - if (s.isDone || s.isMakingProgress) { - countOfNotMakingProgress = 0 - } else { + if (buildResolvableOnly && !s.canResolveWithoutWriting) { + queue.enqueue(s) countOfNotMakingProgress += 1 + } else { + suspensionStatRuns += 1 + s.runSuspension() + if (!s.isDone) queue.enqueue(s) + if (s.isDone || s.isMakingProgress) { + countOfNotMakingProgress = 0 + } else { + countOfNotMakingProgress += 1 + } } } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildState.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildState.scala new file mode 100644 index 0000000000..be6085490b --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildState.scala @@ -0,0 +1,207 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream + +import org.apache.daffodil.api +import org.apache.daffodil.io.DirectOrBufferedDataOutputStream +import org.apache.daffodil.io.StringDataInputStreamForUnparse +import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.LocalStack +import org.apache.daffodil.lib.util.MStackOfMaybe +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.runtime1.dpath.UnparserBlocking +import org.apache.daffodil.runtime1.infoset.DIDocument +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.infoset.InfosetAccessor +import org.apache.daffodil.runtime1.infoset.InfosetInputter +import org.apache.daffodil.runtime1.processors.DelimiterStackUnparseNode +import org.apache.daffodil.runtime1.processors.EscapeSchemeUnparserHelper +import org.apache.daffodil.runtime1.processors.Suspension +import org.apache.daffodil.runtime1.processors.SuspensionTracker +import org.apache.daffodil.runtime1.processors.TermRuntimeData +import org.apache.daffodil.runtime1.processors.VariableBox +import org.apache.daffodil.runtime1.processors.VariableMap +import org.apache.daffodil.runtime1.processors.dfa.DFADelimiter + +/** + * A `UState` subclass that consumes an actual `InfosetInputter` and provides + * live Cursor/TRD/index-stack behavior; this is the "build" side of the + * build/write unparse split. The write-only surface (delimiter stack, + * escape scheme cache, and the scratch buffers used for measuring/escaping + * text) is stubbed to error, since nothing build does should ever touch it; + * build never writes content. + * + * `getDataOutputStream` is NOT stubbed: generic `UState` utility methods + * (toString, currentLocation, bitPos0b) call into it unconditionally, so + * `BuildState` constructs an actual DOS wrapping a no-op sink purely to + * satisfy that. + * + * Used only when the `useBuildWritePrefetch` tunable is enabled (default + * false); otherwise unused, and unparsing constructs `UStateMain` + * exclusively as before. + */ +final class BuildState( + private val inputter: InfosetInputter, + sharedCtx: UnparseSharedContext, + diagnosticsArg: Seq[api.Diagnostic], + areDebugging: Boolean +) extends UState( + // Build never reads or writes a variable, so an empty map is enough; + // it's just here to satisfy UState's constructor. + new VariableBox(VariableMap()), + diagnosticsArg, + Maybe(sharedCtx.dataProc), + sharedCtx.tunable, + areDebugging + ) + with SuspensionCapableUState + with TraversalIndexStacks { + + dState.setMode(UnparserBlocking) + setSharedContext(sharedCtx) + + // Purely so generic UState utility methods (toString, currentLocation, + // bitPos0b) have something non-null to call into; never actually + // written to for real output. + setDataOutputStream( + DirectOrBufferedDataOutputStream( + new java.io.OutputStream { override def write(b: Int): Unit = () }, + null, + false, + sharedCtx.tunable.outputStreamChunkSizeInBytes, + sharedCtx.tunable.maxByteArrayOutputStreamBufferSizeInBytes, + sharedCtx.tunable.tempFilePath + ) + ) + + // Build runs ahead of write, so freeing a node here would null out a + // child reference write hasn't read yet; write still frees as normal. + override def releaseUnneededInfoset: Boolean = false + + private def writeOnly = + Assert.usageError("BuildState never writes content, so this write-only state doesn't exist") + + override def escapeSchemeEVCache: MStackOfMaybe[EscapeSchemeUnparserHelper] = writeOnly + override def withUnparserDataInputStream: LocalStack[StringDataInputStreamForUnparse] = + writeOnly + override def withByteArrayOutputStream + : LocalStack[(ByteArrayOutputStream, DirectOrBufferedDataOutputStream)] = writeOnly + override def allTerminatingMarkup: List[DFADelimiter] = writeOnly + override def localDelimiters: DelimiterStackUnparseNode = writeOnly + override def pushDelimiters(node: DelimiterStackUnparseNode): Unit = writeOnly + override def popDelimiters(): Unit = writeOnly + + override def advance: Boolean = inputter.advance + override def advanceAccessor: InfosetAccessor = inputter.advanceAccessor + override def inspect: Boolean = inputter.inspect + override def inspectAccessor: InfosetAccessor = inputter.inspectAccessor + override def fini(): Unit = Assert.usageError("Not to be used on UState") + + override def inspectOrError: InfosetAccessor = { + if (inspect) { + inspectAccessor + } else { + Assert.invariantFailed( + "An InfosetEvent was required for building, but no InfosetEvent was available." + ) + } + } + + override def advanceOrError: InfosetAccessor = { + if (advance) { + advanceAccessor + } else { + Assert.invariantFailed( + "An InfosetEvent was required for building, but no InfosetEvent was available." + ) + } + } + + override def isInspectArrayEnd: Boolean = { + if (!inspect) { + false + } else { + inspectAccessor match { + case e if e.isEnd && e.isArray => true + case _ => false + } + } + } + + def currentInfosetNode: DINode = { + if (currentInfosetNodeMaybe.isEmpty) { + null + } else { + currentInfosetNodeMaybe.get + } + } + + def currentInfosetNodeMaybe: Maybe[DINode] = { + if (currentInfosetNodeStack.isEmpty) { + Nope + } else { + currentInfosetNodeStack.top + } + } + + override val currentInfosetNodeStack = new MStackOfMaybe[DINode] + + // Shared, not owned; one SuspensionTracker queue, both build and write + // see the same one via sharedCtx. + def suspensionTracker: SuspensionTracker = sharedCtx.suspensionTracker + + // Build never evaluates an expression or writes a suspendable child, so + // nothing build does can ever suspend, and this is never called: only + // Suspension.suspend calls it, and that always calls cloneForSuspension + // first, which already throws. + def addSuspension(se: Suspension): Unit = writeOnly + + /** + * Uses evalBuildResolvableSuspensions: canResolveWithoutWriting is a + * static, direction-blind heuristic, so a suspension it marks false + * (usually a forward reference, occasionally an already-resolved + * backward one) is skipped here rather than genuinely retried, and + * left pending for a later, unfiltered sweep instead. + */ + def evalSuspensions(isFinal: Boolean): Unit = { + sharedCtx.suspensionTracker.evalBuildResolvableSuspensions() + if (isFinal) sharedCtx.suspensionTracker.requireFinal() + } + def suspensions = sharedCtx.suspensionTracker.suspensions + + /** + * Build never evaluates an expression or writes a suspendable child, so + * nothing build does can ever suspend, and this is never called. + */ + override def cloneForSuspension(suspendedDOS: DirectOrBufferedDataOutputStream): UState = + writeOnly + + final override def pushTRD(trd: TermRuntimeData): Unit = inputter.pushTRD(trd) + final override def maybeTopTRD(): Maybe[TermRuntimeData] = inputter.maybeTopTRD() + final override def popTRD(trd: TermRuntimeData): TermRuntimeData = { + val poppedTRD = inputter.popTRD() + if (poppedTRD ne trd) + Assert.invariantFailed("TRDs do not match. Expected: " + trd + " got " + poppedTRD) + poppedTRD + } + + final override def documentElement: DIDocument = inputter.documentElement +} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildWriteCoroutines.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildWriteCoroutines.scala new file mode 100644 index 0000000000..861fb32301 --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildWriteCoroutines.scala @@ -0,0 +1,100 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.lib.util.Coroutine +import org.apache.daffodil.lib.util.MainCoroutine + +/* + * The build/write handoff for build/write-prefetch: build and write's + * own content dispatch hand off via Coroutine[T], guaranteeing only one + * of the pair ever runs at a time, via a blocking-queue rendezvous + * between two threads. + */ + +/** Signals build sends to write when resuming it. */ +sealed trait BuildSignal + +/** Build made progress since write last ran; try again. */ +case object MoreTreeAvailable extends BuildSignal + +/** + * Build's top-level recursion has fully returned; no further nodes will + * ever be added. Sent exactly once, and may be the very first signal + * write's coroutine ever receives, if build never crossed the + * prefetch-lead threshold. Write must check for this on every resume. + */ +case object BuildFinished extends BuildSignal + +/** + * Build's own thread failed before its normal completion handoff. Write + * is guaranteed to be paused at this exact moment, so this is how + * build's exception wakes it back up; write skips its usual + * finalization and re-throws build's exception instead. + */ +case object BuildAborted extends BuildSignal + +/** Signals write sends to build when yielding control back. */ +sealed trait WriteSignal + +/** + * Write's own recursive dispatch has blocked, needing more tree before + * it can continue; build should keep building, and will be resumed + * again once it does. + */ +case object WriteNeedsMore extends WriteSignal + +/** + * Write has finished all remaining work, or failed trying to. Sent + * exactly once, as write's last act. The captured error, if any, is + * re-thrown on build's thread instead. + */ +final case class WriteDone(error: Option[Throwable]) extends WriteSignal + +/** + * Build's side of the handoff. Runs on the ORIGINAL calling thread + * (`isMain = true`, so no thread is spawned for it); build already drives + * everything from the caller's thread; only write gets a genuinely new + * thread (`WriteCoroutine` below). + */ +final class BuildCoroutine extends MainCoroutine[WriteSignal] + +/** + * Write's side of the handoff. An actual, separate thread, spawned lazily + * (via `Coroutine`'s own `init()`) the first time build resumes it. + * `runBody` (the caller-supplied build/write driver) must construct + * write's own `UState` as its FIRST action on this thread, not before it + * starts and not on the build thread, since `DataOutputStream` is pinned + * to its creating thread; constructing it eagerly on build's thread would + * bind it to the wrong one. + * + * `runBody` receives the signal that started this thread as its second + * argument, rather than `run()` silently discarding it, because that + * first signal might already be `BuildFinished`; `runBody` must treat it + * like any later resume result, not assume the first is always + * `MoreTreeAvailable`. It must end by calling + * `resumeFinal(buildCoroutine, WriteDone(...))` (call it, then return + * from `run()` immediately; its own contract). + */ +final class WriteCoroutine(runBody: (WriteCoroutine, BuildSignal) => Unit) + extends Coroutine[BuildSignal] { + override protected def run(): Unit = { + val firstSignal = waitForResume() + runBody(this, firstSignal) + } +} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Builder.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Builder.scala new file mode 100644 index 0000000000..dc103fad06 --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Builder.scala @@ -0,0 +1,263 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.One +import org.apache.daffodil.runtime1.infoset.ChoiceBranchEndEvent +import org.apache.daffodil.runtime1.infoset.ChoiceBranchEvent +import org.apache.daffodil.runtime1.infoset.ChoiceBranchStartEvent +import org.apache.daffodil.runtime1.processors.ElementRuntimeData +import org.apache.daffodil.runtime1.processors.ModelGroupRuntimeData +import org.apache.daffodil.runtime1.processors.TermRuntimeData +import org.apache.daffodil.unparsers.runtime1.RepeatingChildUnparser +import org.apache.daffodil.unparsers.runtime1.SequenceChildUnparser + +/** + * A node in a much smaller, dedicated tree of Builders that parallels the + * full Unparser tree: only Grams that actually create or select infoset + * content (elements, sequences, choices, hidden groups) contribute one, so + * building the infoset never has to dispatch through the many write-only + * wrapper unparsers (delimiters, escape schemes, layers, padding, + * specified-length) that sit between them in the Unparser tree. Driven + * exclusively from `BuildState`, never from a write-side `UState`. + */ +trait Builder extends Serializable { + def build(state: UState): Unit +} + +/** + * A no-op stand-in for a choice branch or sequence child whose content is + * empty (e.g. an empty sequence), so its build-time presence can still be + * recorded without anything actually needing to happen. + */ +object EmptyBuilder extends Builder { + override def build(state: UState): Unit = () +} + +/** + * Builds each of several sibling Grams' content in order. Used only where a + * `~` composition has more than one child that actually builds infoset + * content; the common case of at most one such child never needs this. + */ +final class SeqCompBuilder(children: Array[Builder]) extends Builder { + override def build(state: UState): Unit = { + var i = 0 + while (i < children.length) { + children(i).build(state) + i += 1 + } + } +} + +/** + * Builds one element's infoset node and, for complex types, recurses into + * contentBuilder to build descendant nodes. unparseBegin/unparseEnd are the + * same element-kind-specific (plain/nillable/OVC/etc.) node-creation logic + * unparse() itself uses, including the bounded-lookahead lead-counter + * hookup and the deferred simple-value finalization; only the "what does + * this element contain" recursion is redirected to the builder tree instead + * of back into the unparser tree. + */ +final class ElementBuilder( + erd: ElementRuntimeData, + unparseBegin: UState => Unit, + unparseEnd: UState => Unit, + contentBuilder: Maybe[Builder] +) extends Builder { + + override def build(state: UState): Unit = { + unparseBegin(state) + + if (erd.isComplexType) { + state.pushTRD(erd.optComplexTypeModelGroupRuntimeData.get) + if (contentBuilder.isDefined) { contentBuilder.get.build(state) } + state.popTRD(erd.optComplexTypeModelGroupRuntimeData.get) + } + + unparseEnd(state) + } +} + +/** + * Pairs a sequence child's existing occurs-count/array bookkeeping (reused + * as-is from the Unparser tree, since it is cheap, pure state bookkeeping + * unrelated to the tree-walking overhead this Builder tree exists to avoid) + * with that same child's own Builder, which SequenceBuilder recurses into + * instead of the child's full Unparser. + */ +final case class SequenceChildBuildInfo( + childUnparser: SequenceChildUnparser, + childBuilder: Builder +) + +/** + * Builds an entire sequence's children, scalar and array/optional alike. + * Mirrors OrderedSequenceUnparserBase's own build loop exactly, except each + * child's recursive build call targets its Builder rather than its full, + * write-only-wrapper-laden Unparser. + */ +final class SequenceBuilder(children: IndexedSeq[SequenceChildBuildInfo]) extends Builder { + + override def build(state: UState): Unit = { + state.groupIndexStack.push(1L) + + var index = 0 + val limit = children.length + while (index < limit) { + val info = children(index) + val cu = info.childUnparser + val trd = cu.trd + state.pushTRD(trd) + cu match { + case rep: RepeatingChildUnparser => { + state.arrayIterationIndexStack.push(1L) + state.occursIndexStack.push(1L) + val erd = rep.erd + var numOccurrences = 0 + val maxReps = rep.maxRepeats(state) + + Assert.invariant(state.inspect, "No event for building.") + val ev = state.inspectAccessor + if (ev.erd eq erd) { + rep.startArrayOrOptional(state) + while (rep.shouldDoUnparser(rep, state)) { + info.childBuilder.build(state) + numOccurrences += 1 + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + state.moveOverOneGroupIndexOnly() + } + rep.checkFinalOccursCountBetweenMinAndMaxOccurs( + state, + rep, + numOccurrences, + maxReps, + state.arrayIterationPos - 1 + ) + rep.endArrayOrOptional(erd, state) + } else { + rep.checkFinalOccursCountBetweenMinAndMaxOccurs( + state, + rep, + numOccurrences, + maxReps, + 0 + ) + } + + state.arrayIterationIndexStack.pop() + state.occursIndexStack.pop() + } + case _ => { + info.childBuilder.build(state) + trd match { + case erd: ElementRuntimeData if !erd.isRepresented => // ok, skip group advance + case _ => state.moveOverOneGroupIndexOnly() + } + } + } + state.popTRD(trd) + index += 1 + } + + state.groupIndexStack.pop() + } +} + +/** + * Builds just the one structurally-present branch of a choice. Mirrors + * ChoiceCombinatorUnparser's own branch resolution exactly, except the + * resolved branch's recursive build call targets its Builder rather than + * its full Unparser. + */ +final class ChoiceBuilder( + mgrd: ModelGroupRuntimeData, + branchMap: Map[ChoiceBranchEvent, (TermRuntimeData, Builder)], + defaultBranch: Maybe[(TermRuntimeData, Builder)] +) extends Builder { + + private def buildBranch(state: UState, branch: (TermRuntimeData, Builder)): Unit = { + val (trd, builder) = branch + state.pushTRD(trd) + builder.build(state) + state.popTRD(trd) + } + + override def build(state: UState): Unit = { + if (state.withinHiddenNest) { + buildBranch(state, defaultBranch.get) + } else { + state.pushTRD(mgrd) + val event = state.inspectOrError + val key: ChoiceBranchEvent = event match { + case e if e.isStart && (e.isElement || e.isArray) => + ChoiceBranchStartEvent(e.erd.namedQName) + case e if e.isEnd && (e.isElement || e.isArray) => + ChoiceBranchEndEvent(e.erd.namedQName) + } + val fromTable = branchMap.get(key) + val resolved = if (fromTable.isDefined) { + fromTable + } else { + defaultBranch.toOption + } + if (resolved.isEmpty) { + UnparseError( + One(mgrd.schemaFileLocation), + One(state.currentLocation), + "Found next element %s, but expected one of %s.", + key.qname.toExtendedSyntax, + branchMap.keys.map { _.qname.toExtendedSyntax }.mkString(", ") + ) + } + state.popTRD(mgrd) + buildBranch(state, resolved.get) + } + } +} + +/** + * Builds the body of a hidden group. withinHiddenNest must stay maintained + * during build too: it is what tells a hidden element's unparseBegin/ + * unparseEnd to manufacture a node instead of consuming an event that will + * never exist. + */ +final class HiddenGroupBuilder(bodyBuilder: Builder) extends Builder { + override def build(state: UState): Unit = { + try { + state.incrementHiddenDef() + bodyBuilder.build(state) + } finally { + state.decrementHiddenDef() + } + } +} + +/** + * A nilled complex element has no children to build; nilled-ness is only + * known once the node exists, so this checks it at build time rather than + * resolving statically to either branch. + */ +final class NilOrContentBuilder(contentBuilder: Builder) extends Builder { + override def build(state: UState): Unit = { + val inode = state.currentInfosetNode.asComplex + if (!inode.isNilled) { contentBuilder.build(state) } + } +} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala index 27a4381877..fb3b205df9 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala @@ -161,6 +161,19 @@ abstract class UState( def moveOverOneGroupIndexOnly(): Unit def moveOverOneElementChildOnly(): Unit + // A dfdl:occursIndex() expression in an occurrence's own content reads + // arrayIterationIndexStack/occursIndexStack's top, which must track the + // occurrence currently being unparsed; both build and write push/pop + // this pair identically around each array/optional occurrence group. + final def pushOccurrenceIndices(): Unit = { + arrayIterationIndexStack.push(1L) + occursIndexStack.push(1L) + } + final def popOccurrenceIndices(): Unit = { + arrayIterationIndexStack.pop() + occursIndexStack.pop() + } + def inspectOrError: InfosetAccessor def advanceOrError: InfosetAccessor def isInspectArrayEnd: Boolean @@ -386,10 +399,10 @@ abstract class UState( // clone the UState so it can no longer change, and pass that clone into // setFinished. val finfo = this match { - case m: UStateMain => m.cloneForSuspension(dos) + case m: SuspensionCapableUState => m.cloneForSuspension(dos) case _ => Assert.invariantFailed( - "State must be a UStateMain when splitting for bit order change" + "State must be SuspensionCapableUState when splitting for bit order change" ) } @@ -404,9 +417,38 @@ abstract class UState( def documentElement: DIDocument - final val releaseUnneededInfoset: Boolean = !areDebugging && tunable.releaseUnneededInfoset + // Build must never free infoset nodes: build runs ahead of write, so + // freeing one would null out a child reference write hasn't read yet. + // Write still frees as normal once done with a node. + def releaseUnneededInfoset: Boolean def delimitedParseResult = Nope + + // UStateMain owns the real one; UStateForSuspension delegates to its + // mainUState so that a Suspension can always reach its tracker via + // savedUstate, even after it's been cloned off for suspension. + def suspensionTracker: SuspensionTracker + + // Optional reference to the shared build/write lead counter. Defaults + // unset (a no-op for every existing call site); only BuildState sets it + // (in its constructor), and only a write-side UState that opts in (via + // setSharedContext) reads it. + private var sharedContextMaybe: Maybe[UnparseSharedContext] = Nope + final def setSharedContext(ctx: UnparseSharedContext): Unit = sharedContextMaybe = One(ctx) + final def sharedContext: Maybe[UnparseSharedContext] = sharedContextMaybe +} + +/** + * Mixed in by any `UState` that tracks `Suspension`s - `UStateMain` and + * `BuildState`. Only write ever creates one; build never does, but + * still needs `evalSuspensions`/`suspensions` to resolve write-created + * suspensions early against the tree it has already built. + */ +trait SuspensionCapableUState { self: UState => + def suspensions: Seq[Suspension] + def addSuspension(se: Suspension): Unit + def evalSuspensions(isFinal: Boolean): Unit + def cloneForSuspension(suspendedDOS: DirectOrBufferedDataOutputStream): UState } /** @@ -418,7 +460,7 @@ abstract class UState( * memory. */ final class UStateForSuspension( - val mainUState: UStateMain, + val mainUState: UState with SuspensionCapableUState, val dataOutputStream: DirectOrBufferedDataOutputStream, vbox: VariableBox, override val currentInfosetNode: DINode, @@ -430,6 +472,10 @@ final class UStateForSuspension( areDebugging: Boolean ) extends UState(vbox, mainUState.diagnostics, mainUState.dataProc, tunable, areDebugging) { + // Always a live write-side UState, since BuildState never creates a + // suspension to clone off of. + override def releaseUnneededInfoset: Boolean = mainUState.releaseUnneededInfoset + _dataOutputStream = dataOutputStream dState.setMode(UnparserBlocking) dState.setCurrentNode(thisElement.asInstanceOf[DINode]) @@ -443,6 +489,11 @@ final class UStateForSuspension( override def getEncoder(cs: BitsCharset): BitsCharsetEncoder = mainUState.getEncoder(cs) override def suspensions = mainUState.suspensions + override val suspensionTracker = mainUState.suspensionTracker + // sharedContext/setSharedContext are final on UState (single mutable + // field, not overridable), so this clone's own copy is primed explicitly + // here rather than delegated, mirroring mainUState's at construction time. + if (mainUState.sharedContext.isDefined) setSharedContext(mainUState.sharedContext.get) // override def charBufferDataOutputStream = mainUState.charBufferDataOutputStream override def withUnparserDataInputStream = mainUState.withUnparserDataInputStream @@ -503,7 +554,44 @@ final class UStateForSuspension( } } -final class UStateMain private ( +/** + * Stack-backed array-iteration/occurs/group/child index tracking, + * shared by UStateMain and BuildState: each stack starts seeded with 1L, + * and moveOverOne*Only bumps its top by one as navigation advances. + * UStateForSuspension needs none of this; it stubs the stacks to die and + * tracks arrayIterationPos/occursPos as plain frozen Longs captured at + * suspension time instead (see its overrides above), so this lives in a + * mixin rather than directly on UState. + */ +trait TraversalIndexStacks { self: UState => + override val arrayIterationIndexStack = MStackOfLong() + arrayIterationIndexStack.push(1L) + override def moveOverOneArrayIterationIndexOnly(): Unit = + arrayIterationIndexStack.setTop(arrayIterationIndexStack.top + 1) + override def arrayIterationPos = arrayIterationIndexStack.top + + override val occursIndexStack = MStackOfLong() + occursIndexStack.push(1L) + override def moveOverOneOccursIndexOnly(): Unit = + occursIndexStack.setTop(occursIndexStack.top + 1) + override def occursPos = occursIndexStack.top + + override val groupIndexStack = MStackOfLong() + groupIndexStack.push(1L) + override def moveOverOneGroupIndexOnly(): Unit = + groupIndexStack.setTop(groupIndexStack.top + 1) + override def groupPos = groupIndexStack.top + + // TODO: it doesn't look anything is actually reading the value of childindex + // stack. Can we get rid of it? + override val childIndexStack = MStackOfLong() + childIndexStack.push(1L) + override def moveOverOneElementChildOnly(): Unit = + childIndexStack.setTop(childIndexStack.top + 1) + override def childPos = childIndexStack.top +} + +final class UStateMain private[unparsers] ( private val inputter: InfosetInputter, outStream: java.io.OutputStream, vbox: VariableBox, @@ -511,7 +599,11 @@ final class UStateMain private ( dataProcArg: DataProcessor, tunable: DaffodilTunables, areDebugging: Boolean -) extends UState(vbox, diagnosticsArg, One(dataProcArg), tunable, areDebugging) { +) extends UState(vbox, diagnosticsArg, One(dataProcArg), tunable, areDebugging) + with SuspensionCapableUState + with TraversalIndexStacks { + + final val releaseUnneededInfoset: Boolean = !areDebugging && tunable.releaseUnneededInfoset dState.setMode(UnparserBlocking) @@ -553,8 +645,10 @@ final class UStateMain private ( // MStack, since the escape scheme cache logic requires an MStack. We // reallyjust need the top for cloning for suspensions, but that // requires changes to how the escape schema cache is accessed, which - // isn't a trivial change. - val esClone = new MStackOfMaybe[EscapeSchemeUnparserHelper]() + // isn't a trivial change. Sized to the source's actual depth since + // nothing ever pushes onto a suspension's cloned escapeSchemeEVCache + // after this point. + val esClone = new MStackOfMaybe[EscapeSchemeUnparserHelper](escapeSchemeEVCache.length) esClone.copyFrom(escapeSchemeEVCache) Maybe(esClone) } else { @@ -563,8 +657,11 @@ final class UStateMain private ( val ds = if (!delimiterStack.isEmpty) { // If there are any delimiters, then we need to clone them all since - // they may be needed for escaping - val dsClone = new MStackOf[DelimiterStackUnparseNode]() + // they may be needed for escaping. Sized to the source's actual + // depth: pushDelimiters/popDelimiters both die on this clone (see + // below), so it's read-only for the rest of the suspension's life + // and can never grow past this depth. + val dsClone = new MStackOf[DelimiterStackUnparseNode](delimiterStack.length) dsClone.copyFrom(delimiterStack) Maybe(dsClone) } else { @@ -665,29 +762,6 @@ final class UStateMain private ( override val currentInfosetNodeStack = new MStackOfMaybe[DINode] - override val arrayIterationIndexStack = MStackOfLong() - arrayIterationIndexStack.push(1L) - override def moveOverOneArrayIterationIndexOnly() = - arrayIterationIndexStack.setTop(arrayIterationIndexStack.top + 1) - override def arrayIterationPos = arrayIterationIndexStack.top - - override val occursIndexStack = MStackOfLong() - occursIndexStack.push(1L) - override def moveOverOneOccursIndexOnly() = occursIndexStack.setTop(occursIndexStack.top + 1) - override def occursPos = occursIndexStack.top - - override val groupIndexStack = MStackOfLong() - groupIndexStack.push(1L) - override def moveOverOneGroupIndexOnly() = groupIndexStack.setTop(groupIndexStack.top + 1) - override def groupPos = groupIndexStack.top - - // TODO: it doesn't look anything is actually reading the value of childindex - // stack. Can we get rid of it? - override val childIndexStack = MStackOfLong() - childIndexStack.push(1L) - override def moveOverOneElementChildOnly() = childIndexStack.setTop(childIndexStack.top + 1) - override def childPos = childIndexStack.top - override lazy val escapeSchemeEVCache = new MStackOfMaybe[EscapeSchemeUnparserHelper] val delimiterStack = new MStackOf[DelimiterStackUnparseNode]() @@ -707,9 +781,20 @@ final class UStateMain private ( * All the other clones used for outputValueCalc, those never * need to add any. */ - private val suspensionTracker = + private val ownSuspensionTracker = new SuspensionTracker(tunable.unparseSuspensionWaitYoung, tunable.unparseSuspensionWaitOld) + // When sharedContext is set, route through the SAME tracker the other + // side of that split uses, or these suspensions would queue where + // nothing ever drains them. + override def suspensionTracker: SuspensionTracker = { + if (sharedContext.isDefined) { + sharedContext.get.suspensionTracker + } else { + ownSuspensionTracker + } + } + def addSuspension(se: Suspension): Unit = { suspensionTracker.trackSuspension(se) } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContext.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContext.scala new file mode 100644 index 0000000000..7ad487859c --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContext.scala @@ -0,0 +1,223 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.iapi.DaffodilTunables +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.* +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.processors.DataProcessor +import org.apache.daffodil.runtime1.processors.SuspensionTracker + +/** + * What build and write genuinely need to share by reference: the + * `SuspensionTracker` (one queue; build opportunistically resolves + * write-created suspensions early against the tree it has already + * built, and write drains whatever remains at the end). + * + * Also owns the lead counter: how far build is ahead of write, + * incremented once per node build constructs and decremented once per + * node write finishes. Once `leadExceedsPrefetchLimit` or the + * pending-suspension backlog exceeds `pendingSuspensionTripLimit` + * (below), build must resume write before continuing its own + * recursion, bounding how far ahead it may run. + */ +final class UnparseSharedContext( + val suspensionTracker: SuspensionTracker, + val dataProc: DataProcessor, + val tunable: DaffodilTunables, + val prefetchLimit: Long +) { + private var buildLead: Long = 0 + + def incrementLead(): Unit = buildLead += 1 + + def decrementLead(): Unit = { + buildLead -= 1 + Assert.invariant(buildLead >= 0) + } + + def currentLead: Long = buildLead + + def leadExceedsPrefetchLimit: Boolean = buildLead > prefetchLimit + + /** + * A second, independent limit on how far build may run ahead of write, + * alongside prefetchLimit: a suspension can be created without moving + * the lead counter, so the lead alone doesn't bound how many pile up + * pending. + */ + def pendingSuspensionTripLimit: Long = tunable.unparsePendingSuspensionTripLimit + + private var buildCoroutine_ : BuildCoroutine = null + private var writeCoroutine_ : WriteCoroutine = null + + def setCoroutines(bc: BuildCoroutine, wc: WriteCoroutine): Unit = { + buildCoroutine_ = bc + writeCoroutine_ = wc + } + def buildCoroutine: BuildCoroutine = buildCoroutine_ + def writeCoroutine: WriteCoroutine = writeCoroutine_ + + private var recordedWriteDone: Maybe[WriteDone] = Nope + + /** + * Resumes writeCoroutine with `signal`, unless write already finished + * (a second resume would park forever), in which case the cached + * WriteDone is returned instead. Always go through this, never resume + * writeCoroutine directly. + */ + def resumeWrite(signal: BuildSignal): WriteSignal = { + if (recordedWriteDone.isDefined) { + recordedWriteDone.get + } else { + val result = buildCoroutine.resume(writeCoroutine, signal) + result match { + case wd: WriteDone => recordedWriteDone = One(wd) + case _ => + } + result + } + } + + /** + * Records write's own result without ever resuming writeCoroutine, + * for when build never needed write started at all and it ran + * directly on build's own thread instead. Still must be recorded + * here so a later abortWrite/resumeWrite sees write as finished. + */ + def recordWriteDone(wd: WriteDone): Unit = { + recordedWriteDone = One(wd) + } + + /** + * Wakes write's thread (spawning it first if never started) after + * build's own thread failed before reaching the normal resumeWrite + * handoff, so it can clean up instead of hanging forever. Blocks until + * that cleanup actually finishes, not just until it starts, so the + * caller never reports an error result while write's cleanup is still + * running on another thread. A no-op if write already finished, or if + * setCoroutines was never reached. + */ + def abortWrite(): Unit = { + if (recordedWriteDone.isEmpty && writeCoroutine_ != null) { + buildCoroutine.resume(writeCoroutine, BuildAborted) + } + } + + /** + * True if `child` may safely be written now: complex/array existing is + * enough; simple needs a value, except hidden/nilled elements (never + * given one) and OVC (deferred via its own Suspension; must not block, + * or an OVC depending on a later sibling's write-time property would deadlock). + */ + private def isChildReady(child: DINode): Boolean = { + if (child.isSimple) { + val s = child.asSimple + child.isHidden || s.isNilled || s.hasValue || s.erd.dpathElementCompileInfo.isOutputValueCalc + } else { + // complex/array, hidden or not; existing is enough + true + } + } + + /** + * True once build's recursion has fully returned. Sticky: once true, + * awaitChild must never again resume buildCoroutine; build's thread has + * moved on to waiting for write's ultimate WriteDone, not another + * WriteNeedsMore cycle, so this can't be a per-call local flag. + */ + private var buildFinished: Boolean = false + + /** + * Records a signal write's coroutine received, without itself resuming + * anyone. Needed for the signal that STARTS write's thread, which can + * legitimately already be BuildFinished; must be recorded before any + * awaitChild call, or tryUnblockWrite would wrongly resume build again. + */ + def observeBuildSignal(signal: BuildSignal): Unit = { + if (signal == BuildFinished) buildFinished = true + else if (signal == BuildAborted) throw new BuildAbortedException + } + + /** + * One attempt at unblocking write: resumes build if not finished yet, + * else retries suspensions directly. False means no progress was made; + * callers must throw AwaitChildStalledException rather than loop again, + * deferring to evalSuspensions(isFinal = true) for the real diagnosis. + */ + private def tryUnblockWrite(): Boolean = { + if (!buildFinished) { + observeBuildSignal(writeCoroutine.resume(buildCoroutine, WriteNeedsMore)) + true + } else { + val before = suspensionTracker.suspensions.length + suspensionTracker.evalSuspensionsUnthrottled() + suspensionTracker.suspensions.length < before + } + } + + /** + * Blocks (parks write's thread) until `parent.child(index)` both exists + * and is ready (isChildReady above), throwing AwaitChildStalledException + * once tryUnblockWrite reports no further progress is possible. + */ + def awaitChild(parent: DINode, index: Int): DINode = { + while (index >= parent.numChildren || !isChildReady(parent.child(index))) { + if (!tryUnblockWrite()) throw new AwaitChildStalledException + } + parent.child(index) + } + + /** + * Blocks until `parent.child(index)` exists, or `parent.isFinal` with no + * child there (none ever will be); for callers asking "is there more + * data at all" rather than awaiting a child known to be coming. A true + * result may still need awaitChild afterward for value readiness. + */ + def childExistsOrFinal(parent: DINode, index: Int): Boolean = { + while (index >= parent.numChildren) { + if (parent.isFinal) return false + if (!tryUnblockWrite()) throw new AwaitChildStalledException + } + true + } +} + +/** + * Thrown when write can make no further progress after build has + * finished and an unthrottled suspension retry made no headway. Caught + * only by the top-level coroutine driver, which still runs its normal + * finalization (the genuine, diagnostic-producing final suspension + * drain) rather than treating this as the final outcome itself. + */ +final class AwaitChildStalledException extends Exception + +/** + * Thrown the moment write's thread observes that build's own thread + * failed with an exception before reaching its normal completion, either + * as the very first signal write's thread ever receives, or as the + * result of a resume from deep inside a block on a pending child. Caught + * only by the top-level coroutine driver, which skips its own normal + * finalization entirely (those invariants and the final suspension drain + * assume a consistently, fully-built tree that a genuine abort may not + * have produced) and just cleans up write's own resources; build's own + * exception, not this one, is what actually gets reported. + */ +final class BuildAbortedException extends Exception diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala index 592544ff0d..7d7d4f8b2f 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala @@ -20,7 +20,9 @@ package org.apache.daffodil.runtime1.processors.unparsers import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.runtime1.dsom.RuntimeSchemaDefinitionError +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.* +import org.apache.daffodil.unparsers.runtime1.WriteUnparser sealed trait Unparser extends Processor { @@ -48,7 +50,6 @@ sealed trait Unparser extends Processor { // keeping track of prior bit order. Finding those has been problematic. // // So this is a temporary fix, until we can figure out where else to do this. - // this match { // bit order only applies to primitives, not combinators, nor "noData" unparsers. case af: AlignmentPrimUnparser => // ok. Don't check bitOrder before Aligning. @@ -72,7 +73,11 @@ sealed trait Unparser extends Processor { ustate.resetFormatInfoCaches() } if (ustate.dataProc.isDefined) ustate.dataProc.get.after(ustate, this) - ustate.setMaybeProcessor(savedProc) + // Restore the prior processor only if one existed. Nope means this is + // the first unparse1 call on a freshly cloned suspension UState, which + // starts with none; resetting to Nope would discard the only context + // it will ever have, which is still needed once the suspension completes. + if (savedProc.isDefined) ustate.setMaybeProcessor(savedProc) } def UE(ustate: UState, s: String, args: Any*) = { @@ -147,7 +152,8 @@ final class ErrorUnparser(override val context: TermRuntimeData = null) final class SeqCompUnparser(context: RuntimeData, val childUnparsers: Array[Unparser]) extends CombinatorUnparser(context) - with ToBriefXMLImpl { + with ToBriefXMLImpl + with WriteUnparser { override val runtimeDependencies = Array() @@ -158,9 +164,27 @@ final class SeqCompUnparser(context: RuntimeData, val childUnparsers: Array[Unpa def unparse(ustate: UState): Unit = { var i = 0 while (i < childUnparsers.length) { - val unparser = childUnparsers(i) + childUnparsers(i).unparse1(ustate) + i += 1 + } + } + + /** + * SeqCompUnparser can wrap any `WriteUnparser` (sequence/choice/ + * hidden-group/delimiter-stack) alongside plain prims: a `WriteUnparser` + * recurses into its own writeContent; everything else (a bare element + * never appears here directly, only wrapped by one) runs via unparse1. + */ + override def writeContent(containerNode: DINode, ustate: UState): Unit = { + var i = 0 + while (i < childUnparsers.length) { + childUnparsers(i) match { + case wu: WriteUnparser => + wu.writeContent(containerNode, ustate) + case cu => + cu.unparse1(ustate) + } i += 1 - unparser.unparse1(ustate) } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala index 9cfff9e970..74b0045a3c 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala @@ -19,6 +19,7 @@ package org.apache.daffodil.unparsers.runtime1 import scala.jdk.CollectionConverters.* +import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.lib.util.MaybeInt @@ -78,13 +79,124 @@ class ChoiceCombinatorUnparser( choiceBranchMap: ChoiceBranchMap, choiceLengthInBits: MaybeInt ) extends CombinatorUnparser(mgrd) - with ToBriefXMLImpl { + with ToBriefXMLImpl + with WriteUnparser { override def nom = "Choice" override val runtimeDependencies = Array() override def childProcessors = choiceBranchMap.childProcessors + /** + * Resolves which branch an already-built child belongs to by keying off + * the child's element identity, blocking until that's known, then + * recurses into the resolved branch. Self-manages advancing/freeing its + * own resolved tree-child position, and applies the same choice-length + * filling as the structural pass. Covers the visible-choice path only; + * the hidden-choice default branch is handled separately below. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = { + val sharedCtx = state.sharedContext.get + val complex = containerNode.asComplex + val childIndex = state.childIndexStack.top.toInt + + val (maybeChildUnparser, resolvedChildIndex): (Maybe[Unparser], Int) = + if (state.withinHiddenNest) { + // A hidden choice's branch is always the single deterministic + // default one (DFDL requires its outcome to be schema-determined, + // not data-driven), so no key/event peek is needed. The tree child at this position, if any, IS this default branch's own content, not something to skip past. + val idx = if (sharedCtx.childExistsOrFinal(complex, childIndex)) { + childIndex + } else { + -1 + } + (Maybe.toMaybe(choiceBranchMap.defaultUnparser), idx) + } else { + // Hidden elements never produce infoset events, so the branch + // lookup keys are built from the first represented child. Build + // still materializes a branch's leading hidden group as actual + // tree children, so skip past those first. + var idx = childIndex + while (sharedCtx.childExistsOrFinal(complex, idx) && complex.child(idx).isHidden) { + idx += 1 + } + if (idx >= complex.numChildren) { + // Build is done and no child ever showed up: the choice resolved + // to a branch with no infoset footprint at all (e.g. an empty + // sequence, or an absent defaultable element); fall back to the + // default/unmapped branch. + (Maybe.toMaybe(choiceBranchMap.defaultUnparser), -1) + } else { + val child = sharedCtx.awaitChild(complex, idx) + val key: ChoiceBranchEvent = ChoiceBranchStartEvent(child.erd.namedQName) + val fromTable = choiceBranchMap.lookupTable.get(key) + if (fromTable != null) { + // An actual match; this tree position genuinely belongs to this + // choice. + (One(fromTable), idx) + } else { + // No branch key matches this child, so it must belong to a + // sibling term after this choice: this choice resolved to a + // branch with no infoset footprint here and consumes no + // tree position. + (Maybe.toMaybe(choiceBranchMap.defaultUnparser), -1) + } + } + } + if (maybeChildUnparser.isEmpty) { + // A real UnparseError, not an internal assertion: a choice with no + // default and no match for what's actually in the tree is + // malformed input, not an invariant violation. + UnparseError( + One(mgrd.schemaFileLocation), + One(state.currentLocation), + "No matching or default choice branch found." + ) + } + val child = if (resolvedChildIndex >= 0) { + complex.child(resolvedChildIndex) + } else { + // A resumable group or ChoiceBranchEmptyUnparser needs no tree child, + // so null is fine, but the unmapped default can also be a bare + // ElementUnparserBase, which DOES need one: writeContent(null, state) + // on that would NPE deep inside it instead of failing clearly here. + if (maybeChildUnparser.get.isInstanceOf[ElementUnparserBase]) { + Assert.invariantFailed( + "Choice resolved to a simple-element default branch with no resolved tree position." + ) + } + null + } + + // True when the resolved branch is itself a resumable group, whose own + // dispatch already advances/frees its tree positions; this choice must + // not also advance/free then, or it double-advances past the branch's + // last child, skipping the following sibling term. + var innerSelfManagesPosition = false + withChoiceLengthFiller(state) { + maybeChildUnparser.get match { + case elemUnp: ElementUnparserBase => elemUnp.writeContent(child, state) + case wu: WriteUnparser => + innerSelfManagesPosition = true + wu.writeContent(containerNode, state) + case emptyUnp: ChoiceBranchEmptyUnparser => + // A branch that optimized to nothing (e.g. a sequence containing + // only an assert); runs its (no-op) unparse, same as any other + // synchronous branch. + emptyUnp.unparse1(state) + case other => + Assert.usageError(s"unhandled choice branch unparser type: $other") + } + } + + // Nothing to advance/free when the branch had no infoset footprint + // (resolvedChildIndex == -1), nor when it's itself a resumable group (innerSelfManagesPosition), which already did so for its own positions. + if (resolvedChildIndex >= 0 && !innerSelfManagesPosition) { + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(resolvedChildIndex, state.releaseUnneededInfoset) + } + } + def unparse(state: UState): Unit = { if (state.withinHiddenNest) { val branchForUnparseIfHidden = choiceBranchMap.defaultUnparser @@ -121,22 +233,36 @@ class ChoiceCombinatorUnparser( val childUnparser = maybeChildUnparser.get state.popTRD(mgrd) state.pushTRD(childUnparser.context.asInstanceOf[TermRuntimeData]) - if (choiceLengthInBits.isDefined) { - val suspendableOp = - new ChoiceUnusedUnparserSuspendableOperation(mgrd, choiceLengthInBits.get) - val choiceUnusedUnparser = - new ChoiceUnusedUnparser(mgrd, choiceLengthInBits.get, suspendableOp) - - suspendableOp.captureDOSStartForChoiceUnused(state) - childUnparser.unparse1(state) - suspendableOp.captureDOSEndForChoiceUnused(state) - choiceUnusedUnparser.unparse(state) - } else { + withChoiceLengthFiller(state) { childUnparser.unparse1(state) } state.popTRD(childUnparser.context.asInstanceOf[TermRuntimeData]) } } + + /** + * Wraps runChosenBranch with the dfdl:choiceLength "unused region" + * filler (no-op if choiceLengthInBits isn't set), capturing DOS + * position before/after via ChoiceUnusedUnparser. The + * state.setProcessor calls are redundant for unparse()'s call site but + * required for writeContent's, which bypasses unparse1 (and its + * equivalent setProcessor) entirely. + */ + private def withChoiceLengthFiller(state: UState)(runChosenBranch: => Unit): Unit = { + if (choiceLengthInBits.isEmpty) { + runChosenBranch + } else { + val suspendableOp = + new ChoiceUnusedUnparserSuspendableOperation(mgrd, choiceLengthInBits.get) + val unusedUnparser = new ChoiceUnusedUnparser(mgrd, choiceLengthInBits.get, suspendableOp) + state.setProcessor(ChoiceCombinatorUnparser.this) + suspendableOp.captureDOSStartForChoiceUnused(state) + runChosenBranch + state.setProcessor(ChoiceCombinatorUnparser.this) + suspendableOp.captureDOSEndForChoiceUnused(state) + unusedUnparser.unparse(state) + } + } } class DelimiterStackUnparser( @@ -145,7 +271,22 @@ class DelimiterStackUnparser( terminatorOpt: Maybe[TerminatorUnparseEv], ctxt: TermRuntimeData, bodyUnparser: Unparser -) extends CombinatorUnparser(ctxt) { +) extends CombinatorUnparser(ctxt) + with WriteUnparser { + + /** + * Pushes the delimiter scope, recurses into the body, and pops only + * once the body's own writeContent (if any) returns, since a pending + * pause still needs the stack for separator writing. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop( + containerNode, + bodyUnparser, + state, + setup = pushDelimiterScope, + teardown = (state, _) => state.popDelimiters() + ) override def nom = "DelimiterStack" override def toBriefXML(depthLimit: Int = -1): String = { @@ -164,7 +305,15 @@ class DelimiterStackUnparser( (initiatorOpt.toList ++ separatorOpt.toList ++ terminatorOpt.toList).toArray def unparse(state: UState): Unit = { - // Evaluate Delimiters + pushDelimiterScope(state) + try { + bodyUnparser.unparse1(state) + } finally { + state.popDelimiters() + } + } + + private def pushDelimiterScope(state: UState): Unit = { val init = if (initiatorOpt.isDefined) initiatorOpt.get.evaluate(state) else EmptyDelimiterStackUnparseNode.empty @@ -174,14 +323,7 @@ class DelimiterStackUnparser( val term = if (terminatorOpt.isDefined) terminatorOpt.get.evaluate(state) else EmptyDelimiterStackUnparseNode.empty - - val node = DelimiterStackUnparseNode(init, sep, term) - - state.pushDelimiters(node) - - bodyUnparser.unparse1(state) - - state.popDelimiters() + state.pushDelimiters(DelimiterStackUnparseNode(init, sep, term)) } } @@ -189,25 +331,43 @@ class DynamicEscapeSchemeUnparser( escapeScheme: EscapeSchemeUnparseEv, ctxt: TermRuntimeData, bodyUnparser: Unparser -) extends CombinatorUnparser(ctxt) { +) extends CombinatorUnparser(ctxt) + with WriteUnparser { override def nom = "EscapeSchemeStack" override def childProcessors = Vector(bodyUnparser) override val runtimeDependencies = Array(escapeScheme) + /** + * Caches the escape scheme, recurses into the body, and invalidates + * the cache only once the body's own writeContent (if any) returns, + * since a pending pause still needs the cache for delimiter writing. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop( + containerNode, + bodyUnparser, + state, + setup = cacheEscapeScheme, + teardown = (state, _) => escapeScheme.invalidateCache(state) + ) + def unparse(state: UState): Unit = { - // evaluate the dynamic escape scheme in the correct scope. the resulting - // value is cached in the Evaluatable (since it is manually cached) and - // future parsers/evaluatables that use this escape scheme will use that - // cached value. + cacheEscapeScheme(state) + try { + bodyUnparser.unparse1(state) + } finally { + escapeScheme.invalidateCache(state) + } + } + + // Evaluates the dynamic escape scheme in the correct scope; the result is + // cached in the Evaluatable (since it is manually cached), so future + // unparsers/evaluatables that use this escape scheme reuse that cached + // value. + private def cacheEscapeScheme(state: UState): Unit = { escapeScheme.newCache(state) escapeScheme.evaluate(state) - - // Unparse - bodyUnparser.unparse1(state) - - // invalidate the escape scheme cache - escapeScheme.invalidateCache(state) } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala index c1343654ba..636d7bc055 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala @@ -24,6 +24,7 @@ import org.apache.daffodil.lib.util.MaybeULong import org.apache.daffodil.runtime1.dpath.SuspendableExpression import org.apache.daffodil.runtime1.dsom.CompiledExpression import org.apache.daffodil.runtime1.infoset.DIComplex +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.infoset.DISimple import org.apache.daffodil.runtime1.infoset.DataValue.DataValuePrimitive import org.apache.daffodil.runtime1.infoset.RetryableException @@ -31,6 +32,7 @@ import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.Evaluatable import org.apache.daffodil.runtime1.processors.UnparseTargetLengthInBitsEv import org.apache.daffodil.runtime1.processors.unparsers.* +import org.apache.daffodil.runtime1.processors.unparsers.MoreTreeAvailable /** * Elements that, when unparsing, have no length specified. @@ -65,6 +67,42 @@ sealed trait RepMoveMixin { } } +/** + * The build-side hookup for bounded lookahead: called once per node, only + * from the Builder tree (via unparseBeginForBuild below), never from + * write's or single-pass's own unparseBegin call. Increments the shared + * lead counter, and once it exceeds the prefetch limit, or too many + * suspensions are pending, resumes write's coroutine and blocks until it + * yields back before continuing build's own recursion. + */ +private object BuildWriteLeadHookup { + + /** + * Must be called after unparseBegin, not unparseEnd: an ancestor's + * increment must happen before any descendant's content is built. A + * write cascade triggered here can reach that ancestor before + * build's own recursive call into it returns; incrementing in + * unparseEnd instead would underflow the counter. + */ + def afterNodeAdded(state: UState): Unit = { + val ctx = state.sharedContext.get + ctx.incrementLead() + // leadExceedsPrefetchLimit alone doesn't bound the retained + // memory of pending suspensions, so pendingSuspensionTripLimit + // separately caps total pendingCount; resuming write here can + // also resolve suspensions immediately. + if ( + ctx.leadExceedsPrefetchLimit || + ctx.suspensionTracker.pendingCount > ctx.pendingSuspensionTripLimit + ) { + // resumeWrite may legitimately send WriteDone here, not just + // WriteNeedsMore, and records that so nothing tries to resume an + // already-finished coroutine again later. + ctx.resumeWrite(MoreTreeAvailable) + } + } +} + /** * The unparser used for an element that has inputValueCalc. * @@ -126,7 +164,8 @@ sealed abstract class ElementUnparserBase( val eReptypeUnparser: Maybe[Unparser] ) extends CombinatorUnparser(erd) with RepMoveMixin - with ElementUnparserStartEndStrategy { + with ElementUnparserStartEndStrategy + with WriteUnparser { final override def childProcessors = (eBeforeUnparser.toList ++ eUnparser.toList ++ eAfterUnparser.toList ++ eReptypeUnparser.toList ++ setVarUnparsers.toList).toVector @@ -161,21 +200,104 @@ sealed abstract class ElementUnparserBase( } } - protected def doBeforeContentUnparser(state: UState): Unit = { + private[runtime1] def doBeforeContentUnparser(state: UState): Unit = { if (eBeforeUnparser.isDefined) eBeforeUnparser.get.unparse1(state) } - protected def doAfterContentUnparser(state: UState): Unit = { + private[runtime1] def doAfterContentUnparser(state: UState): Unit = { if (eAfterUnparser.isDefined) eAfterUnparser.get.unparse1(state) } - protected def runContentUnparser(state: UState): Unit = { - if (eReptypeUnparser.isDefined) { - eReptypeUnparser.get.unparse1(state) - } else if (eUnparser.isDefined) - eUnparser.get.unparse1(state) + /** + * Registers whatever suspensions this element's content depends on + * (dfdl:length, dfdl:outputValueCalc) before content bytes get + * written. No-op by default; overridden by the SpecifiedLength + * unparsers. Called unconditionally from writeContent too, before + * its group-unparser check, so this isn't skipped for + * delimited/escape-scheme-wrapped elements. + */ + private[runtime1] def contentSetup(state: UState): Unit = () + + private[runtime1] def dispatchContentUnparser(state: UState): Unit = { + (eReptypeUnparser.toOption, eUnparser.toOption) match { + case (Some(rep), _) => + rep.unparse1(state) + case (None, Some(eu)) => + eu.unparse1(state) + case _ => // nothing to do: no content unparser applies + } + } + + private[runtime1] def runContentUnparser(state: UState): Unit = { + contentSetup(state) + dispatchContentUnparser(state) + } + + /** + * Writes this element's content against an already-built + * `containerNode`, without consuming any InfosetInputter events: + * dispatches to whatever `eUnparser` turns out to be, a group + * unparser or a plain simple-element value-writer. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = { + state.currentInfosetNodeStack.push(One(containerNode)) + state.childIndexStack.push(0L) + try { + // writeContent is only ever called from write's side, so this and + // computeSetVariables below always run here, at write's own + // document-order position. + captureRuntimeValuedExpressionValues(state) + doBeforeContentUnparser(state) + // contentSetup can suspend, and suspending reads state.processor. + // That's normally set by the ordinary unparse dispatch, which this + // call bypasses entirely, so it's set explicitly here to match. + state.setProcessor(this) + // Must run before the dispatch below: a group-wrapped eUnparser + // (delimiter stack, escape scheme, specified-length prefix) + // delegates straight to its own writeContent and never reaches + // dispatchContentUnparser, which would otherwise run this instead. + contentSetup(state) + // eReptypeUnparser takes priority here too, matching + // dispatchContentUnparser: a repType'd element's raw eUnparser can + // itself be group-wrapped, and without this check would wrongly + // delegate to that raw content instead of converting via repType. + eUnparser.toOption match { + case Some(wu: WriteUnparser) if eReptypeUnparser.isEmpty => + wu.writeContent(containerNode, state) + case _ => + dispatchContentUnparser(state) + } + // The after-content (padding/fill) region depends on the content + // having been written, so it must run only once the (possibly + // nested) content dispatch above has fully returned; true for both + // the simple-element and group-content cases. + doAfterContentUnparser(state) + computeSetVariables(state) + // Only a simple node needs finalizing here (complex/array nodes were + // already finalized in build's unparseEnd; re-finalizing trips + // setFinal()'s !isFinal assert). An OVC node may still be valueless + // here, so check hasValue rather than assert it. + if (containerNode.isSimple && !containerNode.isFinal && containerNode.asSimple.hasValue) + containerNode.setFinal() + // Write-side half of the shared lead counter, matching build's own + // once-per-node increment; no-op unless a write-side UState has + // opted in via setSharedContext. + if (state.sharedContext.isDefined) state.sharedContext.get.decrementLead() + // Content/value-length suspensions can only unblock once write has + // actually written the bytes, so this must run here too, not just + // build's unparseEnd, or they pile up unresolved until the final + // isFinal=true call instead of resolving as data becomes available. + state.asInstanceOf[SuspensionCapableUState].evalSuspensions(isFinal = false) + } finally { + // A stall (AwaitChildStalledException) or other exception mid-recursion + // must still unwind this node's own push, or these stacks end up + // unbalanced and later invariant checks fail with a confusing internal + // assertion instead of the real diagnostic. + state.childIndexStack.pop() + state.currentInfosetNodeStack.pop + } } override def unparse(state: UState): Unit = { @@ -188,11 +310,8 @@ sealed abstract class ElementUnparserBase( doBeforeContentUnparser(state) - // - // We must push the TermRuntimeData for all model-groups. - // The starting point for this is the model-group of a complex type. - // Simple types don't have model groups, so no pushing those. - // + // We must push the TermRuntimeData for all model-groups, starting from + // the complex type's model-group; simple types have none to push. if (erd.isComplexType) state.pushTRD(erd.optComplexTypeModelGroupRuntimeData.get) @@ -205,7 +324,7 @@ sealed abstract class ElementUnparserBase( computeSetVariables(state) - unparseEnd(state) + unparseEnd(state, isBuild = false) if (state.dataProc.isDefined) state.dataProc.value.endElement(state, this) @@ -298,11 +417,10 @@ class ElementSpecifiedLengthUnparser( override val runtimeDependencies = maybeTargetLengthEv.toArray - override def runContentUnparser(state: UState): Unit = { - computeTargetLength( - state - ) // must happen before run() so that we can take advantage of knowing the length - super.runContentUnparser(state) // setup unparsing, which will block for no valu + // Must happen before dispatchContentUnparser so we can take + // advantage of knowing the length. + override private[runtime1] def contentSetup(state: UState): Unit = { + computeTargetLength(state) } } @@ -324,12 +442,10 @@ class ElementOVCSpecifiedLengthUnparserSuspendableExpression( val diSimple = state.currentInfosetNode.asSimple diSimple.setDataValue(v) - - // - // These are now done in the main unparse, but they will - // suspend if they cannot be evaluated because there is not data value yet. - // - // callingUnparser.computeSetVariables(state) + // Do NOT setFinal here: a chained retry for this value's own + // conversion may still be pending, and finalizing now would trip + // that retry's own not-yet-final assertion; writeContent's own + // hasValue-guarded setFinal covers the synchronous common case. } override protected def maybeKnownLengthInBits(ustate: UState): MaybeULong = MaybeULong(0L) @@ -362,12 +478,14 @@ class ElementOVCSpecifiedLengthUnparser( Assert.invariant(context.dpathElementCompileInfo.isOutputValueCalc) - override def runContentUnparser(state: UState): Unit = { - computeTargetLength( - state - ) // must happen before run() so that we can take advantage of knowing the length - suspendableExpression.run(state) // run the expression. It might or might not have a value. - super.runContentUnparser(state) // setup unparsing, which will block for no valu + override private[runtime1] def contentSetup(state: UState): Unit = { + // Must happen before dispatchContentUnparser so we can take + // advantage of knowing the length. + computeTargetLength(state) + if (!state.currentInfosetNode.asSimple.hasValue) { + // run the expression. It might or might not have a value. + suspendableExpression.run(state) + } } } @@ -381,12 +499,25 @@ sealed trait ElementUnparserStartEndStrategy { * Consumes the required infoset events and changes context so that the * element's DIElement node is the context element. */ - protected def unparseBegin(state: UState): Unit + def unparseBegin(state: UState): Unit /** - * Restores prior context. Consumes end-element event. + * Restores prior context. Consumes end-element event. A freshly-built + * simple node has no value yet (write still has to set it), so isBuild + * defers finalizing it; a complex/array node is always final here. + */ + def unparseEnd(state: UState, isBuild: Boolean): Unit + + /** + * The Builder tree's entry points: same node-creation logic as + * unparseBegin/unparseEnd, plus the build-side lead-counter hookup that + * only ever applies on this side. */ - protected def unparseEnd(state: UState): Unit + final def unparseBeginForBuild(state: UState): Unit = { + unparseBegin(state) + BuildWriteLeadHookup.afterNodeAdded(state) + } + final def unparseEndForBuild(state: UState): Unit = unparseEnd(state, isBuild = true) protected def captureRuntimeValuedExpressionValues(ustate: UState): Unit @@ -403,7 +534,7 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart * Consumes the required infoset events and changes context so that the * element's DIElement node is the context element. */ - final override protected def unparseBegin(state: UState): Unit = { + final override def unparseBegin(state: UState): Unit = { if (erd.isQuasiElement) { // Quasi elements are used for RepType and PrefixedLength, and have no corresponding // events in the infoset inputter. The parent parser will push a DIElement for us to @@ -490,7 +621,7 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart /** * Restores prior context. Consumes end-element event. */ - final override protected def unparseEnd(state: UState): Unit = { + final override def unparseEnd(state: UState, isBuild: Boolean): Unit = { if (erd.isQuasiElement) { // Quasi elements are used for TypeValueCalc, and have no corresponding events in the infoset inputter // The parent parser will handle pushing and poping the Infoset, so we do not need to do anything here. @@ -530,15 +661,14 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart } } - // cur is finished, mark it as final and free if possible. Note that we - // need the container and not the parent of the current element to free - // it. This way if this element is in an array, we free this element - // from the array. We also do not set hidden IVC elements as - // final--although we allow hidden IVC elements when unparsing, they - // never get a value so we can't set them as final without breaking - // assertions. Nothing can access hidden IVC elements, so this should - // not break anything - if (!state.withinHiddenNest || erd.isRepresented) cur.setFinal() + // cur is finished: mark it final and free via its container (not + // parent, so an array-member frees from the array), except hidden + // IVC elements (never get a value) and, for build, SIMPLE elements + // (write still needs to set their actual value afterward). + if ( + (!state.withinHiddenNest || erd.isRepresented) && + !(isBuild && cur.isSimple) + ) cur.setFinal() val curContainer = if (cur.erd.isArray) cur.diParent.maybeLastChild.get else cur.diParent @@ -558,7 +688,7 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart move(state) - state.asInstanceOf[UStateMain].evalSuspensions(isFinal = false) + state.asInstanceOf[SuspensionCapableUState].evalSuspensions(isFinal = false) } } @@ -573,7 +703,7 @@ trait OVCStartEndStrategy extends ElementUnparserStartEndStrategy { /** * For OVC, the behavior w.r.t. consuming infoset events is different. */ - protected final override def unparseBegin(state: UState): Unit = { + final override def unparseBegin(state: UState): Unit = { val ovcElem = if (!state.withinHiddenNest) { // outputValueCalc elements are optional in the infoset. If the next event @@ -628,7 +758,7 @@ trait OVCStartEndStrategy extends ElementUnparserStartEndStrategy { state.currentInfosetNodeStack.push(One(ovcElem)) } - protected final override def unparseEnd(state: UState): Unit = { + final override def unparseEnd(state: UState, isBuild: Boolean): Unit = { // if an OVC element existed, the start AND end events were consumed in // unparseBegin. No need to advance the cursor here. diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala index 070e787290..1a4eff66a9 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala @@ -17,6 +17,7 @@ package org.apache.daffodil.unparsers.runtime1 +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.ModelGroupRuntimeData import org.apache.daffodil.runtime1.processors.unparsers.* @@ -27,12 +28,26 @@ import org.apache.daffodil.runtime1.processors.unparsers.* * we unwind from the refs, we'll decrement. */ class HiddenGroupCombinatorUnparser(ctxt: ModelGroupRuntimeData, bodyUnparser: Unparser) - extends CombinatorUnparser(ctxt) { + extends CombinatorUnparser(ctxt) + with WriteUnparser { override def childProcessors = Vector(bodyUnparser) override val runtimeDependencies = Array() + // The hidden-depth counter must stay incremented across any pauses, since + // anything the body writes needs it (e.g. choice-branch resolution + // branches on state.withinHiddenNest, and RepType conversion asserts + // it's never true). + override def writeContent(containerNode: DINode, start: UState): Unit = + writeWithPushPop( + containerNode, + bodyUnparser, + start, + setup = _.incrementHiddenDef(), + teardown = (start, _) => start.decrementHiddenDef() + ) + def unparse(start: UState): Unit = { try { start.incrementHiddenDef() diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala index 3856c0b272..373c56b94d 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala @@ -17,6 +17,7 @@ package org.apache.daffodil.unparsers.runtime1 +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.layers.LayerDriver import org.apache.daffodil.runtime1.processors.SequenceRuntimeData import org.apache.daffodil.runtime1.processors.unparsers.* @@ -28,8 +29,40 @@ class LayeredSequenceUnparser( override def nom = "LayeredSequence" + private def handleLayerThrowable(layerDriver: LayerDriver, t: Throwable): Unit = { + if (layerDriver ne null) { + layerDriver.handleThrowable(t) + } else { + LayerDriver.handleThrowableWithoutLayer(t) + } + } + + // Same setup/teardown as unparse() below, via withLayerTransform. Without + // this override, write-side dispatch would treat this as a plain + // WriteUnparser (inherited), bypassing the layer transform entirely: raw + // bytes, no compression/checksum, no layer error handling. + override def writeContent(containerNode: DINode, state: UState): Unit = { + // Needed for the same reason as ChoiceCombinatorUnparser's writeContent: + // setFinished/cloneForSuspension reach state.bitOrder/state.processor, + // normally set by Unparser.unparse1's wrapper, which this + // recursive-dispatch code bypasses. + state.setProcessor(LayeredSequenceUnparser.this) + withLayerTransform(state) { + LayeredSequenceUnparser.super.writeContent(containerNode, state) + } + } + override def unparse(state: UState): Unit = { + withLayerTransform(state) { + super.unparse(state) + } + } + // Splits off a buffered DOS for the layer to flush through, so fragment + // bits/bitOrder on the original DOS can't affect how the layer flushes + // bytes, then runs the layer driver's transform around runBody. The + // `finally` restoration ensures later writes always reach `layerFollowingDOS`, win or lose. + private def withLayerTransform(state: UState)(runBody: => Unit): Unit = { val originalDOS = state.getDataOutputStream // create a new buffered DOS that this layer will flush to when the layer @@ -49,7 +82,8 @@ class LayeredSequenceUnparser( // TODO: we're not unparsing here, just writing bytes, so perhaps we do not // need this cloned state? Everything in layers is byte-centric, so there is // no issue of fragment bytes. - val formatInfoPre = state.asInstanceOf[UStateMain].cloneForSuspension(layerUnderlyingDOS) + val formatInfoPre = + state.asInstanceOf[SuspensionCapableUState].cloneForSuspension(layerUnderlyingDOS) // mark the original DOS as finished--no more data will be unparsed to it. // If known, this will carry bit position forward to the layerUnderlyingDOS, @@ -74,7 +108,7 @@ class LayeredSequenceUnparser( // unparse the layer body into layerDOS state.setDataOutputStream(layerDOS) - super.unparse(state) + runBody // now we're done unparsing the layer recursively. // While doing that unparsing, the data output stream may have been split, so the // DOS in the state may no longer be the layerDOS. @@ -85,7 +119,7 @@ class LayeredSequenceUnparser( // val endOfLayerUnparseDOS = state.getDataOutputStream val formatInfoPost = - state.asInstanceOf[UStateMain].cloneForSuspension(endOfLayerUnparseDOS) + state.asInstanceOf[SuspensionCapableUState].cloneForSuspension(endOfLayerUnparseDOS) // setFinished on this end-of-layer-unparse data-output-stream ensures // that the layerDOS gets close() called on it. @@ -95,8 +129,13 @@ class LayeredSequenceUnparser( // layer stack is potentially still needed, so // nothing can be cleaned up at this point. } catch { - case t: Throwable if (layerDriver ne null) => layerDriver.handleThrowable(t) - case t: Throwable => LayerDriver.handleThrowableWithoutLayer(t) + // Pure write-side control-flow signals, unrelated to the layer + // itself; rewrapping either would defeat write's own handling + // (a stall diagnostic, or build's abort cleanup) with a raw + // "layer failed" exception. + case e: AwaitChildStalledException => throw e + case e: BuildAbortedException => throw e + case t: Throwable => handleLayerThrowable(layerDriver, t) // otherwise we have no layer driver, so we were unable to load the layer. // just let that propagate. } finally { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala index 1d31e39d1e..851b153561 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala @@ -19,6 +19,7 @@ package org.apache.daffodil.unparsers.runtime1 import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.unparsers.* @@ -52,7 +53,8 @@ case class ComplexNilOrContentUnparser( ctxt: ElementRuntimeData, nilUnparser: Unparser, contentUnparser: Unparser -) extends CombinatorUnparser(ctxt) { +) extends CombinatorUnparser(ctxt) + with WriteUnparser { override val runtimeDependencies = Array() @@ -66,4 +68,17 @@ case class ComplexNilOrContentUnparser( else contentUnparser.unparse1(state) } + + // Without this override, WriteUnparser dispatch would call + // contentUnparser.unparse1 synchronously on write's state, but it can + // itself be a resumable group unparser expecting live InfosetInputter + // events that don't exist yet (see SpecifiedLengthExplicitImplicitUnparser). + override def writeContent(containerNode: DINode, state: UState): Unit = { + val bodyUnparser = if (containerNode.asComplex.isNilled) { + nilUnparser + } else { + contentUnparser + } + writeWithPushPop(containerNode, bodyUnparser, state) + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala index 7c451f1ddb..e20d040c92 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala @@ -26,6 +26,9 @@ import org.apache.daffodil.lib.schema.annotation.props.gen.SeparatorPosition import org.apache.daffodil.lib.schema.annotation.props.gen.SeparatorPosition.* import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.MaybeInt +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.runtime1.infoset.DIComplex +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.ModelGroupRuntimeData import org.apache.daffodil.runtime1.processors.SequenceRuntimeData @@ -120,7 +123,8 @@ class OrderedSeparatedSequenceUnparser( sepMtaUnparserMaybe: Maybe[Unparser], sep: Unparser, childUnparsers: Array[SequenceChildUnparser with Separated] -) extends OrderedSequenceUnparserBase(rd) { +) extends OrderedSequenceUnparserBase(rd) + with WriteUnparser { // Sequences of nothing (no initiator, no terminator, nothing at all) should // have been optimized away Assert.invariant(childUnparsers.length > 0) @@ -129,6 +133,428 @@ class OrderedSeparatedSequenceUnparser( override def childProcessors = childUnparsers.toVector + /** + * In-flight separator-suppression state for one writeContent call: + * whether any term has already written something (`wroteAny`), whichever + * separator is currently deferred pending its term's content + * (`pendingPostfixSeparatorAtTerm` for ssp Never, + * `pendingSuppressibleOp` for AnyEmpty/TrailingEmpty(Strict)), and the + * end-of-sequence TrailingEmpty(Strict) queue (`trailingSuspendedOps`). + */ + private class SeparatorSuppressionState(state: UState) { + + private var wroteAny = false + + // for spos == Postfix, the separator for a represented term must come + // AFTER its content is written, not before. Set by beforeSeparator, + // cleared by afterSeparator once that term's content has been written. + private var pendingPostfixSeparatorAtTerm = false + + // For ssp AnyEmpty/TrailingEmpty(Strict): the in-flight suspension + // beforeSeparator started for the current term, completed by + // afterSeparator. Mirrors pendingPostfixSeparatorAtTerm's role for ssp + // Never, but as a suspension object rather than a flag. + private var pendingSuppressibleOp: SuppressableSeparatorUnparserSuspendableOperation = null + + // For ssp TrailingEmpty/TrailingEmptyStrict: separators deferred until + // the sequence's own end (DFDL requires trailing through the whole + // sequence, not just the local group), matching + // unparseWithSuppression's end-of-loop resolution. + private val trailingSuspendedOps = + scala.collection.mutable.Buffer[SuppressableSeparatorUnparserSuspendableOperation]() + + /** + * ssp=never never omits separators based on content, so a term with + * fewer than maxOccurs actual occurrences still gets one separator per + * missing occurrence. Other policies decide via content length + * instead; irrelevant here. + */ + def writeExtraSeparatorsIfNeverSuppressed( + rep: RepeatingChildUnparser, + numOccurrences: Long + ): Unit = { + if (ssp != Never) return + // Uses erd.maxOccurs, not rep.maxRepeats(state): for + // occursCountKind="expression"/"parsed", maxRepeats(state) is + // Long.MaxValue, which would loop until OOM. An unbounded array's + // maxOccurs is -1, so sepsNeeded goes negative and this loop no-ops. + val sepsNeeded = rep.erd.maxOccurs - numOccurrences + if (sepsNeeded <= 0) return + val numExtraSeps = if ((spos eq Infix) && !wroteAny) { + sepsNeeded - 1 + } else { + sepsNeeded + } + var n = numExtraSeps + while (n > 0) { + unparseJustSeparator(state) + n -= 1 + } + // Deliberately NOT wroteAny = true: this never advances + // state.groupPos, so a following Infix term must still see this as + // unwritten, or it would wrongly get its own separator too. + } + + /** + * occursCountKind="implicit" is positional, so a non-trailing bounded + * array/optional still needs speculative missing-occurrence separators, + * or a later term shifts position. stacksAlreadyPushed: true only right + * after an actual occurrence loop already positioned the stacks. + */ + def writePositionallyRequiredSepsIfSuppressed( + rep: RepeatingChildUnparser with Separated, + numOccurrences: Long, + stacksAlreadyPushed: Boolean + ): Unit = { + if (ssp == Never) return + if ( + (rep.ock ne OccursCountKind.Implicit) || + !rep.isPositional || !rep.isBoundedMax || + (rep.isDeclaredLast && rep.isPotentiallyTrailing) + ) return + // safe: isBoundedMax being true guarantees maxRepeats(state) is the + // actual, finite erd.maxOccurs, never Long.MaxValue. + val maxReps = rep.maxRepeats(state) + if (numOccurrences >= maxReps) return + if (!stacksAlreadyPushed) { + state.pushOccurrenceIndices() + } + var n = numOccurrences + while (n < maxReps) { + beforeSeparator(rep.erd, rep.isKnownStaticallyNotToSuppressSeparator) + afterSeparator() + n += 1 + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + } + if (!stacksAlreadyPushed) { + state.popOccurrenceIndices() + } + } + + /** + * Called before a term/occurrence's content is written, once known + * to actually be written. For ssp Never (or staticallyNotSuppressible), + * Prefix/Infix write immediately and Postfix defers; otherwise + * Prefix/Infix speculatively unparse a suppressible separator. + */ + def beforeSeparator( + trd: TermRuntimeData, + staticallyNotSuppressible: Boolean + ): Unit = { + if (staticallyNotSuppressible || (ssp eq Never)) { + spos match { + case Prefix => unparseJustSeparator(state) + case Infix => if (wroteAny) unparseJustSeparator(state) + case Postfix => pendingPostfixSeparatorAtTerm = true + } + wroteAny = true + return + } + ssp match { + case Never => + Assert.invariantFailed("handled above") + case AnyEmpty | TrailingEmpty | TrailingEmptyStrict => + spos match { + case Prefix | Infix => + if ((spos eq Infix) && !wroteAny) { + // no separator possible; hence, no suppression + } else { + val suspendableOp = + new SuppressableSeparatorUnparserSuspendableOperation( + sepMtaAlignmentMaybe, + sep, + trd + ) + val suppressableSep = SuppressableSeparatorUnparser(sep, trd, suspendableOp) + suppressableSep.unparse1(state) + pendingSuppressibleOp = suspendableOp + } + case Postfix => + val suspendableOp = + new SuppressableSeparatorUnparserSuspendableOperation( + sepMtaAlignmentMaybe, + sep, + trd + ) + suspendableOp.captureDOSForStartOfSeparatedRegionBeforePostfixSeparator(state) + pendingSuppressibleOp = suspendableOp + } + } + wroteAny = true + } + + /** + * Completes whatever beforeSeparator started for a term. Handles ssp + * Never (pendingPostfixSeparatorAtTerm, immediate write) and + * AnyEmpty/TrailingEmpty(Strict) (pendingSuppressibleOp); exactly one + * is ever set, matching beforeSeparator's own ssp branch. + */ + def afterSeparator(): Unit = { + if (pendingPostfixSeparatorAtTerm) { + pendingPostfixSeparatorAtTerm = false + unparseJustSeparator(state) + } + if (pendingSuppressibleOp ne null) { + val op = pendingSuppressibleOp + pendingSuppressibleOp = null + ssp match { + case AnyEmpty => + spos match { + case Prefix | Infix => + op.captureStateAtEndOfPotentiallyZeroLengthRegionFollowingTheSeparator(state) + case Postfix => + op.captureDOSForEndOfSeparatedRegionBeforePostfixSeparator(state) + SuppressableSeparatorUnparser(sep, op.rd, op).unparse1(state) + op.captureStateAtEndOfPotentiallyZeroLengthRegionFollowingTheSeparator(state) + } + case TrailingEmpty | TrailingEmptyStrict => + spos match { + case Prefix | Infix => + trailingSuspendedOps += op + case Postfix => + op.captureDOSForEndOfSeparatedRegionBeforePostfixSeparator(state) + SuppressableSeparatorUnparser(sep, op.rd, op).unparse1(state) + trailingSuspendedOps += op + } + case Never => + Assert.invariantFailed("pendingSuppressibleOp should never be set for ssp Never") + } + } + } + + // The term produced zero occurrences (a different term's child took + // this position, or build finished with nothing more coming); no + // tree node was added, so position doesn't move, but it may still owe + // separators, like an actual occurrence loop's own end-of-term handling. + def zeroOccurrences(rep: RepeatingChildUnparser with Separated): Unit = { + if (ssp eq Never) { + writeExtraSeparatorsIfNeverSuppressed(rep, 0) + } else { + writePositionallyRequiredSepsIfSuppressed(rep, 0, stacksAlreadyPushed = false) + } + } + + // ssp TrailingEmpty(Strict): now that nothing at all remains in this + // sequence, every deferred separator's "after" boundary is this exact + // point. Resolve them all. + def resolveTrailingSuspended(): Unit = { + if ((ssp eq TrailingEmpty) || (ssp eq TrailingEmptyStrict)) { + trailingSuspendedOps.foreach { + _.captureStateAtEndOfPotentiallyZeroLengthRegionFollowingTheSeparator(state) + } + trailingSuspendedOps.clear() + } + } + } + + /** + * Walks childUnparsers positionally against an already-built + * containerNode, writing each term's content/separator per spos; blocks + * via childExistsOrFinal/awaitChild wherever a needed child isn't ready. + * childIndexStack.top tracks position (an array is ONE slot). + */ + override def writeContent(containerNode: DINode, state: UState): Unit = { + val sharedCtx = state.sharedContext.get + val complex = containerNode.asComplex + val sepState = new SeparatorSuppressionState(state) + + var index = 0 + while (index < childUnparsers.length) { + childUnparsers(index) match { + case rep: RepeatingChildUnparser => + writeRepeatingTerm(rep, complex, sharedCtx, state, sepState) + case cu => + writeRequiredTerm(cu, complex, sharedCtx, state, sepState) + } + index += 1 + } + + sepState.resolveTrailingSuspended() + } + + /** + * Writes an array/optional term: either its full occurrence loop (an + * actual `DIArray`, or a scalar optional's single occurrence) or, if the + * term produced no occurrences at all, its zero-occurrences separator + * bookkeeping. + */ + private def writeRepeatingTerm( + rep: RepeatingChildUnparser, + complex: DIComplex, + sharedCtx: UnparseSharedContext, + state: UState, + sepState: SeparatorSuppressionState + ): Unit = { + val idx = state.childIndexStack.top.toInt + if (sharedCtx.childExistsOrFinal(complex, idx)) { + val next = complex.child(idx) + if (next.erd eq rep.erd) { + next match { + case arrayNode: DIArray => + val repSep = rep.asInstanceOf[RepeatingChildUnparser with Separated] + // A dfdl:occursIndex() expression in the occurrence's content + // reads state.occursIndexStack.top, which must track it or + // every occurrence would evaluate as if it were the first. + state.pushOccurrenceIndices() + try { + var arrayOcc = 0 + while (sharedCtx.childExistsOrFinal(arrayNode, arrayOcc)) { + val occNode = sharedCtx.awaitChild(arrayNode, arrayOcc) + // Postfix's separator comes after this occurrence's + // content; afterSeparator runs once it's written. + sepState.beforeSeparator( + repSep.erd, + repSep.isKnownStaticallyNotToSuppressSeparator + ) + repSep.childUnparser + .asInstanceOf[ElementUnparserBase] + .writeContent(occNode, state) + sepState.afterSeparator() + arrayNode.freeChildIfNoLongerNeeded(arrayOcc, state.releaseUnneededInfoset) + arrayOcc += 1 + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + } + // this whole array term is done; it occupies exactly one + // slot among containerNode's own children, regardless of + // how many occurrences it held + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + state.moveOverOneElementChildOnly() + if (ssp eq Never) { + sepState.writeExtraSeparatorsIfNeverSuppressed(repSep, arrayOcc) + } else { + sepState.writePositionallyRequiredSepsIfSuppressed( + repSep, + arrayOcc, + stacksAlreadyPushed = true + ) + } + } finally { + state.popOccurrenceIndices() + } + case scalarOptional => + val repSep = rep.asInstanceOf[RepeatingChildUnparser with Separated] + val readyChild = sharedCtx.awaitChild(complex, idx) + // Same push as a true array's entry (see above); a + // scalar optional is still a RepeatingChildUnparser (just + // one that can never hold more than one occurrence). + state.pushOccurrenceIndices() + try { + // Postfix's separator comes after this element's + // content; afterSeparator runs once it's written. + sepState.beforeSeparator( + repSep.erd, + repSep.isKnownStaticallyNotToSuppressSeparator + ) + repSep.childUnparser + .asInstanceOf[ElementUnparserBase] + .writeContent(readyChild, state) + sepState.afterSeparator() + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + // Matches the DIArray arm's own per-occurrence advancement: + // occursIndexStack.top must reflect the next (missing) + // occurrence's index before writePositionallyRequiredSepsIfSuppressed's + // loop runs, or its first separator would see index 1, not 2. + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + if (ssp eq Never) { + sepState.writeExtraSeparatorsIfNeverSuppressed(repSep, 1) + } else { + sepState.writePositionallyRequiredSepsIfSuppressed( + repSep, + 1, + stacksAlreadyPushed = true + ) + } + } finally { + state.popOccurrenceIndices() + } + } + } else { + // this term produced zero occurrences: a different term's + // child appeared in this tree position instead + sepState.zeroOccurrences(rep.asInstanceOf[RepeatingChildUnparser with Separated]) + } + } else { + // no more children will ever come; this optional/array term is absent + sepState.zeroOccurrences(rep.asInstanceOf[RepeatingChildUnparser with Separated]) + } + } + + /** + * Writes a required term (exactly one occurrence): a simple/complex + * element, a nested bare group, or a statement-only term with no tree + * child of its own. Includes its own separator bookkeeping if represented. + */ + private def writeRequiredTerm( + cu: SequenceChildUnparser with Separated, + complex: DIComplex, + sharedCtx: UnparseSharedContext, + state: UState, + sepState: SeparatorSuppressionState + ): Unit = { + cu.childUnparser match { + case nvi: NewVariableInstanceStartUnparser => + nvi.unparse1(state) + case sv: SetVariableUnparser => + sv.unparse1(state) + case nvi: NewVariableInstanceEndUnparser => + nvi.unparse1(state) + case align: AlignmentPrimUnparser => + // Padding has no non-idempotent side effect (unlike assert/discriminator + // below), so re-running it here is required, not forbidden: build's + // own recursion only ran it against build's no-op-sink DOS. Always + // represented, and never suppressible since its content is deterministic. + if (cu.trd.isRepresented) { + sepState.beforeSeparator(cu.trd, staticallyNotSuppressible = true) + } + align.unparse1(state) + if (cu.trd.isRepresented) { + sepState.afterSeparator() + } + case statementOnly if !statementOnly.isInstanceOf[WriteUnparser] => + // No tree child, so no separator, and nothing to wait on. + () + case _ => + val idx = state.childIndexStack.top.toInt + // A nested bare group (e.g. a choice) may resolve to a branch with no + // infoset footprint at all; once build is done and no child showed up, + // let the group's own dispatch decide. A plain element term always + // needs an actual child, so it always waits unconditionally. + val isGroupTerm = !cu.childUnparser.isInstanceOf[ElementUnparserBase] && + cu.childUnparser.isInstanceOf[WriteUnparser] + if (isGroupTerm) { + if (sharedCtx.childExistsOrFinal(complex, idx)) { + sharedCtx.awaitChild(complex, idx) + } + } else { + sharedCtx.awaitChild(complex, idx) + } + // A non-represented term gets no separator, so wroteAny must not + // flip true. Suppression only applies to terms whose presence is + // uncertain (array/optional, or a bare group not statically known + // to need it), never to a plain element's own content length. + val useSuppression = isGroupTerm && !cu.isKnownStaticallyNotToSuppressSeparator + if (cu.trd.isRepresented) { + sepState.beforeSeparator(cu.trd, staticallyNotSuppressible = !useSuppression) + } + cu.childUnparser match { + case elemUnp: ElementUnparserBase => + elemUnp.writeContent(complex.child(idx), state) + sepState.afterSeparator() + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + case wu: WriteUnparser => + wu.writeContent(complex, state) + sepState.afterSeparator() + case other => + Assert.usageError(s"unhandled sequence term unparser type: $other") + } + } + } + /** * Unparses one occurrence with associated separator (non-suppressable). */ @@ -315,8 +741,7 @@ class OrderedSeparatedSequenceUnparser( val zlDetector = childUnparser.zeroLengthDetector childUnparser match { case unparser: RepOrderedSeparatedSequenceChildUnparser => { - state.arrayIterationIndexStack.push(1L) - state.occursIndexStack.push(1L) + state.pushOccurrenceIndices() val erd = unparser.erd var numOccurrences = 0 val maxReps = unparser.maxRepeats(state) @@ -476,8 +901,7 @@ class OrderedSeparatedSequenceUnparser( // no event (state.inspect returned false) Assert.invariantFailed("No event for unparsing.") } - state.arrayIterationIndexStack.pop() - state.occursIndexStack.pop() + state.popOccurrenceIndices() } case scalarUnparser => trd match { @@ -606,8 +1030,7 @@ class OrderedSeparatedSequenceUnparser( // childUnparser match { case unparser: RepOrderedSeparatedSequenceChildUnparser => { - state.arrayIterationIndexStack.push(1L) - state.occursIndexStack.push(1L) + state.pushOccurrenceIndices() val erd = unparser.erd Assert.invariant(erd.isArray || erd.isOptional) Assert.invariant(erd.isRepresented) // arrays/optionals cannot have inputValueCalc @@ -698,8 +1121,7 @@ class OrderedSeparatedSequenceUnparser( unparser.endArrayOrOptional(erd, state) } - state.arrayIterationIndexStack.pop() - state.occursIndexStack.pop() + state.popOccurrenceIndices() } case scalarUnparser => { unparseOne(scalarUnparser, trd, state) diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala index 0f9b8c0eae..b173d8cda3 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala @@ -22,6 +22,7 @@ import org.apache.daffodil.lib.schema.annotation.props.gen.LengthUnits import org.apache.daffodil.lib.schema.annotation.props.gen.Representation import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.runtime1.infoset.DIElement +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.infoset.DISimple import org.apache.daffodil.runtime1.infoset.Infoset import org.apache.daffodil.runtime1.processors.CharsetEv @@ -35,7 +36,8 @@ final class SpecifiedLengthExplicitImplicitUnparser( eUnparser: Unparser, erd: ElementRuntimeData, targetLengthInBitsEv: UnparseTargetLengthInBitsEv -) extends CombinatorUnparser(erd) { +) extends CombinatorUnparser(erd) + with WriteUnparser { override val runtimeDependencies = Array() @@ -52,7 +54,7 @@ final class SpecifiedLengthExplicitImplicitUnparser( dcs } - override final def unparse(state: UState): Unit = { + private def checkVariableWidthComplexType(state: UState): Unit = { lazy val dcs = getCharset(state) if ( erd.impliedRepresentation == Representation.Text && @@ -66,10 +68,26 @@ final class SpecifiedLengthExplicitImplicitUnparser( lengthKind.toString, lengthUnits.toString ) - } else { - eUnparser.unparse1(state) } } + + override final def unparse(state: UState): Unit = { + checkVariableWidthComplexType(state) + eUnparser.unparse1(state) + } + + // Without this, a SeqCompUnparser wrapping this class would treat + // eUnparser as a synchronous call via its generic fallback, but + // eUnparser can itself be a resumable group unparser expecting live + // InfosetInputter events, desyncing build's event stream entirely. + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop( + containerNode, + eUnparser, + state, + setup = checkVariableWidthComplexType, + teardown = (_, _) => () + ) } /** @@ -139,13 +157,39 @@ class SpecifiedLengthPrefixedUnparser( override val lengthUnits: LengthUnits, override val prefixedLengthAdjustmentInUnits: Long ) extends CombinatorUnparser(erd) - with CalculatedPrefixedLengthUnparserMixin { + with CalculatedPrefixedLengthUnparserMixin + with WriteUnparser { override val runtimeDependencies = Array() override def childProcessors = Vector(prefixedLengthUnparser, eUnparser) override def unparse(state: UState): Unit = { + val plElem = pushDetachedPrefixLengthElement(state) + eUnparser.unparse1(state) + resolvePrefixLength(state, state.currentInfosetNode.asInstanceOf[DIElement], plElem) + } + + // Without this, WriteUnparser dispatch (a plain recursive-dispatch + // fallback for a group-wrapped eUnparser) would call eUnparser.unparse1 + // synchronously, but it can itself be a resumable group unparser + // expecting live InfosetInputter events. + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop( + containerNode, + eUnparser, + state, + setup = pushDetachedPrefixLengthElement, + teardown = { (state, plElem) => + // resolvePrefixLength (via assignPrefixLength/suspension.run) + // expects state.processor to already be set, normally done by + // Unparser.unparse1, which this recursive-dispatch path bypasses. + state.setProcessor(SpecifiedLengthPrefixedUnparser.this) + resolvePrefixLength(state, containerNode.asInstanceOf[DIElement], plElem) + } + ) + + private def pushDetachedPrefixLengthElement(state: UState): DISimple = { // Create a "detached" DIDocument with a single child element that the // prefix length will be parsed to. This creates a completely new // infoset and parses to that, so care is taken to ensure this infoset @@ -160,10 +204,10 @@ class SpecifiedLengthPrefixedUnparser( state.currentInfosetNodeStack.push(One(plElem)) prefixedLengthUnparser.unparse1(state) state.currentInfosetNodeStack.pop + plElem + } - val elem = state.currentInfosetNode.asInstanceOf[DIElement] - eUnparser.unparse1(state) - + private def resolvePrefixLength(state: UState, elem: DIElement, plElem: DISimple): Unit = { if (elem.contentLength.maybeLengthInBits().isDefined) { // If we were able to immediately calculate the length of the element, // then just set it as the value of the detached element created above so diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala index 8b4488a884..cf84670fa2 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala @@ -18,6 +18,9 @@ package org.apache.daffodil.unparsers.runtime1 import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.schema.annotation.props.gen.OccursCountKind +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.runtime1.infoset.DIComplex +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.SequenceRuntimeData import org.apache.daffodil.runtime1.processors.TermRuntimeData @@ -56,7 +59,8 @@ class RepOrderedUnseparatedSequenceChildUnparser( class OrderedUnseparatedSequenceUnparser( rd: SequenceRuntimeData, childUnparsers: Array[SequenceChildUnparser] -) extends OrderedSequenceUnparserBase(rd) { +) extends OrderedSequenceUnparserBase(rd) + with WriteUnparser { // Sequences of nothing (no initiator, no terminator, nothing at all) should // have been optimized away @@ -66,6 +70,141 @@ class OrderedUnseparatedSequenceUnparser( override def childProcessors = childUnparsers.toVector + /** + * Walks childUnparsers positionally against an already-built + * containerNode, same shape as a separated sequence but without any + * separator writing. Without a dedicated writeContent override here, + * the outer dispatch's WriteUnparser check would be false, falling + * through to the single-pass-style dispatch path against an untouched + * inputter. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = { + val sharedCtx = state.sharedContext.get + val complex = containerNode.asComplex + + var index = 0 + while (index < childUnparsers.length) { + childUnparsers(index) match { + case rep: RepeatingChildUnparser => + writeRepeatingTerm(rep, complex, sharedCtx, state) + case cu => + writeRequiredTerm(cu, complex, sharedCtx, state) + } + index += 1 + } + } + + /** + * Writes an array/optional term's full occurrence loop, if it produced + * any occurrences; nothing to do otherwise. There's no separator + * bookkeeping to account for either way, since this sequence has none. + */ + private def writeRepeatingTerm( + rep: RepeatingChildUnparser, + complex: DIComplex, + sharedCtx: UnparseSharedContext, + state: UState + ): Unit = { + val idx = state.childIndexStack.top.toInt + if (sharedCtx.childExistsOrFinal(complex, idx)) { + val next = complex.child(idx) + if (next.erd eq rep.erd) { + next match { + case arrayNode: DIArray => + // A dfdl:occursIndex() expression in the occurrence's own + // content reads state.occursIndexStack.top, which must track + // the actual occurrence being written. + state.pushOccurrenceIndices() + try { + var arrayOcc = 0 + while (sharedCtx.childExistsOrFinal(arrayNode, arrayOcc)) { + val occNode = sharedCtx.awaitChild(arrayNode, arrayOcc) + rep.childUnparser + .asInstanceOf[ElementUnparserBase] + .writeContent(occNode, state) + arrayNode.freeChildIfNoLongerNeeded(arrayOcc, state.releaseUnneededInfoset) + arrayOcc += 1 + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + } + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + state.moveOverOneElementChildOnly() + } finally { + state.popOccurrenceIndices() + } + case scalarOptional => + val readyChild = sharedCtx.awaitChild(complex, idx) + // Same reason as the array case above: a dfdl:occursIndex() + // expression in the occurrence's own content reads + // state.occursIndexStack.top. + state.pushOccurrenceIndices() + try { + rep.childUnparser + .asInstanceOf[ElementUnparserBase] + .writeContent(readyChild, state) + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + } finally { + state.popOccurrenceIndices() + } + } + } + // else: this term produced zero occurrences; a different term's + // child appeared in this tree position instead, and there's no + // separator bookkeeping to resolve for it here, unlike the + // separated sequence's own zeroOccurrences. + } + // else: no more children will ever come; this optional/array term is + // absent, again with nothing further to do about it here. + } + + /** + * Writes a required term: a simple or complex element, a nested bare + * group, or a statement-only term with no tree child of its own. + */ + private def writeRequiredTerm( + cu: SequenceChildUnparser, + complex: DIComplex, + sharedCtx: UnparseSharedContext, + state: UState + ): Unit = { + cu.childUnparser match { + // A plain scalar element term: await its one tree child, write + // its content, then advance the element-child position. + case elemUnp: ElementUnparserBase => + val idx = state.childIndexStack.top.toInt + val child = sharedCtx.awaitChild(complex, idx) + elemUnp.writeContent(child, state) + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + case wu: WriteUnparser => + val idx = state.childIndexStack.top.toInt + // This nested group (e.g. a choice) may have resolved to a branch + // with no infoset footprint at all; only wait for a child's + // readiness when childExistsOrFinal says one genuinely exists. + if (sharedCtx.childExistsOrFinal(complex, idx)) { + sharedCtx.awaitChild(complex, idx) + } + wu.writeContent(complex, state) + case nvi: NewVariableInstanceStartUnparser => + nvi.unparse1(state) + case sv: SetVariableUnparser => + sv.unparse1(state) + case nvi: NewVariableInstanceEndUnparser => + nvi.unparse1(state) + case align: AlignmentPrimUnparser => + // Padding has no non-idempotent side effect (unlike assert/discriminator + // below), so re-running it here is required, not forbidden: build's + // own recursion only ran it against build's no-op-sink DOS. + align.unparse1(state) + case _ => + // No tree child, so nothing to wait on. + () + } + } + /** * Unparses one iteration of an array/optional element */ @@ -101,8 +240,7 @@ class OrderedUnseparatedSequenceUnparser( // childUnparser match { case unparser: RepeatingChildUnparser => { - state.arrayIterationIndexStack.push(1L) - state.occursIndexStack.push(1L) + state.pushOccurrenceIndices() val erd = unparser.erd var numOccurrences = 0 val maxReps = unparser.maxRepeats(state) @@ -177,8 +315,7 @@ class OrderedUnseparatedSequenceUnparser( ) } - state.arrayIterationIndexStack.pop() - state.occursIndexStack.pop() + state.popOccurrenceIndices() } // case scalarUnparser => { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala new file mode 100644 index 0000000000..eb60b2aaed --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala @@ -0,0 +1,79 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.unparsers.runtime1 + +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.processors.unparsers.* + +/** + * Write-side dispatch for the build/write-prefetch unparse path, by + * ordinary recursive calls, using `UnparseSharedContext.awaitChild` to + * block (park write's coroutine thread, see `BuildWriteCoroutines.scala`) + * wherever a needed child doesn't yet exist or isn't ready. + */ +trait WriteUnparser { + + // Writes containerNode's content via direct recursive calls on write's + // own coroutine thread - "where we are" is just the JVM call stack, not + // a return-value state machine. May block (via awaitChild) until a + // needed child exists and is ready, then resumes where it left off. + def writeContent(containerNode: DINode, state: UState): Unit + + // Shared push-once/pop-once skeleton: setup runs before recursing into + // bodyUnparser (dispatched to writeContent if it's a WriteUnparser, else + // plain unparse1), teardown runs once that call returns with setup's + // result (e.g. threading a detached element from setup to teardown). + protected def writeWithPushPop[A]( + containerNode: DINode, + bodyUnparser: Unparser, + state: UState, + setup: UState => A, + teardown: (UState, A) => Unit + ): Unit = { + val setupResult = setup(state) + // Unlike single-pass unparse(), a stall here is caught higher up and + // followed by finishWriteSide's invariant checks against this same + // state, so teardown must still run, or those checks fail for an + // unrelated reason. + try { + bodyUnparser match { + case wu: WriteUnparser => wu.writeContent(containerNode, state) + case _ => bodyUnparser.unparse1(state) + } + } finally { + teardown(state, setupResult) + } + } + + /** + * `writeWithPushPop` for a combinator with nothing to push or pop; just + * the dispatch-to-`writeContent`-or-`unparse1` part. + */ + protected def writeWithPushPop( + containerNode: DINode, + bodyUnparser: Unparser, + state: UState + ): Unit = + writeWithPushPop( + containerNode, + bodyUnparser, + state, + (_: UState) => (), + (_: UState, _: Unit) => () + ) +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/core/outputValueCalc/TestOutputValueCalcSuspensionRegressions.scala b/daffodil-core/src/test/scala/org/apache/daffodil/core/outputValueCalc/TestOutputValueCalcSuspensionRegressions.scala new file mode 100644 index 0000000000..527fc5bf8c --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/core/outputValueCalc/TestOutputValueCalcSuspensionRegressions.scala @@ -0,0 +1,230 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.core.outputValueCalc + +import java.nio.channels.Channels + +import org.apache.daffodil.core.compiler.Compiler +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter +import org.apache.daffodil.runtime1.processors.DataProcessor + +import org.junit.Assert.* +import org.junit.Test + +/** + * Regression guards for evalSuspensionQueue's buildResolvableOnly + * restructuring: single-pass unparse doesn't use that mode, but must + * still resolve these scenarios exactly as before. + */ +class TestOutputValueCalcSuspensionRegressions { + + private val example = XMLUtils.EXAMPLE_NAMESPACE + + private val readsOvcRecordCount = 3 + + private def readsOvcSchema = { + SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + } + + private def readsOvcInfoset = { + val recordXml = (1 to readsOvcRecordCount).map { i => + + {f"T$i%03d"} + {f"data$i%04d"} + + } + +
+ {recordXml} + + } + + @Test def testOvcReadsOvcResolvesViaFinalDrain(): Unit = { + val compiler = Compiler().withTunables(Map("unparseSuspensionWaitOld" -> "1000000")) + val pf = compiler.compileNode(readsOvcSchema) + if (pf.isError) fail(pf.getDiagnostics.toString) + val dp = pf.onPath("/").asInstanceOf[DataProcessor] + if (dp.isError) fail(dp.getDiagnostics.toString) + + val outputStream = new java.io.ByteArrayOutputStream() + val out = Channels.newChannel(outputStream) + val inputter = new ScalaXMLInfosetInputter(readsOvcInfoset) + val actual = dp.unparse(inputter, out) + out.close() + assertFalse(actual.getDiagnostics.toString, actual.isProcessingError) + + val unparsed = outputStream.toString + // lenA reads lenB's own value (not its length): both should equal + // record[readsOvcRecordCount]/data's length, 8. + assertEquals(" 8 8", unparsed.substring(0, 8)) + + val recordsPart = unparsed.substring(8) + val expectedRecords = + (1 to readsOvcRecordCount).map(i => f"T$i%03d" + f"data$i%04d").mkString + assertEquals(expectedRecords, recordsPart) + } + + private val pendingRetryRecordCount = 12 + + private def pendingRetrySchema = { + SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + } + + private def pendingRetryInfoset = { + val recordXml = (1 to pendingRetryRecordCount).map { i => + + {f"T$i%03d"} + {f"data$i%04d"} + + } + +
+ {recordXml} + + } + + // Every len field is a fixed-length "data" element (8 bytes), so all + // four should compute to 8 regardless of which record they target. + private def pendingRetryExpectedOutput = { + val header = " 8 8 8 8" + val records = (1 to pendingRetryRecordCount).map(i => f"T$i%03d" + f"data$i%04d").mkString + header + records + } + + @Test def testManyForceRetryCyclesResolveCorrectly(): Unit = { + TestUtils.testUnparsing( + pendingRetrySchema, + pendingRetryInfoset, + pendingRetryExpectedOutput, + tunables = Map("unparseSuspensionWaitOld" -> "1", "unparseSuspensionWaitYoung" -> "1") + ) + } + + @Test def testManyForceRetryCyclesDefaultTunablesStillMatch(): Unit = { + // Same schema at default tunables: confirms the pass above isn't + // vacuous (e.g. a schema mistake unrelated to the tunable). + TestUtils.testUnparsing(pendingRetrySchema, pendingRetryInfoset, pendingRetryExpectedOutput) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/lib/util/TestMStack.scala b/daffodil-core/src/test/scala/org/apache/daffodil/lib/util/TestMStack.scala index b4cebcdbcb..5905e6deb3 100644 --- a/daffodil-core/src/test/scala/org/apache/daffodil/lib/util/TestMStack.scala +++ b/daffodil-core/src/test/scala/org/apache/daffodil/lib/util/TestMStack.scala @@ -17,6 +17,11 @@ package org.apache.daffodil.lib.util +import org.apache.daffodil.lib.util.Maybe.* + +import org.junit.Assert.* +import org.junit.Test + /** * Compare MStack performance to ArrayStack. It should be faster for primitives */ @@ -24,6 +29,107 @@ class TestMStack { var junk: Long = 0 + // Copying into a sized destination must match copying into a default one. + @Test def testMStackOfCopyFromSizedDestinationMatchesDefault(): Unit = { + val source = new MStackOf[String] + source.push("a") + source.push("b") + source.push("c") + + val sizedDest = new MStackOf[String](source.length) + sizedDest.copyFrom(source) + + val defaultDest = new MStackOf[String] + defaultDest.copyFrom(source) + + assertEquals(defaultDest.toList, sizedDest.toList) + assertEquals(3, sizedDest.length) + assertEquals("c", sizedDest.top) + } + + // A sized-to-exact-depth destination must still grow past capacity. + @Test def testMStackOfSizedDestinationStillGrowsCorrectly(): Unit = { + val source = new MStackOf[String] + source.push("a") + source.push("b") + + val sizedDest = new MStackOf[String](source.length) + sizedDest.copyFrom(source) + + sizedDest.push("c") + sizedDest.push("d") + sizedDest.push("e") + + assertEquals(5, sizedDest.length) + assertEquals("e", sizedDest.top) + assertEquals(List("e", "d", "c", "b", "a"), sizedDest.toList) + } + + /** Same sized-clone pattern, but for MStackOfMaybe. */ + @Test def testMStackOfMaybeCopyFromSizedDestinationMatchesDefault(): Unit = { + val source = new MStackOfMaybe[String] + source.push(One("x")) + source.push(Nope) + source.push(One("z")) + + val sizedDest = new MStackOfMaybe[String](source.length) + sizedDest.copyFrom(source) + + val defaultDest = new MStackOfMaybe[String] + defaultDest.copyFrom(source) + + assertEquals(defaultDest.toListMaybe, sizedDest.toListMaybe) + assertEquals(3, sizedDest.length) + assertEquals(One("z"), sizedDest.top) + assertEquals(One("z"), sizedDest.pop) + assertEquals(Nope, sizedDest.pop) + assertEquals(One("x"), sizedDest.pop) + assertTrue(sizedDest.isEmpty) + } + + /** Same sized-construction pattern, for the primitive-specialized variants. */ + @Test def testMStackOfBooleanSizedConstructionWorks(): Unit = { + val stk = MStackOfBoolean(3) + stk.push(true) + stk.push(false) + stk.push(true) + assertEquals(3, stk.length) + assertEquals(true, stk.pop()) + assertEquals(false, stk.pop()) + assertEquals(true, stk.pop()) + } + + @Test def testMStackOfIntSizedConstructionWorks(): Unit = { + val stk = MStackOfInt(2) + stk.push(1) + stk.push(2) + stk.push(3) + assertEquals(3, stk.length) + assertEquals(3, stk.pop()) + assertEquals(2, stk.pop()) + assertEquals(1, stk.pop()) + } + + @Test def testMStackOfLongSizedConstructionWorks(): Unit = { + val stk = MStackOfLong(1) + stk.push(10L) + stk.push(20L) + assertEquals(2, stk.length) + assertEquals(20L, stk.pop()) + assertEquals(10L, stk.pop()) + } + + // trackMaxSizeReached is a `final val`; confirms it's off by default + // and maxSizeReached stays 0 while it is. + @Test def testMaxSizeReachedDisabledByDefault(): Unit = { + assertEquals(false, MStack.trackMaxSizeReached) + val stk = new MStackOf[String] + stk.push("a") + stk.push("b") + stk.push("c") + assertEquals(0, stk.maxSizeReached) + } + /** * This test compares MStackOfLong to ArrayStack[Long]. * diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/TestBuildWritePrefetchDataProcessor.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/TestBuildWritePrefetchDataProcessor.scala new file mode 100644 index 0000000000..c94be64fa5 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/TestBuildWritePrefetchDataProcessor.scala @@ -0,0 +1,1580 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors + +import java.io.ByteArrayOutputStream +import java.nio.charset.StandardCharsets +import scala.jdk.CollectionConverters.* + +import org.apache.daffodil.api +import org.apache.daffodil.core.compiler.Compiler +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.externalvars.ExternalVariablesLoader +import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter + +import org.junit.Assert.* +import org.junit.Test + +/** + * Exercises DataProcessor.unparse with useBuildWritePrefetch enabled, + * confirming byte-identical output vs. single-pass across many schema shapes. + */ +class TestBuildWritePrefetchDataProcessor { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + /** A hidden, constant-valued OVC probe wired in via `dfdl:hiddenGroupRef="ex:ovcProbe"`. + * Needed because `hasAnyPrefetchBeneficialOVC` is false for a schema with no resolvable + * OVC, which would silently fall back to single-pass regardless of the + * `useBuildWritePrefetch` tunable; contributes a leading "Z" byte to expected output. */ + val ovcProbeGroup: scala.xml.Elem = + + + + + + + @Test def testOVCSuspensionSchemaMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = 005 + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + // dfdl:textNumberPadCharacter="0"/textNumberJustification="right" produces + // actual, schema-configured zero-padding, not generic fillByte-based padding. + assertEquals("006005", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + @Test def testArrayChoiceSeparatorSchemaMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + +
H
+ a + b + c + X +
+ + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("ZH,a,b,c,X", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A choice reached while withinHiddenNest: both sides pick + // defaultUnparser unconditionally, since the outcome is schema-determined. + @Test def testHiddenChoiceMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = 3 + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("[hello,3", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A postfix-separated array as the last term: the separator must + // still be written after the final occurrence, not dropped. + @Test def testTrailingArrayPostfixSeparatorMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + a + b + c + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Za\nb\nc\n", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A choice branch with zero tree footprint (absent optional element) + // must still write its own initiator, or write's dispatch can hang. + @Test def testChoiceBranchWithAbsentOptionalElementMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + , + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = 1 + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Zfirst_defaultable1", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + /** Same schema as above, but with "opt" PRESENT (an actual match, not the empty-branch fallback). */ + @Test def testChoiceBranchWithPresentOptionalElementMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + , + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + 0 + 1 + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Zfirst_defaultable01", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + @Test def testManyOccurrenceArrayWithSmallPrefetchLimitMatchesSinglePass(): Unit = { + val numItems = 40 + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val items = (0 until numItems).map(i => {s"i$i"}) + val infoset = + + {items} + + + // A tiny lookahead window forces many build<->write handoffs across this array's 40 + // occurrences. BoundedPrefetchTest proves the lead counter stays bounded; this test + // just confirms the tunable threads through DataProcessor.unparse correctly and the + // output is still byte-perfect. + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes( + sch, + infoset, + extraTunables = Map("unparsePrefetchWindowNodes" -> "3") + ) + assertEquals( + "Z" + (0 until numItems).map(i => s"i$i").mkString(","), + new String(singlePassBytes, StandardCharsets.US_ASCII) + ) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A bare nested xs:sequence has no tree node of its own; the enclosing + // writeContent must not advance/free a tree-child position recursing into it. + @Test def testNestedBareSequenceMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + B + I1 + I2 + A + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + // The nested xs:sequence has no dfdl:separator of its own and doesn't inherit the + // outer's "," (a local property of the outer xs:sequence's annotation, not pushed + // down to nested groups), so it compiles as unseparated: no separator between + // inner1/inner2, but the outer's separator still appears before/after the nested group. + assertEquals("ZB,I1I2,A", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A fixed-length choice must pad unused space after a shorter branch, + // not just write the branch content. + @Test def testFixedLengthChoicePaddingMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + // typeB is 2 bytes; the choice's declared length is 5 bytes, so 3 + // bytes of ChoiceUnusedUnparser filler should follow it. Plus the + // leading 1-byte ovcProbe. + val infoset = XY + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals(6, singlePassBytes.length) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A hidden group's elements never get hasValue=true; write's dispatch + // must special-case that instead of blocking on it forever. + @Test def testHiddenGroupMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + + + + + , + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = V + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("HV", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Initiator/terminator with no separator: same delimiter-frame code + // path as the separator tests above, via a different property combination. + @Test def testInitiatorTerminatorMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + A + B + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z[AB]", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Escape-scheme state is confined entirely to write-only unparsers; + // confirms that end to end with no frame changes needed. + @Test def testEscapeSchemeMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + + + + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + one, two + three + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Zone#, two,three", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A dfdlx:layer-wrapped sequence: confirms the layer's DOS-splitting + // setup produces byte-identical output under prefetch. + @Test def testLayeredSequenceMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + // s1 is exactly 8 bytes (the layer's declared fixedLength), long enough that + // FixedLengthOutputStream.write's accumulate-then-auto-close-on-count-==fixedLength + // logic runs across several bytes, not just one or two; a short value could pass + // while still masking an off-by-one in the accumulation. + val infoset = + + ABCDEFGH + Q + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("ZABCDEFGHQ", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // FixedLengthLayer's length-exceeded error must surface the same way + // under prefetch, not hang write's coroutine or leak a raw exception. + @Test def testLayeredSequenceLengthMismatchErrorMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + // s1 is 10 bytes, 2 more than the layer's declared fixedLength=8. + val infoset = + + ABCDEFGHIJ + Q + + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassOut = new ByteArrayOutputStream() + val singlePassRes = + singlePassDp.unparse(new ScalaXMLInfosetInputter(infoset), singlePassOut) + assertTrue("expected a failed UnparseResult, not a successful one", singlePassRes.isError) + assertTrue( + singlePassRes.getDiagnostics + .get(0) + .getMessage + .contains("exceeded fixed layer length of 8") + ) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchOut = new ByteArrayOutputStream() + val prefetchRes = prefetchDp.unparse(new ScalaXMLInfosetInputter(infoset), prefetchOut) + assertTrue("expected a failed UnparseResult, not a successful one", prefetchRes.isError) + assertTrue( + prefetchRes.getDiagnostics + .get(0) + .getMessage + .contains("exceeded fixed layer length of 8") + ) + } + + // A chained suspension: computed1's OVC references computed2 (itself + // an OVC forward reference), two levels deep instead of one. + @Test def testNestedOVCSuspensionsMatchSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = 005 + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("007006005", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Binary integers (every other test here is textual), through an + // array to also exercise the repeating-child frame path. + @Test def testBinaryIntArrayMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + + + + + , + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + 1 + 2 + 3 + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertArrayEquals( + Array[Byte]('Z'.toByte, 0, 0, 0, 1, 0, 0, 0, 2, 0, 0, 0, 3), + singlePassBytes + ) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A variable-length dfdl:length expression (every other test here is + // constant), exercising computeTargetLength's non-constant branch. + @Test def testVariableLengthExpressionMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + 5 + hello + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z05hello", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A prefixed-length element: the prefix depends on the content's + // unparsed length, a forward dependency like OVC but length-driven. + @Test def testPrefixedLengthMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + , + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = hello + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z05hello", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A prefixed-length element with complex (not simple) content, so + // write's dispatch must recurse into the wrapped content unparser. + @Test def testPrefixedLengthComplexContentMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + , + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + + hi + bye + + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z06hi,bye", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // An OVC referencing fn:count of a preceding array, resolvable + // directly against the tree: the ordinary non-suspending OVC path. + @Test def testOVCCountOfPrecedingArrayMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = + + 1 + 2 + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("122", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A large array where build fully finishes before write's coroutine + // starts, so write's first signal is BuildFinished, not a live coroutine. + @Test def testOVCCountOfPrecedingArrayAfterBuildFullyFinishesMatchesSinglePass(): Unit = { + val numItems = 500 + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val items = (0 until numItems).map(i => {i % 10}) + val infoset = + + {items} + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A dynamic dfdl:terminator expression referencing a preceding + // array's count: navigation-only, resolvable directly against the tree. + @Test def testDynamicTerminatorReferencingArrayCountMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Zy", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // An IVC field (no unparse effect) referencing fn:exists over a + // nested array: exercises ordinary navigation past a nested array. + @Test def testIVCExistsOverNestedArrayMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + + 1 + 2 + 3 + 4 + + true + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z1,2,3,4", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Aggressive throttling forces every "len" suspension to be retried + // repeatedly; guards against re-running an already-resolved one. + @Test def testManyValueLengthForwardReferencesWithAggressiveThrottlingMatchesSinglePass() + : Unit = { + val numRecords = 25 + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val records = (0 until numRecords).map(i => {s"value$i"}) + val infoset = + + {records} + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes( + sch, + infoset, + extraTunables = Map( + "unparsePrefetchWindowNodes" -> "2", + "unparseSuspensionWaitYoung" -> "1", + "unparseSuspensionWaitOld" -> "1" + ) + ) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A repeating NVI scope with a valueLength OVC reading it, letting + // build race many iterations ahead: guards against a stale iteration's value. + @Test def testNVIScopedVariableWithValueLengthOVCMatchesSinglePass(): Unit = { + val numRecords = 25 + val sch = SchemaUtils.dfdlTestSchema( + , { + + + }, + + + + + + + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val records = + (0 until numRecords).map(i => + {f"$i%02d"} + {s"value$i"} + ) + val infoset = + + {records} + + + // Small window forces build to race many iterations ahead of write, + // each pushing its own runningVar instance, before write catches up. + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes( + sch, + infoset, + extraTunables = Map("unparsePrefetchWindowNodes" -> "2") + ) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Simpler variant of the repeating-NVI test above: a single + // (non-repeating) scope, as a distinct data point. + @Test def testSingleNVIScopedVariableWithValueLengthOVCMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , { + + + }, + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = + + 42 + hello + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Regression guard for NVI-scope setVariable resolution: no forward + // OVC or separator, the minimal shape that exposes the race. + @Test def testNVIScopedSetVariableWithNoForwardReferenceMatchesSinglePass(): Unit = { + val numRecords = 25 + val sch = SchemaUtils.dfdlTestSchema( + , { + + + }, + + + + + + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val records = (0 until numRecords).map(i => {f"$i%02d"}) + val infoset = + + {records} + + + val extVars = + ExternalVariablesLoader.mapToBindings(Map(s"{$example}runningVar" -> "-1").asJava) + + val singlePassDp = Compiler() + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + .withExternalVariables(extVars) + val singlePassBytes = TestUtils.unparseToBytes(singlePassDp, infoset) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .withTunable("unparsePrefetchWindowNodes", "2") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + .withExternalVariables(extVars) + val prefetchBytes = TestUtils.unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A setup failure (inputter never produces StartDocument) must yield + // a failed UnparseResult, not an NPE from a null error-path state. + @Test def testMalformedInfosetInputterGetsCleanErrorNotNPE(): Unit = { + // Needs at least one prefetch-beneficial (value-only, not length-dependent) OVC, or + // DataProcessor.unparse's dispatch (ssrd.hasAnyPrefetchBeneficialOVC) falls back to + // single-pass regardless of the tunable, and unparseViaBuildThenWrite (the method + // under test) would never run. + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + + // hasNext() = false immediately means initialize()'s + // "!delegate.hasNext" check fires straight away, before any actual + // infoset event is produced; exactly the "never starts with + // StartDocument" failure this guards. + val neverStartsInputter = new api.infoset.InfosetInputter { + override def getEventType() = null + override def getLocalName() = null + override def getNamespaceURI() = null + override def getSimpleText( + primType: org.apache.daffodil.runtime1.dpath.NodeInfo.Kind, + runtimeProperties: java.util.Map[String, String] + ) = null + override def isNilled(): java.lang.Boolean = null + override def hasNext() = false + override def next(): Unit = () + override def fini(): Unit = () + } + + val out = new ByteArrayOutputStream() + val res = dp.unparse(neverStartsInputter, out) + + assertTrue("expected a failed UnparseResult, not a successful one", res.isError) + assertTrue( + res.getDiagnostics.get(0).getMessage.contains("does not start with StartDocument") + ) + } + + // A variable-length dfdl:length expression inside a separated + // sequence must also call computeTargetLength from write's dispatch. + @Test def testDelimitedVariableLengthExpressionMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + 5 + hello + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z05|hello", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Same dispatch path as the test above, but for a complex element + // whose expression-based dfdl:length wraps a whole group's content chain. + @Test def testDelimitedComplexVariableLengthExpressionMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + 3 + xyz + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z03|xyz", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A schema where every OVC is content-length-dependent (never + // resolvable early) must fall back to single-pass automatically. + @Test def testPurelyContentLengthOVCSchemaFallsBackAutomatically(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = xyz + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = TestUtils.unparseToBytes(singlePassDp, infoset) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertFalse( + "schema has only a content-length-dependent OVC; should never report prefetch-beneficial", + prefetchDp.ssrd.hasAnyPrefetchBeneficialOVC + ) + val prefetchBytes = TestUtils.unparseToBytes(prefetchDp, infoset) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A schema mixing a resolvable OVC with an unrelated + // content-length-dependent one must still use the prefetch path. + @Test def testMixedSchemaWithResolvableAndContentLengthOVCStillUsesPrefetchPath(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = + + 005 + xyz + + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = TestUtils.unparseToBytes(singlePassDp, infoset) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertTrue( + "schema has a resolvable-without-writing OVC alongside a content-length one; " + + "must still report prefetch-beneficial", + prefetchDp.ssrd.hasAnyPrefetchBeneficialOVC + ) + val prefetchBytes = TestUtils.unparseToBytes(prefetchDp, infoset) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A schema with no OVC at all: hasAnyPrefetchBeneficialOVC must not + // throw, and simply reports false. + @Test def testSchemaWithNoOVCAtAllReportsNotBeneficial(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertFalse(dp.ssrd.hasAnyPrefetchBeneficialOVC) + } + + // A schema where every OVC is resolvable-without-writing must report + // true: the common case prefetch already handles. + @Test def testSchemaWithOnlyResolvableOVCReportsBeneficial(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertTrue(dp.ssrd.hasAnyPrefetchBeneficialOVC) + } + + // hasAnyPrefetchBeneficialOVC is baked in at compile time; confirms + // the fallback still applies after a save/reload round trip. + @Test def testHasAnyPrefetchBeneficialOVCSurvivesSaveReload(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertFalse(dp.ssrd.hasAnyPrefetchBeneficialOVC) + + val os = new ByteArrayOutputStream() + dp.save(java.nio.channels.Channels.newChannel(os)) + val reloadedDp = Compiler() + .reload(new java.io.ByteArrayInputStream(os.toByteArray)) + .asInstanceOf[DataProcessor] + assertFalse(reloadedDp.ssrd.hasAnyPrefetchBeneficialOVC) + + val infoset = xyz + assertArrayEquals( + TestUtils.unparseToBytes(dp, infoset), + TestUtils.unparseToBytes(reloadedDp, infoset) + ) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBoundedPrefetch.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBoundedPrefetch.scala new file mode 100644 index 0000000000..93e959e032 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBoundedPrefetch.scala @@ -0,0 +1,122 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream +import java.nio.charset.StandardCharsets + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase + +import org.junit.Assert.* +import org.junit.Test + +/** + * Proves build pauses to let write catch up: with a small prefetchLimit, + * the lead counter stays bounded mid-recursion, not equal to the full count. + */ +class TestBoundedPrefetch { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testBuildLeadStaysBoundedDuringRecursion(): Unit = { + val numItems = 40 + val prefetchLimit = 3L + + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + , + elementFormDefault = "unqualified" + ) + + val items = (0 until numItems).map(i => {s"i$i"}) + val infoset = + + {items} + + val expectedBytes = (0 until numItems).map(i => s"i$i").mkString(",") + + val dp = TestUtils.compileForUnparse( + sch, + Map("releaseUnneededInfoset" -> "false", "useBuildWritePrefetch" -> "true") + ) + + val buildInputter = TestUtils.newInitializedInputter(infoset, dp) + + val sharedCtx = + UnparseSharedContextTestFixture.build(dp, prefetchLimit)() + + val walkerOut = new ByteArrayOutputStream() + val writeInputter = TestUtils.newInitializedInputter(infoset, dp) + val writeState = UState.createInitialUState(walkerOut, dp, writeInputter, false) + writeState.setSharedContext(sharedCtx) + writeState.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + val rootUnparser = dp.ssrd.unparser.asInstanceOf[ElementUnparserBase] + UnparseSharedContextTestFixture.wireCoroutines( + sharedCtx, + buildInputter.documentElement, + rootUnparser, + writeState + ) + + val buildState = new BuildState(buildInputter, sharedCtx, Nil, false) + + dp.ssrd.builder.get.build(buildState) + + // See class doc above for why this proves interleaving; numItems + 1 + // (row + all items) is what currentLead would equal here if + // resumeWrite never fired mid-recursion. + assertTrue( + s"expected lead close to prefetchLimit=$prefetchLimit after build, but was ${sharedCtx.currentLead} " + + s"(numItems=$numItems); build did not actually pause for write", + sharedCtx.currentLead <= prefetchLimit + 1 + ) + assertTrue( + "expected build to have gotten ahead of write by at least one node", + sharedCtx.currentLead > 0 + ) + + // Drain the rest and confirm the output is byte-for-byte correct + // despite having been produced across many separate resumeWrite calls + // rather than a single one-shot write pass. + val finalSignal = sharedCtx.resumeWrite(BuildFinished) + finalSignal match { + case WriteDone(Some(t)) => throw t + case WriteDone(None) => // continue below + case other => fail(s"unexpected final signal: $other") + } + writeState.evalSuspensions(isFinal = true) + writeState.getDataOutputStream.setFinished(writeState) + + assertEquals(expectedBytes, new String(walkerOut.toByteArray, StandardCharsets.US_ASCII)) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildState.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildState.scala new file mode 100644 index 0000000000..c2d27e5621 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildState.scala @@ -0,0 +1,89 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils + +import org.junit.Assert.* +import org.junit.Test + +/** + * Validates BuildState in isolation: confirms it surfaces the same + * event sequence a real UStateMain would, for a separated schema. + */ +class TestBuildState { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testBuildStateSurfacesCorrectEventSequence(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + val infoset = + + Alice + 30 + Boston + + + val dp = TestUtils.compileForUnparse( + sch, + Map("releaseUnneededInfoset" -> "false", "useBuildWritePrefetch" -> "true") + ) + + val inputter = TestUtils.newInitializedInputter(infoset, dp) + // initialize() pushes the root TRD as its last step - the same + // setup a real unparse's invariant check relies on. + + val sharedCtx = + UnparseSharedContextTestFixture.build(dp, prefetchLimit = 100)() + val buildState = new BuildState(inputter, sharedCtx, Nil, false) + + // Drives through the ACTUAL Builder recursion, not hand-driven + // advance() calls, since next-element resolution depends on the same + // TRD push/pop dance ElementBuilder.build performs. This schema's + // separator never reaches BuildState at all: the Builder tree skips + // straight past the delimiter-stack wrapper unparser entirely. + dp.ssrd.builder.get.build(buildState) + + assertEquals(4L, sharedCtx.currentLead) // row, name, age, city + + val rootNode = inputter.documentElement.child(0).asComplex + assertEquals(3, rootNode.numChildren) + assertEquals("name", rootNode.child(0).erd.name) + assertEquals("age", rootNode.child(1).erd.name) + assertEquals("city", rootNode.child(2).erd.name) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildWriteArrayChoice.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildWriteArrayChoice.scala new file mode 100644 index 0000000000..f239041689 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildWriteArrayChoice.scala @@ -0,0 +1,211 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream +import java.nio.charset.StandardCharsets + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase + +import org.junit.Assert.* +import org.junit.Test + +/** + * Validates write-side dispatch against a single-pass tree, and a + * standalone BuildState run, for both scalar and array/choice content. + */ +class TestBuildWriteArrayChoice { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testWriteWalkerMatchesActualUnparse(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + val infoset = + + Alice + 30 + Boston + + + val dp = TestUtils.compileForUnparse( + sch, + Map("releaseUnneededInfoset" -> "false", "useBuildWritePrefetch" -> "true") + ) + + val (singlePassBytes, walkerBytes) = + TestUtils.getSinglePassAndWriteContentBytes(dp, infoset) + + assertArrayEquals(singlePassBytes, walkerBytes) + } + + @Test def testArrayAndChoiceWriteContentMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + // Three "item" occurrences exercise the array loop; typeB (not typeA) + // exercises actual choice resolution. lengthKind=delimited (not + // explicit) avoids double-firing CaptureStartOfContentLengthUnparser's + // non-idempotent marker, since this tree is reused from a completed unparse. + val infoset = + +
H
+ a + b + c + X +
+ + val dp = TestUtils.compileForUnparse( + sch, + Map("releaseUnneededInfoset" -> "false", "useBuildWritePrefetch" -> "true") + ) + + val (singlePassBytes, walkerBytes) = + TestUtils.getSinglePassAndWriteContentBytes(dp, infoset) + + assertEquals("H,a,b,c,X", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, walkerBytes) + } + + // Standalone-build regression: drives BuildState directly, then feeds + // its tree into write's writeContent (end-to-end build-then-write). + @Test def testStandaloneBuildStateNavigatesArrayChoiceSeparator(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + val infoset = + +
H
+ a + b + c + X +
+ + val dp = TestUtils.compileForUnparse( + sch, + Map("releaseUnneededInfoset" -> "false", "useBuildWritePrefetch" -> "true") + ) + + // Build phase: standalone BuildState drives the actual Unparser recursion, + // navigating past the sequence's separator and through the + // array/choice content, purely to build the tree. + val buildInputter = TestUtils.newInitializedInputter(infoset, dp) + + val sharedCtx = + UnparseSharedContextTestFixture.build(dp, prefetchLimit = 100)() + val buildState = new BuildState(buildInputter, sharedCtx, Nil, false) + + dp.ssrd.builder.get.build(buildState) + + // row, header, item x3, typeB = 6 elements total. + assertEquals(6L, sharedCtx.currentLead) + + val rootNode = buildInputter.documentElement.child(0).asComplex + assertEquals(3, rootNode.numChildren) + assertEquals("header", rootNode.child(0).erd.name) + assertEquals("item", rootNode.child(1).erd.name) + assertEquals(3, rootNode.child(1).asInstanceOf[DIArray].numChildren) + assertEquals("typeB", rootNode.child(2).erd.name) + + // Build was driven directly (no coroutine handoff), so sharedCtx can't + // know build is done; tell it so write's awaitChild call takes the + // post-BuildFinished (suspension-retry) path instead of resuming a + // coroutine that was never set up. + sharedCtx.observeBuildSignal(BuildFinished) + + // Write phase: write the tree BuildState just constructed, confirming + // it's a usable, fully-built tree, not just a navigation exercise. + val walkerOut = new ByteArrayOutputStream() + val writeInputter = TestUtils.newInitializedInputter(infoset, dp) + val writeState = UState.createInitialUState(walkerOut, dp, writeInputter, false) + writeState.setSharedContext(sharedCtx) + writeState.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + val rootUnparser = dp.ssrd.unparser.asInstanceOf[ElementUnparserBase] + val rootWriteNode = sharedCtx.awaitChild(buildInputter.documentElement, 0) + rootUnparser.writeContent(rootWriteNode, writeState) + // The separator (default separatorSuppressionPolicy "anyEmpty") is + // written speculatively via a suspension deciding, once known, if the + // region it precedes is zero-length; this drains that chain before the + // DOS is finalized, mirroring DataProcessor's finishWriteSide. + writeState.evalSuspensions(isFinal = true) + writeState.getDataOutputStream.setFinished(writeState) + + assertEquals("H,a,b,c,X", new String(walkerOut.toByteArray, StandardCharsets.US_ASCII)) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestLeadCounter.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestLeadCounter.scala new file mode 100644 index 0000000000..50dad4b3a2 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestLeadCounter.scala @@ -0,0 +1,115 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream +import java.nio.charset.StandardCharsets + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase + +import org.junit.Assert.* +import org.junit.Test + +/** + * Validates the shared build/write lead counter end to end: build + * increments and write decrements against the same shared instance. + */ +class TestLeadCounter { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testLeadCounterIncrementsOnBuildAndDecrementsOnWrite(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + // Fixed dfdl:length is safe here because this tree comes from + // BuildState, which never runs content-writing (including + // CaptureStartOfContentLengthUnparser). + val infoset = + + Alice + 30 + Boston + + + val dp = TestUtils.compileForUnparse( + sch, + Map("releaseUnneededInfoset" -> "false", "useBuildWritePrefetch" -> "true") + ) + + // Build phase: drive BuildState through the actual recursion, incrementing + // the shared lead counter via the actual unparseBegin hookup, as in + // BuildStateTest. + val buildInputter = TestUtils.newInitializedInputter(infoset, dp) + + val sharedCtx = + UnparseSharedContextTestFixture.build(dp, prefetchLimit = 100)() + val buildState = new BuildState(buildInputter, sharedCtx, Nil, false) + + assertEquals(0L, sharedCtx.currentLead) + dp.ssrd.builder.get.build(buildState) + + // row itself, name, age, city = 4 elements total, each incrementing + // once via unparseBegin's actual hookup. + assertEquals(4L, sharedCtx.currentLead) + + // Build was driven directly (no coroutine handoff), so sharedCtx can't + // know build is done; tell it so write's awaitChild calls take the + // post-BuildFinished (suspension-retry) path instead of resuming a + // coroutine that was never set up. + sharedCtx.observeBuildSignal(BuildFinished) + + // Write phase: write the SAME already-built tree + // (buildInputter.documentElement) against the SAME sharedCtx, + // decrementing the lead counter as it goes. + val walkerOut = new ByteArrayOutputStream() + val writeInputter = TestUtils.newInitializedInputter(infoset, dp) + val writeState = UState.createInitialUState(walkerOut, dp, writeInputter, false) + writeState.setSharedContext(sharedCtx) + writeState.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + val rootUnparser = dp.ssrd.unparser.asInstanceOf[ElementUnparserBase] + val rootNode = sharedCtx.awaitChild(buildInputter.documentElement, 0) + rootUnparser.writeContent(rootNode, writeState) + writeState.getDataOutputStream.setFinished(writeState) + + assertEquals("Alice30Boston", new String(walkerOut.toByteArray, StandardCharsets.US_ASCII)) + + // Write decremented once per element too, so the counter is back to 0: + // build and write agree on how many nodes exist, coordinated through + // the shared UnparseSharedContext. + assertEquals(0L, sharedCtx.currentLead) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContextTestFixture.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContextTestFixture.scala new file mode 100644 index 0000000000..29d1a9c899 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContextTestFixture.scala @@ -0,0 +1,70 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.runtime1.infoset.DIDocument +import org.apache.daffodil.runtime1.processors.DataProcessor +import org.apache.daffodil.runtime1.processors.SuspensionTracker +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase + +/** + * Shared BuildState construction for tests, mirroring production's `* 2` + * doubling of tunable-derived suspension-wait thresholds (default only). + */ +object UnparseSharedContextTestFixture { + def build(dp: DataProcessor, prefetchLimit: Long)( + suspensionWaitYoung: Int = dp.tunables.unparseSuspensionWaitYoung * 2, + suspensionWaitOld: Int = dp.tunables.unparseSuspensionWaitOld * 2 + ): UnparseSharedContext = { + new UnparseSharedContext( + new SuspensionTracker(suspensionWaitYoung, suspensionWaitOld), + dp, + dp.tunables, + prefetchLimit + ) + } + + /** Wires a BuildCoroutine/WriteCoroutine pair onto sharedCtx, mirroring + * production's WriteCoroutine, so mid-recursion resumeWrite calls do + * something; caller must still perform the final handoff afterward. + */ + def wireCoroutines( + sharedCtx: UnparseSharedContext, + documentElement: DIDocument, + rootUnparser: ElementUnparserBase, + writeState: UState + ): Unit = { + val buildCoroutine = new BuildCoroutine() + val writeCoroutine = new WriteCoroutine({ (wc, firstSignal) => + try { + sharedCtx.observeBuildSignal(firstSignal) + try { + val rootNode = sharedCtx.awaitChild(documentElement, 0) + rootUnparser.writeContent(rootNode, writeState) + } catch { + case _: AwaitChildStalledException => + } + wc.resumeFinal(buildCoroutine, WriteDone(None)) + } catch { + case t: Throwable => + wc.resumeFinal(buildCoroutine, WriteDone(Some(t))) + } + }) + sharedCtx.setCoroutines(buildCoroutine, writeCoroutine) + } +} diff --git a/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd b/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd index db5568846a..1fda1bea74 100644 --- a/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd +++ b/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd @@ -534,6 +534,50 @@ + + + + If true, unparsing uses a build/write-prefetch path instead of the + default single-pass unparse: a build pass runs ahead of a write pass + over the same infoset tree, resolving forward references against the + tree directly instead of via Suspensions where possible. How far + build may run ahead of write is bounded by + unparsePrefetchWindowNodes. + + + + + + + Only consulted when useBuildWritePrefetch is true. The maximum + number of infoset nodes build may construct ahead of write before + build pauses to let write catch up. Bounds peak memory use to + roughly this many resident nodes rather than the whole infoset. + + + + + + + Only consulted when useBuildWritePrefetch is true. A second, + independent limit on how far build may run ahead of write, + alongside unparsePrefetchWindowNodes: the maximum number of + pending suspensions (forward references that cannot resolve + until write itself writes the referenced bytes, e.g. + dfdl:valueLength/dfdl:contentLength) build may accumulate + before it pauses to let write catch up, regardless of how + large unparsePrefetchWindowNodes is set. Without this, + unparsePrefetchWindowNodes alone does not bound this + backlog, since a node's lead-counter contribution is + released once write's structural pass reaches it, whether or + not that node's own suspension has resolved yet, so a large + window can otherwise let many resolved-but-still- + pending nodes accumulate unboundedly before anything + forces write to actually catch up to what they're + waiting on. + + + @@ -644,7 +688,11 @@ evaluating suspensions that are likely to fail. The unparseSuspensionWaitYoung and unparseSuspensionWaitOld values determine how many elements are unparsed before evaluating - young and old suspensions, respectively. + young and old suspensions, respectively. When useBuildWritePrefetch + is true, both values are used internally at double what is + configured here, since build and write independently advance this + same counter, so it would otherwise tick twice as fast as a single + traversal. diff --git a/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala b/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala index d6632308b8..3f04975a5b 100644 --- a/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala +++ b/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala @@ -79,7 +79,8 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm | if (configOpt.isDefined) { | val loader = new DaffodilXMLLoader() | val node = loader.load(URISchemaSource(Paths.get(configPath).toFile, configOpt.get), Some(XMLUtils.dafextURI)) - | tunablesMap(node) + | val optTunablesNode = (node \ "tunables").headOption + | optTunablesNode.map(tunablesMap(_)).getOrElse(Map.empty) | } else { | Map.empty | } From d3ef91158aeeb8bb638a678b196b2721c7efbc33 Mon Sep 17 00:00:00 2001 From: olabusayoT <50379531+olabusayoT@users.noreply.github.com> Date: Mon, 28 Sep 2026 14:43:16 -0400 Subject: [PATCH 2/6] Gate builder construction on hasAnyPrefetchBeneficialOVC too builder was only gated on the useBuildWritePrefetch tunable, while the runtime flag that actually decides whether prefetch gets used additionally required hasAnyPrefetchBeneficialOVC. A schema with no prefetch-beneficial OVC never reaches unparseViaBuildThenWrite, so building the parallel Builder tree for it was pure waste even with the tunable globally on. Gate builder on the same combined condition. TestBuildState, TestLeadCounter, TestBoundedPrefetch, and TestBuildWriteArrayChoice's standalone-build test each call dp.ssrd.builder.get directly, bypassing that selection; their schemas had no OVC at all, so builder is now correctly Nope for them. Give each a dfdl:defineVariable-backed marker element (a variable reference has no element references and isn't a compile-time constant, unlike a literal, which the compiler folds to isConstant=true) so their schemas are actually prefetch-beneficial, and update the resulting element counts/output assertions. DAFFODIL-3065 --- .../runtime1/SchemaSetRuntime1Mixin.scala | 11 +++++--- .../unparsers/TestBoundedPrefetch.scala | 22 +++++++++++----- .../processors/unparsers/TestBuildState.scala | 23 ++++++++++++----- .../unparsers/TestBuildWriteArrayChoice.scala | 25 +++++++++++++------ .../unparsers/TestLeadCounter.scala | 24 ++++++++++++------ 5 files changed, 75 insertions(+), 30 deletions(-) diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala index 1f2ae97867..b827547363 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala @@ -103,18 +103,21 @@ trait SchemaSetRuntime1Mixin { Assert.invariant(!root.isError) // Only reference builder/hasAnyPrefetchBeneficialOVC (both real // work: a full parallel Builder tree, a schema component scan) when - // the tunable that would actually use them is on; otherwise every - // schema compile would pay for them regardless. + // prefetch will actually be used for this schema; a schema with no + // prefetch-beneficial OVC never reaches unparseViaBuildThenWrite (see + // DataProcessor.unparse), so building the Builder tree for it would + // be pure waste even with the tunable globally on. + val isPrefetchInUse = tunable.useBuildWritePrefetch && root.hasAnyPrefetchBeneficialOVC val ssrd = new SchemaSetRuntimeData( parser, unparser, - if (tunable.useBuildWritePrefetch) builder else Nope, + if (isPrefetchInUse) builder else Nope, root.elementRuntimeData, variableMap, allLayers, layerRuntimeCompiler, - tunable.useBuildWritePrefetch && root.hasAnyPrefetchBeneficialOVC + isPrefetchInUse ) if (root.numComponents > root.numUniqueComponents) Logger.log.debug( diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBoundedPrefetch.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBoundedPrefetch.scala index 93e959e032..607f5f2995 100644 --- a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBoundedPrefetch.scala +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBoundedPrefetch.scala @@ -42,15 +42,25 @@ class TestBoundedPrefetch { val sch = SchemaUtils.dfdlTestSchema( , - , + { + + + }, + + , @@ -62,7 +72,7 @@ class TestBoundedPrefetch { {items} - val expectedBytes = (0 until numItems).map(i => s"i$i").mkString(",") + val expectedBytes = (0 until numItems).map(i => s"i$i").mkString(",") + ",M" val dp = TestUtils.compileForUnparse( sch, @@ -92,8 +102,8 @@ class TestBoundedPrefetch { dp.ssrd.builder.get.build(buildState) - // See class doc above for why this proves interleaving; numItems + 1 - // (row + all items) is what currentLead would equal here if + // See class doc above for why this proves interleaving; numItems + 2 + // (row + all items + marker) is what currentLead would equal here if // resumeWrite never fired mid-recursion. assertTrue( s"expected lead close to prefetchLimit=$prefetchLimit after build, but was ${sharedCtx.currentLead} " + diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildState.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildState.scala index c2d27e5621..aed0e1dde4 100644 --- a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildState.scala +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildState.scala @@ -35,16 +35,26 @@ class TestBuildState { @Test def testBuildStateSurfacesCorrectEventSequence(): Unit = { val sch = SchemaUtils.dfdlTestSchema( , - , + { + + + }, + + , @@ -78,12 +88,13 @@ class TestBuildState { // straight past the delimiter-stack wrapper unparser entirely. dp.ssrd.builder.get.build(buildState) - assertEquals(4L, sharedCtx.currentLead) // row, name, age, city + assertEquals(5L, sharedCtx.currentLead) // row, name, age, city, marker val rootNode = inputter.documentElement.child(0).asComplex - assertEquals(3, rootNode.numChildren) + assertEquals(4, rootNode.numChildren) assertEquals("name", rootNode.child(0).erd.name) assertEquals("age", rootNode.child(1).erd.name) assertEquals("city", rootNode.child(2).erd.name) + assertEquals("marker", rootNode.child(3).erd.name) } } diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildWriteArrayChoice.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildWriteArrayChoice.scala index f239041689..3722cb7523 100644 --- a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildWriteArrayChoice.scala +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildWriteArrayChoice.scala @@ -127,9 +127,12 @@ class TestBuildWriteArrayChoice { @Test def testStandaloneBuildStateNavigatesArrayChoiceSeparator(): Unit = { val sch = SchemaUtils.dfdlTestSchema( , - , + { + + + }, @@ -141,6 +144,13 @@ class TestBuildWriteArrayChoice { + + , @@ -172,15 +182,16 @@ class TestBuildWriteArrayChoice { dp.ssrd.builder.get.build(buildState) - // row, header, item x3, typeB = 6 elements total. - assertEquals(6L, sharedCtx.currentLead) + // row, header, item x3, typeB, marker = 7 elements total. + assertEquals(7L, sharedCtx.currentLead) val rootNode = buildInputter.documentElement.child(0).asComplex - assertEquals(3, rootNode.numChildren) + assertEquals(4, rootNode.numChildren) assertEquals("header", rootNode.child(0).erd.name) assertEquals("item", rootNode.child(1).erd.name) assertEquals(3, rootNode.child(1).asInstanceOf[DIArray].numChildren) assertEquals("typeB", rootNode.child(2).erd.name) + assertEquals("marker", rootNode.child(3).erd.name) // Build was driven directly (no coroutine handoff), so sharedCtx can't // know build is done; tell it so write's awaitChild call takes the @@ -206,6 +217,6 @@ class TestBuildWriteArrayChoice { writeState.evalSuspensions(isFinal = true) writeState.getDataOutputStream.setFinished(writeState) - assertEquals("H,a,b,c,X", new String(walkerOut.toByteArray, StandardCharsets.US_ASCII)) + assertEquals("H,a,b,c,X,M", new String(walkerOut.toByteArray, StandardCharsets.US_ASCII)) } } diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestLeadCounter.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestLeadCounter.scala index 50dad4b3a2..a26da08f31 100644 --- a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestLeadCounter.scala +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestLeadCounter.scala @@ -39,15 +39,25 @@ class TestLeadCounter { @Test def testLeadCounterIncrementsOnBuildAndDecrementsOnWrite(): Unit = { val sch = SchemaUtils.dfdlTestSchema( , - , + { + + + }, + + , @@ -81,9 +91,9 @@ class TestLeadCounter { assertEquals(0L, sharedCtx.currentLead) dp.ssrd.builder.get.build(buildState) - // row itself, name, age, city = 4 elements total, each incrementing - // once via unparseBegin's actual hookup. - assertEquals(4L, sharedCtx.currentLead) + // row itself, name, age, city, marker = 5 elements total, each + // incrementing once via unparseBegin's actual hookup. + assertEquals(5L, sharedCtx.currentLead) // Build was driven directly (no coroutine handoff), so sharedCtx can't // know build is done; tell it so write's awaitChild calls take the @@ -105,7 +115,7 @@ class TestLeadCounter { rootUnparser.writeContent(rootNode, writeState) writeState.getDataOutputStream.setFinished(writeState) - assertEquals("Alice30Boston", new String(walkerOut.toByteArray, StandardCharsets.US_ASCII)) + assertEquals("Alice30BostonM", new String(walkerOut.toByteArray, StandardCharsets.US_ASCII)) // Write decremented once per element too, so the counter is back to 0: // build and write agree on how many nodes exist, coordinated through From 0c25cdec31564536d7421709eb94df47172610bd Mon Sep 17 00:00:00 2001 From: olabusayoT <50379531+olabusayoT@users.noreply.github.com> Date: Mon, 28 Sep 2026 14:43:01 -0400 Subject: [PATCH 3/6] Deduplicate writeContent/unparse setup-dispatch-teardown logic ElementUnparserBase's writeContent and unparse independently ran the same before-content, content-dispatch, after-content, and setVariables sequence. Extract runElementContent(state, dispatch), taking only the content-dispatch step as a closure: writeContent's WriteUnparser-bypass check (needed to reach a nested group's own writeContent) versus unparse's plain, event-driven dispatch wrapped in unparse's own TermRuntimeData push/pop. WriteUnparser.writeWithPushPop already gave writeContent implementations a shared setup/dispatch/teardown skeleton; each of those same classes' unparse() hand-rolled the identical sequence separately. Extract withPushPop(state, setup, dispatch, teardown); writeWithPushPop becomes a thin specialization that builds its own dispatch closure (writeContent if bodyUnparser is a WriteUnparser, else unparse1). HiddenGroupCombinatorUnparser, DelimiterStackUnparser, DynamicEscapeSchemeUnparser, and the two SpecifiedLength unparsers' unparse() now call withPushPop directly instead of duplicating the try/finally by hand. ComplexNilOrContentUnparser's unparse() and writeContent() separately re-derived the same isNilled branch decision; extract chooseBodyUnparser(node) so both read it once. No behavior change, except SpecifiedLengthPrefixedUnparser's teardown (resolving the prefix length) now also runs if eUnparser.unparse1 throws during unparse(), matching writeContent's own teardown, which already ran unconditionally for the same reason. DAFFODIL-3065 --- .../ChoiceAndOtherVariousUnparsers.scala | 30 +++--- .../unparsers/runtime1/ElementUnparser.scala | 92 ++++++++++--------- .../HiddenGroupCombinatorUnparser.scala | 16 ++-- .../NilEmptyCombinatorUnparsers.scala | 19 ++-- .../runtime1/SpecifiedLengthUnparsers.scala | 25 +++-- .../unparsers/runtime1/WriteUnparser.scala | 50 ++++++---- 6 files changed, 125 insertions(+), 107 deletions(-) diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala index 74b0045a3c..07b3b9138f 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala @@ -304,14 +304,13 @@ class DelimiterStackUnparser( override val runtimeDependencies = (initiatorOpt.toList ++ separatorOpt.toList ++ terminatorOpt.toList).toArray - def unparse(state: UState): Unit = { - pushDelimiterScope(state) - try { - bodyUnparser.unparse1(state) - } finally { - state.popDelimiters() - } - } + def unparse(state: UState): Unit = + withPushPop( + state, + setup = pushDelimiterScope, + dispatch = bodyUnparser.unparse1, + teardown = (state, _) => state.popDelimiters() + ) private def pushDelimiterScope(state: UState): Unit = { val init = @@ -353,14 +352,13 @@ class DynamicEscapeSchemeUnparser( teardown = (state, _) => escapeScheme.invalidateCache(state) ) - def unparse(state: UState): Unit = { - cacheEscapeScheme(state) - try { - bodyUnparser.unparse1(state) - } finally { - escapeScheme.invalidateCache(state) - } - } + def unparse(state: UState): Unit = + withPushPop( + state, + setup = cacheEscapeScheme, + dispatch = bodyUnparser.unparse1, + teardown = (state, _) => escapeScheme.invalidateCache(state) + ) // Evaluates the dynamic escape scheme in the correct scope; the result is // cached in the Evaluatable (since it is manually cached), so future diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala index 636d7bc055..fce7dc3278 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala @@ -235,6 +235,22 @@ sealed abstract class ElementUnparserBase( dispatchContentUnparser(state) } + /** + * The steps identical whether reached via writeContent's already-built + * containerNode or unparse's own event-consuming attach: before-content, + * the content dispatch itself, after-content, and setVariables. dispatch + * abstracts writeContent's WriteUnparser-bypass check (needed to reach a + * nested group's own writeContent) from unparse's plain, purely + * event-driven runContentUnparser. + */ + private[runtime1] def runElementContent(state: UState, dispatch: UState => Unit): Unit = { + captureRuntimeValuedExpressionValues(state) + doBeforeContentUnparser(state) + dispatch(state) + doAfterContentUnparser(state) + computeSetVariables(state) + } + /** * Writes this element's content against an already-built * `containerNode`, without consuming any InfosetInputter events: @@ -245,36 +261,26 @@ sealed abstract class ElementUnparserBase( state.currentInfosetNodeStack.push(One(containerNode)) state.childIndexStack.push(0L) try { - // writeContent is only ever called from write's side, so this and - // computeSetVariables below always run here, at write's own - // document-order position. - captureRuntimeValuedExpressionValues(state) - doBeforeContentUnparser(state) // contentSetup can suspend, and suspending reads state.processor. // That's normally set by the ordinary unparse dispatch, which this // call bypasses entirely, so it's set explicitly here to match. state.setProcessor(this) - // Must run before the dispatch below: a group-wrapped eUnparser - // (delimiter stack, escape scheme, specified-length prefix) - // delegates straight to its own writeContent and never reaches - // dispatchContentUnparser, which would otherwise run this instead. - contentSetup(state) - // eReptypeUnparser takes priority here too, matching - // dispatchContentUnparser: a repType'd element's raw eUnparser can - // itself be group-wrapped, and without this check would wrongly - // delegate to that raw content instead of converting via repType. - eUnparser.toOption match { - case Some(wu: WriteUnparser) if eReptypeUnparser.isEmpty => - wu.writeContent(containerNode, state) - case _ => - dispatchContentUnparser(state) - } - // The after-content (padding/fill) region depends on the content - // having been written, so it must run only once the (possibly - // nested) content dispatch above has fully returned; true for both - // the simple-element and group-content cases. - doAfterContentUnparser(state) - computeSetVariables(state) + runElementContent( + state, + dispatch = { s => + contentSetup(s) + // eReptypeUnparser takes priority here too, matching + // dispatchContentUnparser: a repType'd element's raw eUnparser can + // itself be group-wrapped, and without this check would wrongly + // delegate to that raw content instead of converting via repType. + eUnparser.toOption match { + case Some(wu: WriteUnparser) if eReptypeUnparser.isEmpty => + wu.writeContent(containerNode, s) + case _ => + dispatchContentUnparser(s) + } + } + ) // Only a simple node needs finalizing here (complex/array nodes were // already finalized in build's unparseEnd; re-finalizing trips // setFinal()'s !isFinal assert). An OVC node may still be valueless @@ -306,23 +312,23 @@ sealed abstract class ElementUnparserBase( unparseBegin(state) - captureRuntimeValuedExpressionValues(state) - - doBeforeContentUnparser(state) - - // We must push the TermRuntimeData for all model-groups, starting from - // the complex type's model-group; simple types have none to push. - if (erd.isComplexType) - state.pushTRD(erd.optComplexTypeModelGroupRuntimeData.get) - - runContentUnparser(state) - - if (erd.isComplexType) - state.popTRD(erd.optComplexTypeModelGroupRuntimeData.get) - - doAfterContentUnparser(state) - - computeSetVariables(state) + runElementContent( + state, + dispatch = { s => + // We must push the TermRuntimeData for all model-groups, starting + // from the complex type's model-group; simple types have none to + // push. Only unparse's own event-driven dispatch needs this: + // nextElement, its sole reader, resolves a raw event's tag name + // against it, and writeContent never consumes events at all. + if (erd.isComplexType) + s.pushTRD(erd.optComplexTypeModelGroupRuntimeData.get) + + runContentUnparser(s) + + if (erd.isComplexType) + s.popTRD(erd.optComplexTypeModelGroupRuntimeData.get) + } + ) unparseEnd(state, isBuild = false) diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala index 1a4eff66a9..ca08d92133 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala @@ -48,13 +48,11 @@ class HiddenGroupCombinatorUnparser(ctxt: ModelGroupRuntimeData, bodyUnparser: U teardown = (start, _) => start.decrementHiddenDef() ) - def unparse(start: UState): Unit = { - try { - start.incrementHiddenDef() - // unparse - bodyUnparser.unparse1(start) - } finally { - start.decrementHiddenDef() - } - } + def unparse(start: UState): Unit = + withPushPop( + start, + setup = _.incrementHiddenDef(), + dispatch = bodyUnparser.unparse1, + teardown = (start, _) => start.decrementHiddenDef() + ) } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala index 851b153561..6bc692202c 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala @@ -60,25 +60,18 @@ case class ComplexNilOrContentUnparser( override def childProcessors = Vector(nilUnparser, contentUnparser) + private def chooseBodyUnparser(node: DINode): Unparser = + if (node.asComplex.isNilled) nilUnparser else contentUnparser + def unparse(state: UState): Unit = { Assert.invariant(Maybe.WithNulls.isDefined(state.currentInfosetNode)) - val inode = state.currentInfosetNode.asComplex - if (inode.isNilled) - nilUnparser.unparse1(state) - else - contentUnparser.unparse1(state) + chooseBodyUnparser(state.currentInfosetNode).unparse1(state) } // Without this override, WriteUnparser dispatch would call // contentUnparser.unparse1 synchronously on write's state, but it can // itself be a resumable group unparser expecting live InfosetInputter // events that don't exist yet (see SpecifiedLengthExplicitImplicitUnparser). - override def writeContent(containerNode: DINode, state: UState): Unit = { - val bodyUnparser = if (containerNode.asComplex.isNilled) { - nilUnparser - } else { - contentUnparser - } - writeWithPushPop(containerNode, bodyUnparser, state) - } + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop(containerNode, chooseBodyUnparser(containerNode), state) } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala index b173d8cda3..c70a18c4d8 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala @@ -71,10 +71,13 @@ final class SpecifiedLengthExplicitImplicitUnparser( } } - override final def unparse(state: UState): Unit = { - checkVariableWidthComplexType(state) - eUnparser.unparse1(state) - } + override final def unparse(state: UState): Unit = + withPushPop( + state, + setup = checkVariableWidthComplexType, + dispatch = eUnparser.unparse1, + teardown = (_, _) => () + ) // Without this, a SeqCompUnparser wrapping this class would treat // eUnparser as a synchronous call via its generic fallback, but @@ -164,11 +167,15 @@ class SpecifiedLengthPrefixedUnparser( override def childProcessors = Vector(prefixedLengthUnparser, eUnparser) - override def unparse(state: UState): Unit = { - val plElem = pushDetachedPrefixLengthElement(state) - eUnparser.unparse1(state) - resolvePrefixLength(state, state.currentInfosetNode.asInstanceOf[DIElement], plElem) - } + override def unparse(state: UState): Unit = + withPushPop( + state, + setup = pushDetachedPrefixLengthElement, + dispatch = eUnparser.unparse1, + teardown = { (state, plElem) => + resolvePrefixLength(state, state.currentInfosetNode.asInstanceOf[DIElement], plElem) + } + ) // Without this, WriteUnparser dispatch (a plain recursive-dispatch // fallback for a group-wrapped eUnparser) would call eUnparser.unparse1 diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala index eb60b2aaed..1a68ed6d4c 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala @@ -34,32 +34,48 @@ trait WriteUnparser { // needed child exists and is ready, then resumes where it left off. def writeContent(containerNode: DINode, state: UState): Unit - // Shared push-once/pop-once skeleton: setup runs before recursing into - // bodyUnparser (dispatched to writeContent if it's a WriteUnparser, else - // plain unparse1), teardown runs once that call returns with setup's - // result (e.g. threading a detached element from setup to teardown). - protected def writeWithPushPop[A]( - containerNode: DINode, - bodyUnparser: Unparser, + // Shared push-once/pop-once skeleton, used by both unparse() and + // writeContent() implementations that need one: setup runs before + // dispatch, teardown runs once dispatch returns (even on exception, + // e.g. a stall caught higher up and followed by finishWriteSide's + // invariant checks against this same state) with setup's result + // threaded through (e.g. a detached element from setup to teardown). + protected def withPushPop[A]( state: UState, setup: UState => A, + dispatch: UState => Unit, teardown: (UState, A) => Unit ): Unit = { val setupResult = setup(state) - // Unlike single-pass unparse(), a stall here is caught higher up and - // followed by finishWriteSide's invariant checks against this same - // state, so teardown must still run, or those checks fail for an - // unrelated reason. try { - bodyUnparser match { - case wu: WriteUnparser => wu.writeContent(containerNode, state) - case _ => bodyUnparser.unparse1(state) - } + dispatch(state) } finally { teardown(state, setupResult) } } + // withPushPop specialized for writeContent's own dispatch: bodyUnparser + // is dispatched to writeContent if it's a WriteUnparser, else plain + // unparse1. + protected def writeWithPushPop[A]( + containerNode: DINode, + bodyUnparser: Unparser, + state: UState, + setup: UState => A, + teardown: (UState, A) => Unit + ): Unit = + withPushPop( + state, + setup = setup, + dispatch = { s => + bodyUnparser match { + case wu: WriteUnparser => wu.writeContent(containerNode, s) + case _ => bodyUnparser.unparse1(s) + } + }, + teardown = teardown + ) + /** * `writeWithPushPop` for a combinator with nothing to push or pop; just * the dispatch-to-`writeContent`-or-`unparse1` part. @@ -73,7 +89,7 @@ trait WriteUnparser { containerNode, bodyUnparser, state, - (_: UState) => (), - (_: UState, _: Unit) => () + setup = (_: UState) => (), + teardown = (_: UState, _: Unit) => () ) } From f0ffbcfde57b3ea50427f2d6871e4cc16e0ad32a Mon Sep 17 00:00:00 2001 From: olabusayoT <50379531+olabusayoT@users.noreply.github.com> Date: Mon, 28 Sep 2026 14:43:29 -0400 Subject: [PATCH 4/6] Hoist per-call closure allocations in withPushPop/runElementContent callers Each unparse()/writeContent() call runs once per matching element in the infoset, so a setup/dispatch/teardown argument built from an eta-expanded instance method or field (e.g. bodyUnparser.unparse1, pushDelimiterScope) closes over this and allocates a fresh closure on every single call, unless the JIT's escape analysis eliminates it across the trait/virtual- dispatch boundary into withPushPop. Hoist each such closure that only captures this/instance fields into a private val computed once per unparser instance (named funcXXX), reused by both writeContent and unparse where the same closure applies to both. A closure that also captures a per-call parameter (containerNode) is left as-is: it cannot be hoisted since containerNode genuinely differs on every call. Affects DelimiterStackUnparser, DynamicEscapeSchemeUnparser, HiddenGroupCombinatorUnparser, SpecifiedLengthExplicitImplicitUnparser, SpecifiedLengthPrefixedUnparser, and ElementUnparserBase's unparse() dispatch closure. No behavior change. DAFFODIL-3065 --- .../ChoiceAndOtherVariousUnparsers.scala | 34 ++++++++++++---- .../unparsers/runtime1/ElementUnparser.scala | 40 +++++++++++-------- .../HiddenGroupCombinatorUnparser.scala | 9 ++++- .../runtime1/SpecifiedLengthUnparsers.scala | 37 ++++++++++++----- 4 files changed, 85 insertions(+), 35 deletions(-) diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala index 07b3b9138f..57ea053ab8 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala @@ -274,6 +274,14 @@ class DelimiterStackUnparser( ) extends CombinatorUnparser(ctxt) with WriteUnparser { + // Hoisted once per instance rather than passed as `pushDelimiterScope`/ + // `bodyUnparser.unparse1` at each call site: an eta-expansion of an + // instance method (or one of its fields) closes over `this`, so it + // allocates a fresh closure on every call otherwise, and both + // writeContent and unparse run once per matching element in the infoset. + private val funcPushDelimiterScope: UState => Unit = pushDelimiterScope + private val funcBodyUnparserUnparse1: UState => Unit = bodyUnparser.unparse1 + /** * Pushes the delimiter scope, recurses into the body, and pops only * once the body's own writeContent (if any) returns, since a pending @@ -284,7 +292,7 @@ class DelimiterStackUnparser( containerNode, bodyUnparser, state, - setup = pushDelimiterScope, + setup = funcPushDelimiterScope, teardown = (state, _) => state.popDelimiters() ) override def nom = "DelimiterStack" @@ -307,8 +315,8 @@ class DelimiterStackUnparser( def unparse(state: UState): Unit = withPushPop( state, - setup = pushDelimiterScope, - dispatch = bodyUnparser.unparse1, + setup = funcPushDelimiterScope, + dispatch = funcBodyUnparserUnparse1, teardown = (state, _) => state.popDelimiters() ) @@ -338,6 +346,16 @@ class DynamicEscapeSchemeUnparser( override val runtimeDependencies = Array(escapeScheme) + // Hoisted once per instance rather than passed at each call site: an + // eta-expansion of an instance method, or a lambda referencing an + // instance field like `escapeScheme`, closes over `this`, so it + // allocates a fresh closure on every call otherwise, and both + // writeContent and unparse run once per matching element in the infoset. + private val funcCacheEscapeScheme: UState => Unit = cacheEscapeScheme + private val funcBodyUnparserUnparse1: UState => Unit = bodyUnparser.unparse1 + private val funcInvalidateCache: (UState, Unit) => Unit = + (state, _) => escapeScheme.invalidateCache(state) + /** * Caches the escape scheme, recurses into the body, and invalidates * the cache only once the body's own writeContent (if any) returns, @@ -348,16 +366,16 @@ class DynamicEscapeSchemeUnparser( containerNode, bodyUnparser, state, - setup = cacheEscapeScheme, - teardown = (state, _) => escapeScheme.invalidateCache(state) + setup = funcCacheEscapeScheme, + teardown = funcInvalidateCache ) def unparse(state: UState): Unit = withPushPop( state, - setup = cacheEscapeScheme, - dispatch = bodyUnparser.unparse1, - teardown = (state, _) => escapeScheme.invalidateCache(state) + setup = funcCacheEscapeScheme, + dispatch = funcBodyUnparserUnparse1, + teardown = funcInvalidateCache ) // Evaluates the dynamic escape scheme in the correct scope; the result is diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala index fce7dc3278..b067b76181 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala @@ -235,6 +235,28 @@ sealed abstract class ElementUnparserBase( dispatchContentUnparser(state) } + // Hoisted once per instance rather than passed inline at the unparse() + // call site below: a closure referencing instance fields/methods (erd, + // runContentUnparser) closes over `this`, so it allocates a fresh + // closure on every call otherwise, and unparse runs once per matching + // element in the infoset. writeContent's own dispatch closure is not + // hoisted the same way: it also captures containerNode, a per-call + // parameter, so a fresh closure there is unavoidable regardless. + private val funcUnparseDispatch: UState => Unit = { s => + // We must push the TermRuntimeData for all model-groups, starting + // from the complex type's model-group; simple types have none to + // push. Only unparse's own event-driven dispatch needs this: + // nextElement, its sole reader, resolves a raw event's tag name + // against it, and writeContent never consumes events at all. + if (erd.isComplexType) + s.pushTRD(erd.optComplexTypeModelGroupRuntimeData.get) + + runContentUnparser(s) + + if (erd.isComplexType) + s.popTRD(erd.optComplexTypeModelGroupRuntimeData.get) + } + /** * The steps identical whether reached via writeContent's already-built * containerNode or unparse's own event-consuming attach: before-content, @@ -312,23 +334,7 @@ sealed abstract class ElementUnparserBase( unparseBegin(state) - runElementContent( - state, - dispatch = { s => - // We must push the TermRuntimeData for all model-groups, starting - // from the complex type's model-group; simple types have none to - // push. Only unparse's own event-driven dispatch needs this: - // nextElement, its sole reader, resolves a raw event's tag name - // against it, and writeContent never consumes events at all. - if (erd.isComplexType) - s.pushTRD(erd.optComplexTypeModelGroupRuntimeData.get) - - runContentUnparser(s) - - if (erd.isComplexType) - s.popTRD(erd.optComplexTypeModelGroupRuntimeData.get) - } - ) + runElementContent(state, dispatch = funcUnparseDispatch) unparseEnd(state, isBuild = false) diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala index ca08d92133..076b89eb26 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala @@ -35,6 +35,13 @@ class HiddenGroupCombinatorUnparser(ctxt: ModelGroupRuntimeData, bodyUnparser: U override val runtimeDependencies = Array() + // Hoisted once per instance rather than passed as `bodyUnparser.unparse1` + // at the unparse() call site below: an eta-expansion of an instance + // field's method closes over `this`, so it allocates a fresh closure on + // every call otherwise, and unparse runs once per matching element in + // the infoset. + private val funcBodyUnparserUnparse1: UState => Unit = bodyUnparser.unparse1 + // The hidden-depth counter must stay incremented across any pauses, since // anything the body writes needs it (e.g. choice-branch resolution // branches on state.withinHiddenNest, and RepType conversion asserts @@ -52,7 +59,7 @@ class HiddenGroupCombinatorUnparser(ctxt: ModelGroupRuntimeData, bodyUnparser: U withPushPop( start, setup = _.incrementHiddenDef(), - dispatch = bodyUnparser.unparse1, + dispatch = funcBodyUnparserUnparse1, teardown = (start, _) => start.decrementHiddenDef() ) } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala index c70a18c4d8..013803b89a 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala @@ -71,11 +71,18 @@ final class SpecifiedLengthExplicitImplicitUnparser( } } + // Hoisted once per instance rather than passed at each call site below: + // an eta-expansion of an instance method (or field) closes over `this`, + // so it allocates a fresh closure on every call otherwise, and both + // writeContent and unparse run once per matching element in the infoset. + private val funcCheckVariableWidthComplexType: UState => Unit = checkVariableWidthComplexType + private val funcEUnparserUnparse1: UState => Unit = eUnparser.unparse1 + override final def unparse(state: UState): Unit = withPushPop( state, - setup = checkVariableWidthComplexType, - dispatch = eUnparser.unparse1, + setup = funcCheckVariableWidthComplexType, + dispatch = funcEUnparserUnparse1, teardown = (_, _) => () ) @@ -88,7 +95,7 @@ final class SpecifiedLengthExplicitImplicitUnparser( containerNode, eUnparser, state, - setup = checkVariableWidthComplexType, + setup = funcCheckVariableWidthComplexType, teardown = (_, _) => () ) } @@ -167,14 +174,26 @@ class SpecifiedLengthPrefixedUnparser( override def childProcessors = Vector(prefixedLengthUnparser, eUnparser) + // Hoisted once per instance rather than passed at the unparse() call + // site below: an eta-expansion of an instance method (or field) closes + // over `this`, so it allocates a fresh closure on every call otherwise, + // and unparse runs once per matching element in the infoset. + // writeContent's own setup/teardown below are NOT hoisted the same way: + // its teardown also captures containerNode, a per-call parameter, so a + // fresh closure there is unavoidable regardless. + private val funcPushDetachedPrefixLengthElement: UState => DISimple = + pushDetachedPrefixLengthElement + private val funcEUnparserUnparse1: UState => Unit = eUnparser.unparse1 + private val funcUnparseTeardown: (UState, DISimple) => Unit = { (state, plElem) => + resolvePrefixLength(state, state.currentInfosetNode.asInstanceOf[DIElement], plElem) + } + override def unparse(state: UState): Unit = withPushPop( state, - setup = pushDetachedPrefixLengthElement, - dispatch = eUnparser.unparse1, - teardown = { (state, plElem) => - resolvePrefixLength(state, state.currentInfosetNode.asInstanceOf[DIElement], plElem) - } + setup = funcPushDetachedPrefixLengthElement, + dispatch = funcEUnparserUnparse1, + teardown = funcUnparseTeardown ) // Without this, WriteUnparser dispatch (a plain recursive-dispatch @@ -186,7 +205,7 @@ class SpecifiedLengthPrefixedUnparser( containerNode, eUnparser, state, - setup = pushDetachedPrefixLengthElement, + setup = funcPushDetachedPrefixLengthElement, teardown = { (state, plElem) => // resolvePrefixLength (via assignPrefixLength/suspension.run) // expects state.processor to already be set, normally done by From bde8363d1135cec50d0b1f5418797b5215499a95 Mon Sep 17 00:00:00 2001 From: olabusayoT <50379531+olabusayoT@users.noreply.github.com> Date: Mon, 28 Sep 2026 18:21:15 -0400 Subject: [PATCH 5/6] Split generated withTunable into chunked part-methods DaffodilTunables.withTunable generated one match over every tunable name in a single method; enough tunables now exist that this crossed the JVM's 64KB-per-method bytecode limit, failing compilation with "Method too large". Generate withTunablePart0..N instead, each holding tunablesPerPart (8) tunables' worth of cases, with a fallthrough to the next part on no match and the final part throwing the same "unknown tunable" error the single match used to. withTunable itself keeps its existing public signature, delegating to withTunablePart0. DAFFODIL-3065 --- .../daffodil/propGen/TunableGenerator.scala | 59 ++++++++++++++----- 1 file changed, 45 insertions(+), 14 deletions(-) diff --git a/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala b/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala index 3f04975a5b..3e44d57583 100644 --- a/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala +++ b/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala @@ -99,15 +99,17 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm | tunables.foldLeft(this) { case (dafTuns, (tunable, value)) => dafTuns.withTunable(tunable, value) } | } | + | // One large match over every tunable name would exceed the JVM's + | // 64KB-per-method bytecode limit as tunables accumulate over time, so + | // this is split into a chain of smaller private methods instead, each + | // falling through to the next on no match. | def withTunable(tunable: String, value: String): DaffodilTunables = { - | tunable match { + | withTunablePart0(tunable, value) + | } + | """.trim.stripMargin val bottom = """ - | case _ => throw new IllegalArgumentException("Unknown tunable: " + tunable) - | } - | } - | | private def throwInvalidTunableValue(tunable: String, value: String) = { | throw new IllegalArgumentException("Invalid value for tunable " + tunable + ": " + value) | } @@ -115,6 +117,11 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm |} """.trim.stripMargin + // How many tunables' worth of case-match bytecode go into each + // withTunablePartN method; small enough to leave headroom against the + // JVM's 64KB-per-method limit as more tunables are added over time. + private val tunablesPerPart = 8 + val tunablesRoot = (schemaRootConfig \ "element").find(_ \@ "name" == "tunables").get val tunableNodes = tunablesRoot \\ "all" \ "element" @@ -166,15 +173,39 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm .map(_.scalaDefinition) .mkString(" ", ",\n ", ")") - val conversionString = - tunables - .map { tunable => - tunable.scalaConversion - .split("\n") - .filter(_.trim.length > 0) - .mkString(" ", "\n ", "") + // Split across several withTunablePartN methods rather than one large + // match (see middle's own comment): each chunk's case bodies, followed + // by a fallthrough to the next chunk's method, or to the final + // "unknown tunable" error on the last chunk. + val tunableChunks = tunables.grouped(tunablesPerPart).toIndexedSeq + val numParts = tunableChunks.length + val partsString = + tunableChunks.zipWithIndex + .map { case (chunk, idx) => + val casesString = + chunk + .map { tunable => + tunable.scalaConversion + .split("\n") + .filter(_.trim.length > 0) + .mkString(" ", "\n ", "") + } + .mkString("\n") + val fallthrough = + if (idx == numParts - 1) + """ case _ => throw new IllegalArgumentException("Unknown tunable: " + tunable)""" + else + s" case _ => withTunablePart${idx + 1}(tunable, value)" + s""" + | private def withTunablePart${idx}(tunable: String, value: String): DaffodilTunables = { + | tunable match { + |${casesString} + |${fallthrough} + | } + | } + """.trim.stripMargin } - .mkString("\n") + .mkString("\n\n") w.write(top) w.write("\n") @@ -182,7 +213,7 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm w.write("\n") w.write(middle) w.write("\n") - w.write(conversionString) + w.write(partsString) w.write("\n") w.write(bottom) w.write("\n") From b244541f6521d7ea05e17870ad7febe76cc9b79e Mon Sep 17 00:00:00 2001 From: olabusayoT <50379531+olabusayoT@users.noreply.github.com> Date: Tue, 29 Sep 2026 10:42:52 -0400 Subject: [PATCH 6/6] TEMP: default useBuildWritePrefetch to true --- .../src/main/resources/org/apache/daffodil/xsd/dafext.xsd | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd b/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd index 1fda1bea74..532d1e0c29 100644 --- a/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd +++ b/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd @@ -534,7 +534,7 @@ - + If true, unparsing uses a build/write-prefetch path instead of the