diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Grammar.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Grammar.scala index 47f1f3740f..d9bdd77ac9 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Grammar.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Grammar.scala @@ -21,9 +21,14 @@ import org.apache.daffodil.core.compiler.ForParser import org.apache.daffodil.core.compiler.ForUnparser import org.apache.daffodil.core.dsom.* import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One import org.apache.daffodil.runtime1.processors.parsers.AssertExpressionEvaluationParser import org.apache.daffodil.runtime1.processors.parsers.NadaParser import org.apache.daffodil.runtime1.processors.parsers.SeqCompParser +import org.apache.daffodil.runtime1.processors.unparsers.Builder +import org.apache.daffodil.runtime1.processors.unparsers.SeqCompBuilder import org.apache.daffodil.runtime1.processors.unparsers.SeqCompUnparser import org.apache.daffodil.unparsers.runtime1.NadaUnparser @@ -124,6 +129,26 @@ class SeqComp private (context: SchemaComponent, children: Seq[Gram]) else if (unparserChildren.length == 1) unparserChildren.head else new SeqCompUnparser(context.runtimeData, unparserChildren.toArray) } + + // Only the (usually at most one) child(ren) that actually build infoset + // content contribute here; write-only children (delimiters, padding, etc.) + // simply have no builder to collect. + lazy val builderChildren: Array[Builder] = { + children + .filter(x => !x.isEmpty && (x.forWhat != ForParser)) + .flatMap { x => x.builder.toList } + .toArray + } + + final override lazy val builder: Maybe[Builder] = { + if (builderChildren.isEmpty) { + Nope + } else if (builderChildren.length == 1) { + One(builderChildren.head) + } else { + One(new SeqCompBuilder(builderChildren)) + } + } } object EmptyGram extends Gram(null) { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Production.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Production.scala index 833b1f71f2..ca526f4317 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Production.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Production.scala @@ -22,7 +22,10 @@ import org.apache.daffodil.core.compiler.ForUnparser import org.apache.daffodil.core.compiler.ParserOrUnparser import org.apache.daffodil.core.dsom.SchemaComponent import org.apache.daffodil.lib.util.Logger +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope import org.apache.daffodil.runtime1.processors.parsers.NadaParser +import org.apache.daffodil.runtime1.processors.unparsers.Builder import org.apache.daffodil.unparsers.runtime1.NadaUnparser /** @@ -107,4 +110,16 @@ final class Prod( else unp } + + final override lazy val builder: Maybe[Builder] = { + if (gram.isEmpty) { + Nope + } else { + (forWhat, gram.forWhat) match { + case (ForParser, _) => Nope + case (_, ForParser) => Nope + case _ => gram.builder + } + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ChoiceCombinator.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ChoiceCombinator.scala index b91428e4ca..d2fd1b0b0a 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ChoiceCombinator.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ChoiceCombinator.scala @@ -29,10 +29,14 @@ import org.apache.daffodil.lib.cookers.ChoiceBranchKeyCooker import org.apache.daffodil.lib.cookers.IntRangeCooker import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.schema.annotation.props.gen.ChoiceLengthKind +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One import org.apache.daffodil.lib.util.MaybeInt import org.apache.daffodil.lib.util.ProperlySerializableMap.* import org.apache.daffodil.runtime1.infoset.ChoiceBranchEvent import org.apache.daffodil.runtime1.processors.RangeBound +import org.apache.daffodil.runtime1.processors.TermRuntimeData import org.apache.daffodil.runtime1.processors.parsers.* import org.apache.daffodil.runtime1.processors.unparsers.* import org.apache.daffodil.unparsers.runtime1.* @@ -332,4 +336,31 @@ case class ChoiceCombinator(ch: ChoiceTermBase, alternatives: Seq[Gram]) new ChoiceCombinatorUnparser(ch.modelGroupRuntimeData, cbm, choiceLengthInBits) } } + + override lazy val builder: Maybe[Builder] = { + val (eventRDMap, optDefaultBranch) = ch.choiceBranchMap + + def builderFor(term: Term): (TermRuntimeData, Builder) = { + val cb = term.termContentBody.builder + val b = if (cb.isDefined) { + cb.get + } else { + EmptyBuilder + } + (term.termRuntimeData, b) + } + + val branchMap: Map[ChoiceBranchEvent, (TermRuntimeData, Builder)] = + eventRDMap.map { case (cbe, branchTerm) => (cbe, builderFor(branchTerm)) } + val defaultBranch: Maybe[(TermRuntimeData, Builder)] = optDefaultBranch match { + case Some(term) => One(builderFor(term)) + case None => Nope + } + + if (branchMap.isEmpty && defaultBranch.isEmpty) { + Nope + } else { + One(new ChoiceBuilder(ch.modelGroupRuntimeData, branchMap, defaultBranch)) + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/DelimiterAndEscapeRelated.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/DelimiterAndEscapeRelated.scala index 6bf946a91d..c49afb7a81 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/DelimiterAndEscapeRelated.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/DelimiterAndEscapeRelated.scala @@ -21,10 +21,12 @@ import org.apache.daffodil.core.dsom.* import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.core.grammar.Terminal import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.lib.util.Misc import org.apache.daffodil.lib.xml.XMLUtils import org.apache.daffodil.runtime1.processors.parsers.{ Parser as DaffodilParser, * } +import org.apache.daffodil.runtime1.processors.unparsers.Builder import org.apache.daffodil.runtime1.processors.unparsers.Unparser as DaffodilUnparser import org.apache.daffodil.unparsers.runtime1.* @@ -55,6 +57,10 @@ case class DelimiterStackCombinatorSequence(sq: SequenceTermBase, body: Gram) override lazy val unparser: DaffodilUnparser = new DelimiterStackUnparser(uInit, uSep, uTerm, sq.termRuntimeData, body.unparser) + + // Delimiters are write-only content; the builder tree skips straight to + // whatever this sequence's body itself builds, if anything. + override lazy val builder: Maybe[Builder] = body.builder } case class DelimiterStackCombinatorChoice(ch: ChoiceTermBase, body: Gram) @@ -77,6 +83,10 @@ case class DelimiterStackCombinatorChoice(ch: ChoiceTermBase, body: Gram) override lazy val unparser: DaffodilUnparser = new DelimiterStackUnparser(uInit, None, uTerm, ch.termRuntimeData, body.unparser) + + // Delimiters are write-only content; the builder tree skips straight to + // whatever this choice's body itself builds, if anything. + override lazy val builder: Maybe[Builder] = body.builder } case class DelimiterStackCombinatorElement(e: ElementBase, body: Gram) @@ -111,6 +121,10 @@ case class DelimiterStackCombinatorElement(e: ElementBase, body: Gram) if (u.isEmpty) u else new DelimiterStackUnparser(uInit, None, uTerm, e.termRuntimeData, u) } + + // Delimiters are write-only content; the builder tree skips straight to + // whatever this element's body itself builds, if anything. + override lazy val builder: Maybe[Builder] = body.builder } case class DynamicEscapeSchemeCombinatorElement(e: ElementBase, body: Gram) @@ -135,4 +149,9 @@ case class DynamicEscapeSchemeCombinatorElement(e: ElementBase, body: Gram) if (u.isEmpty || schemeUnparseIsConstant) u else new DynamicEscapeSchemeUnparser(schemeUnparseOpt.get, e.termRuntimeData, u) } + + // The escape scheme only governs delimiter matching in written bytes; the + // builder tree skips straight to whatever this element's body itself + // builds, if anything. + override lazy val builder: Maybe[Builder] = body.builder } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ElementCombinator.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ElementCombinator.scala index 0dcc7ad5cd..a2e27e860b 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ElementCombinator.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ElementCombinator.scala @@ -28,6 +28,8 @@ import org.apache.daffodil.lib.schema.annotation.props.gen.LengthKind import org.apache.daffodil.lib.schema.annotation.props.gen.Representation import org.apache.daffodil.lib.schema.annotation.props.gen.TestKind import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One import org.apache.daffodil.runtime1.processors.parsers.CaptureEndOfContentLengthParser import org.apache.daffodil.runtime1.processors.parsers.CaptureEndOfValueLengthParser import org.apache.daffodil.runtime1.processors.parsers.CaptureStartOfContentLengthParser @@ -36,6 +38,8 @@ import org.apache.daffodil.runtime1.processors.parsers.ElementParser import org.apache.daffodil.runtime1.processors.parsers.ElementParserInputValueCalc import org.apache.daffodil.runtime1.processors.parsers.NadaParser import org.apache.daffodil.runtime1.processors.parsers.Parser +import org.apache.daffodil.runtime1.processors.unparsers.Builder +import org.apache.daffodil.runtime1.processors.unparsers.ElementBuilder import org.apache.daffodil.runtime1.processors.unparsers.Unparser import org.apache.daffodil.unparsers.runtime1.CaptureEndOfContentLengthUnparser import org.apache.daffodil.unparsers.runtime1.CaptureEndOfValueLengthUnparser @@ -44,6 +48,7 @@ import org.apache.daffodil.unparsers.runtime1.CaptureStartOfValueLengthUnparser import org.apache.daffodil.unparsers.runtime1.ElementOVCSpecifiedLengthUnparser import org.apache.daffodil.unparsers.runtime1.ElementOVCUnspecifiedLengthUnparser import org.apache.daffodil.unparsers.runtime1.ElementSpecifiedLengthUnparser +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase import org.apache.daffodil.unparsers.runtime1.ElementUnparserInputValueCalc import org.apache.daffodil.unparsers.runtime1.ElementUnspecifiedLengthUnparser import org.apache.daffodil.unparsers.runtime1.ElementUnusedUnparser @@ -143,6 +148,37 @@ class ElementCombinator( } } + private lazy val eBuilder: Maybe[Builder] = { + if (eValue.isEmpty) { + Nope + } else { + eValue.builder + } + } + private lazy val eReptypeBuilder: Maybe[Builder] = repTypeElementGram.builder + + // Reuses the same already-memoized instance above for unparseBegin/ + // unparseEnd, so build() and write() see identical node-creation + // behavior; the third branch's builder is whatever subComb builds. + override lazy val builder: Maybe[Builder] = unparser match { + case eu @ (_: ElementOVCSpecifiedLengthUnparser | _: ElementSpecifiedLengthUnparser) => { + val eub = eu.asInstanceOf[ElementUnparserBase] + val contentBuilder = if (eReptypeBuilder.isDefined) { + eReptypeBuilder + } else { + eBuilder + } + One( + new ElementBuilder( + context.erd, + eub.unparseBeginForBuild, + eub.unparseEndForBuild, + contentBuilder + ) + ) + } + case _ => subComb.builder + } } case class ElementUnused(ctxt: ElementBase) @@ -374,6 +410,26 @@ class ElementParseAndUnspecifiedLength( new ElementUnparserInputValueCalc(context.erd, uSetVar) } } + + // Reuses the same already-memoized ElementUnparserBase instance above for + // unparseBegin/unparseEnd, so build() and write() see identical + // nilled/OVC/IVC node-creation behavior. + override lazy val builder: Maybe[Builder] = { + val eu = unparser.asInstanceOf[ElementUnparserBase] + val contentBuilder = if (eRepTypeBuilder.isDefined) { + eRepTypeBuilder + } else { + eBuilder + } + One( + new ElementBuilder( + context.erd, + eu.unparseBeginForBuild, + eu.unparseEndForBuild, + contentBuilder + ) + ) + } } abstract class ElementCombinatorBase( @@ -449,4 +505,8 @@ abstract class ElementCombinatorBase( def unparser: Unparser + lazy val eBuilder: Maybe[Builder] = eGram.builder + + lazy val eRepTypeBuilder: Maybe[Builder] = repTypeElementGram.builder + } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/HiddenGroupCombinator.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/HiddenGroupCombinator.scala index a993c98cfe..1de1cc37c9 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/HiddenGroupCombinator.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/HiddenGroupCombinator.scala @@ -20,8 +20,13 @@ package org.apache.daffodil.core.grammar.primitives import org.apache.daffodil.core.dsom.ModelGroup import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.core.grammar.Terminal +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One import org.apache.daffodil.runtime1.processors.parsers.HiddenGroupCombinatorParser import org.apache.daffodil.runtime1.processors.parsers.Parser +import org.apache.daffodil.runtime1.processors.unparsers.Builder +import org.apache.daffodil.runtime1.processors.unparsers.HiddenGroupBuilder import org.apache.daffodil.runtime1.processors.unparsers.Unparser import org.apache.daffodil.unparsers.runtime1.HiddenGroupCombinatorUnparser @@ -34,4 +39,12 @@ final class HiddenGroupCombinator(ctxt: ModelGroup, body: Gram) override lazy val unparser: Unparser = new HiddenGroupCombinatorUnparser(ctxt.modelGroupRuntimeData, body.unparser) + override lazy val builder: Maybe[Builder] = { + val bb = body.builder + if (bb.isEmpty) { + Nope + } else { + One(new HiddenGroupBuilder(bb.get)) + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/LayeredSequence.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/LayeredSequence.scala index 2b75da3f4b..f86a08f97f 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/LayeredSequence.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/LayeredSequence.scala @@ -20,9 +20,14 @@ package org.apache.daffodil.core.grammar.primitives import org.apache.daffodil.core.dsom.* import org.apache.daffodil.core.grammar.Terminal import org.apache.daffodil.core.layers.LayerSchemaCompiler +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One import org.apache.daffodil.lib.util.Misc import org.apache.daffodil.runtime1.processors.parsers.LayeredSequenceParser import org.apache.daffodil.runtime1.processors.parsers.Parser as DaffodilParser +import org.apache.daffodil.runtime1.processors.unparsers.Builder +import org.apache.daffodil.runtime1.processors.unparsers.SequenceBuilder import org.apache.daffodil.runtime1.processors.unparsers.Unparser as DaffodilUnparser import org.apache.daffodil.unparsers.runtime1.LayeredSequenceUnparser @@ -47,4 +52,18 @@ case class LayeredSequence(sq: SequenceGroupTermBase, bodyTerm: SequenceChild) override lazy val unparser: DaffodilUnparser = new LayeredSequenceUnparser(srd, bodyUnparser) + + // The layer transform itself is a write-only, byte-level concern, but + // (unlike delimiters/escape schemes) this is still a genuine one-child + // sequence position: it must push/pop bodyTerm's TRD and advance the + // group index the same way SequenceBuilder does for any other sequence + // child, or next-element resolution on the shared InfosetInputter breaks. + override lazy val builder: Maybe[Builder] = { + val info = bodyTerm.optSequenceChildBuildInfo + if (info.isEmpty) { + Nope + } else { + One(new SequenceBuilder(IndexedSeq(info.get))) + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/NilEmptyCombinators.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/NilEmptyCombinators.scala index 1ed7a027bc..59d784f16a 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/NilEmptyCombinators.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/NilEmptyCombinators.scala @@ -21,8 +21,13 @@ import org.apache.daffodil.core.dsom.ElementBase import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.core.grammar.Terminal import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One import org.apache.daffodil.runtime1.processors.parsers.ComplexNilOrContentParser import org.apache.daffodil.runtime1.processors.parsers.SimpleNilOrValueParser +import org.apache.daffodil.runtime1.processors.unparsers.Builder +import org.apache.daffodil.runtime1.processors.unparsers.NilOrContentBuilder import org.apache.daffodil.unparsers.runtime1.ComplexNilOrContentUnparser import org.apache.daffodil.unparsers.runtime1.SimpleNilOrValueUnparser @@ -59,4 +64,15 @@ case class ComplexNilOrContent(ctxt: ElementBase, nilGram: Gram, contentGram: Gr override lazy val unparser = ComplexNilOrContentUnparser(ctxt.erd, nilUnparser, contentUnparser) + // A nilled complex element has no children to build; nilled-ness is only + // known once the node exists, so this needs a real runtime check, not a + // static pass-through. + override lazy val builder: Maybe[Builder] = { + val cb = contentGram.builder + if (cb.isEmpty) { + Nope + } else { + One(new NilOrContentBuilder(cb.get)) + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceChild.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceChild.scala index 725eff5363..6aaf594e05 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceChild.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceChild.scala @@ -24,8 +24,12 @@ import org.apache.daffodil.lib.schema.annotation.props.SeparatorSuppressionPolic import org.apache.daffodil.lib.schema.annotation.props.gen.LengthKind import org.apache.daffodil.lib.schema.annotation.props.gen.OccursCountKind import org.apache.daffodil.lib.schema.annotation.props.gen.Representation +import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.runtime1.dpath.NodeInfo import org.apache.daffodil.runtime1.processors.parsers.* +import org.apache.daffodil.runtime1.processors.unparsers.Builder +import org.apache.daffodil.runtime1.processors.unparsers.EmptyBuilder +import org.apache.daffodil.runtime1.processors.unparsers.SequenceChildBuildInfo import org.apache.daffodil.unparsers.runtime1.* /** @@ -69,6 +73,7 @@ abstract class SequenceChild(protected val sq: SequenceTermBase, child: Term, gr protected lazy val childParser = child.termContentBody.parser protected lazy val childUnparser = child.termContentBody.unparser + protected lazy val childBuilder: Maybe[Builder] = child.termContentBody.builder final override lazy val parser = sequenceChildParser final override lazy val unparser = sequenceChildUnparser @@ -82,6 +87,19 @@ abstract class SequenceChild(protected val sq: SequenceTermBase, child: Term, gr final lazy val optSequenceChildUnparser: Option[SequenceChildUnparser] = if (childUnparser.isEmpty) None else Some(unparser) + final lazy val optSequenceChildBuildInfo: Option[SequenceChildBuildInfo] = { + if (childUnparser.isEmpty) { + None + } else { + val cb = if (childBuilder.isDefined) { + childBuilder.get + } else { + EmptyBuilder + } + Some(SequenceChildBuildInfo(unparser, cb)) + } + } + /** * There's only parse result helpers here, so let's abbreviate */ diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceCombinator.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceCombinator.scala index aab310aa2c..7d64103def 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceCombinator.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceCombinator.scala @@ -124,6 +124,15 @@ class OrderedSequence(sq: SequenceTermBase, sequenceChildrenArg: Seq[SequenceChi } } } + + override lazy val builder: Maybe[Builder] = { + val childBuildInfos = sequenceChildren.flatMap { _.optSequenceChildBuildInfo } + if (childBuildInfos.isEmpty) { + Maybe.Nope + } else { + Maybe.One(new SequenceBuilder(childBuildInfos.toIndexedSeq)) + } + } } class UnorderedSequence( @@ -220,4 +229,13 @@ class UnorderedSequence( } } } + + override lazy val builder: Maybe[Builder] = { + val childBuildInfos = sequenceChildren.flatMap { _.optSequenceChildBuildInfo } + if (childBuildInfos.isEmpty) { + Maybe.Nope + } else { + Maybe.One(new SequenceBuilder(childBuildInfos.toIndexedSeq)) + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SpecifiedLength.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SpecifiedLength.scala index 0ae3e6a4d9..7ed2b65a52 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SpecifiedLength.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SpecifiedLength.scala @@ -24,6 +24,7 @@ import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.core.grammar.Terminal import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.schema.annotation.props.gen.LengthUnits +import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.runtime1.dpath.NodeInfo.PrimType import org.apache.daffodil.runtime1.processors.parsers.* import org.apache.daffodil.runtime1.processors.unparsers.* @@ -44,6 +45,11 @@ abstract class SpecifiedLengthCombinatorBase(val e: ElementBase, eGramArg: => Gr u } + // None of the length-kind wrapping below (explicit/implicit/prefixed + // lengths, pattern matching) creates infoset nodes; the builder tree skips + // straight to whatever the wrapped element content itself builds. + override lazy val builder: Maybe[Builder] = eGram.builder + def kind: String def toBriefXML(depthLimit: Int = -1): String = { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ElementBaseRuntime1Mixin.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ElementBaseRuntime1Mixin.scala index b17c2b4d15..26c29fb488 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ElementBaseRuntime1Mixin.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ElementBaseRuntime1Mixin.scala @@ -23,10 +23,12 @@ import org.apache.daffodil.core.dsom.PrefixLengthQuasiElementDecl import org.apache.daffodil.core.dsom.PrimitiveType import org.apache.daffodil.core.dsom.Root import org.apache.daffodil.core.dsom.SimpleTypeDefBase +import org.apache.daffodil.core.dsom.TransitiveClosureSchemaComponents import org.apache.daffodil.lib.schema.annotation.props.gen.LengthKind import org.apache.daffodil.lib.schema.annotation.props.gen.Representation.Text import org.apache.daffodil.lib.util.Delay import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.runtime1.dpath.SuspendableExpression import org.apache.daffodil.runtime1.dsom.DPathElementCompileInfo import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.RuntimeData @@ -134,6 +136,20 @@ trait ElementBaseRuntime1Mixin { self: ElementBase => isReferenced || mightHaveSuspensions } + /** + * True if a component reachable from this element has a + * dfdl:outputValueCalc resolvable without writing; gates + * useBuildWritePrefetch. Scoped to a closure seeded from this + * element, not schemaSet.allSchemaComponents, which is shared and + * schema-set-wide and would leak another root's OVC into this one. + */ + final lazy val hasAnyPrefetchBeneficialOVC: Boolean = + TransitiveClosureSchemaComponents(Seq(this)).exists { + case e: ElementBase if e.isOutputValueCalc => + SuspendableExpression.isPrefetchBeneficial(e.ovcCompiledExpression) + case _ => false + } + final override lazy val dpathCompileInfo = dpathElementCompileInfo /** diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/GramRuntime1Mixin.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/GramRuntime1Mixin.scala index 54b45c8148..5d685f28b6 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/GramRuntime1Mixin.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/GramRuntime1Mixin.scala @@ -21,6 +21,7 @@ import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.runtime1.processors.parsers.Parser +import org.apache.daffodil.runtime1.processors.unparsers.Builder import org.apache.daffodil.runtime1.processors.unparsers.Unparser trait GramRuntime1Mixin { self: Gram => @@ -59,4 +60,13 @@ trait GramRuntime1Mixin { self: Gram => else Maybe(u) } } + + /** + * Provides this Gram's node in the much smaller, dedicated Builder tree + * that parallels the full Unparser tree (see Builder's own doc). Most + * Grams contribute nothing to the infoset's structure and simply inherit + * this Nope default; only Grams that create or select infoset content + * override it. + */ + def builder: Maybe[Builder] = Maybe.Nope } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala index 3e9c5634fe..b827547363 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala @@ -21,6 +21,8 @@ import org.apache.daffodil.core.dsom.SchemaSet import org.apache.daffodil.core.dsom.SequenceTermBase import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Logger +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope import org.apache.daffodil.runtime1.iapi.DFDL import org.apache.daffodil.runtime1.layers.LayerRuntimeCompiler import org.apache.daffodil.runtime1.layers.LayerRuntimeData @@ -29,6 +31,7 @@ import org.apache.daffodil.runtime1.processors.Processor import org.apache.daffodil.runtime1.processors.SchemaSetRuntimeData import org.apache.daffodil.runtime1.processors.VariableMap import org.apache.daffodil.runtime1.processors.parsers.NotParsableParser +import org.apache.daffodil.runtime1.processors.unparsers.Builder import org.apache.daffodil.runtime1.processors.unparsers.NotUnparsableUnparser trait SchemaSetRuntime1Mixin { @@ -61,6 +64,20 @@ trait SchemaSetRuntime1Mixin { unp }.value + // Unlike parser/unparser, not forced eagerly: onPath below only + // references this when tunable.useBuildWritePrefetch is on, since + // there's no public API to reuse this SchemaSet's ssrd under a + // different tunable (tunable is fixed at compile time, see + // Compiler().withTunables), so schemas that never enable it never + // pay to construct the Builder tree. + lazy val builder: Maybe[Builder] = { + if (generateUnparser) { + root.document.builder + } else { + Nope + } + } + private lazy val layerRuntimeCompiler = new LayerRuntimeCompiler private lazy val allLayers: Seq[LayerRuntimeData] = LV(Symbol("allLayers")) { @@ -84,14 +101,23 @@ trait SchemaSetRuntime1Mixin { // null parser/unparser, and that it's impossible for a DataProcessor // to have an error Assert.invariant(!root.isError) + // Only reference builder/hasAnyPrefetchBeneficialOVC (both real + // work: a full parallel Builder tree, a schema component scan) when + // prefetch will actually be used for this schema; a schema with no + // prefetch-beneficial OVC never reaches unparseViaBuildThenWrite (see + // DataProcessor.unparse), so building the Builder tree for it would + // be pure waste even with the tunable globally on. + val isPrefetchInUse = tunable.useBuildWritePrefetch && root.hasAnyPrefetchBeneficialOVC val ssrd = new SchemaSetRuntimeData( parser, unparser, + if (isPrefetchInUse) builder else Nope, root.elementRuntimeData, variableMap, allLayers, - layerRuntimeCompiler + layerRuntimeCompiler, + isPrefetchInUse ) if (root.numComponents > root.numUniqueComponents) Logger.log.debug( diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala index bc70863fe3..63dc5e759b 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala @@ -18,6 +18,7 @@ package org.apache.daffodil.core.util import java.io.ByteArrayInputStream +import java.io.ByteArrayOutputStream import java.io.InputStream import java.net.URL import java.nio.channels.Channels @@ -43,10 +44,21 @@ import org.apache.daffodil.lib.iapi.* import org.apache.daffodil.lib.util.* import org.apache.daffodil.lib.xml.* import org.apache.daffodil.runtime1.iapi.DFDL +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.runtime1.infoset.DIComplex +import org.apache.daffodil.runtime1.infoset.DIDocument +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.infoset.InfosetInputter import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetOutputter import org.apache.daffodil.runtime1.processors.DataProcessor +import org.apache.daffodil.runtime1.processors.SuspensionTracker import org.apache.daffodil.runtime1.processors.VariableMap +import org.apache.daffodil.runtime1.processors.unparsers.BuildFinished +import org.apache.daffodil.runtime1.processors.unparsers.UState +import org.apache.daffodil.runtime1.processors.unparsers.UStateMain +import org.apache.daffodil.runtime1.processors.unparsers.UnparseSharedContext +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase import org.apache.commons.io.FileUtils /* @@ -118,9 +130,11 @@ object TestUtils { testSchema: scala.xml.Elem, infosetXML: Node, unparseTo: String, - areTracing: Boolean = false + areTracing: Boolean = false, + tunables: Map[String, String] = Map.empty ): java.util.List[api.Diagnostic] = { - val compiler = Compiler().withTunable("allowExternalPathExpressions", "true") + val compiler = + Compiler().withTunable("allowExternalPathExpressions", "true").withTunables(tunables) val pf = compiler.compileNode(testSchema) if (pf.isError) throwDiagnostics(pf.getDiagnostics) var u = saveAndReload(pf.onPath("/").asInstanceOf[DataProcessor]) @@ -198,6 +212,138 @@ object TestUtils { p } + /** + * Compiles testSchema with the given tunables and returns the resulting + * DataProcessor, with no saveAndReload round-trip (unlike compileSchema) + * since some callers build test-only state directly off the live object. + */ + def compileForUnparse( + testSchema: Node, + tunables: Map[String, String] = Map.empty + ): DataProcessor = { + val pf = Compiler().withTunables(tunables).compileNode(testSchema) + if (pf.isError) throwDiagnostics(pf.getDiagnostics) + val dp = pf.onPath("/").asInstanceOf[DataProcessor] + if (dp.isError) throwDiagnostics(dp.getDiagnostics) + dp + } + + /** + * Builds a fresh InfosetInputter walking infosetXML against dp, already + * initialized (root TRD pushed) the same way a real unparse would. + */ + def newInitializedInputter(infosetXML: Node, dp: DataProcessor): InfosetInputter = { + val inputter = new InfosetInputter(new ScalaXMLInfosetInputter(infosetXML)) + inputter.initialize(dp.ssrd.elementRuntimeData, dp.tunables) + inputter + } + + /** + * Unparses infosetXML against dp, throwing if the result is an error, + * and returns the raw unparsed bytes. + */ + def unparseToBytes(dp: DataProcessor, infosetXML: Node): Array[Byte] = { + val out = new ByteArrayOutputStream() + val res = dp.unparse(new ScalaXMLInfosetInputter(infosetXML), out) + if (res.isError) throwDiagnostics(res.getDiagnostics) + out.toByteArray + } + + /** + * Compiles testSchema twice - once with default tunables, once with + * useBuildWritePrefetch (plus any extraTunables) - and unparses + * infosetXML both ways. Returns (singlePassBytes, prefetchBytes) for + * the caller to assert equality (and any expected-string checks) on. + */ + def getSinglePassAndPrefetchBytes( + testSchema: Node, + infosetXML: Node, + extraTunables: Map[String, String] = Map.empty + ): (Array[Byte], Array[Byte]) = { + val singlePassDp = compileForUnparse(testSchema) + val singlePassBytes = unparseToBytes(singlePassDp, infosetXML) + + val prefetchDp = + compileForUnparse(testSchema, extraTunables + ("useBuildWritePrefetch" -> "true")) + val prefetchBytes = unparseToBytes(prefetchDp, infosetXML) + + (singlePassBytes, prefetchBytes) + } + + /** + * Single-pass check for write's writeContent path: unparses infosetXML + * via dp's normal single-pass unparse (dp must have releaseUnneededInfoset + * disabled, so the built tree survives) to get single-pass bytes and an + * intact tree, then re-walks that same tree through a fresh write-only + * UState's writeContent, draining suspensions and finalizing the same way + * DataProcessor's write side does. Returns (singlePassBytes, walkerBytes) + * for the caller to assert equality on. + */ + def getSinglePassAndWriteContentBytes( + dp: DataProcessor, + infosetXML: Node, + prefetchLimit: Long = 1000 + ): (Array[Byte], Array[Byte]) = { + val singlePassOut = new ByteArrayOutputStream() + val singlePassResult = dp.unparse(new ScalaXMLInfosetInputter(infosetXML), singlePassOut) + if (singlePassResult.isError) throwDiagnostics(singlePassResult.getDiagnostics) + val singlePassBytes = singlePassOut.toByteArray + + val ustate = singlePassResult.resultState.asInstanceOf[UStateMain] + val builtTree: DIDocument = ustate.documentElement + + val walkerOut = new ByteArrayOutputStream() + val writeInputter = newInitializedInputter(infosetXML, dp) + val writeState = UState.createInitialUState(walkerOut, dp, writeInputter, false) + writeState.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + val sharedCtx = new UnparseSharedContext( + new SuspensionTracker( + dp.tunables.unparseSuspensionWaitYoung, + dp.tunables.unparseSuspensionWaitOld + ), + dp, + dp.tunables, + prefetchLimit + ) + sharedCtx.observeBuildSignal(BuildFinished) + writeState.setSharedContext(sharedCtx) + + primeLeadCounter(sharedCtx, builtTree.child(0)) + + val rootUnparser = dp.ssrd.unparser.asInstanceOf[ElementUnparserBase] + val rootNode = sharedCtx.awaitChild(builtTree, 0) + rootUnparser.writeContent(rootNode, writeState) + writeState.evalSuspensions(isFinal = true) + writeState.getDataOutputStream.setFinished(writeState) + + (singlePassBytes, walkerOut.toByteArray) + } + + /** + * Pre-increments UnparseSharedContext's lead counter once per element in + * node's subtree, for a tree built outside BuildState (writeContent's + * decrementLead call requires the counter already be symmetric). + */ + private def primeLeadCounter(sharedCtx: UnparseSharedContext, node: DINode): Unit = + node match { + case complex: DIComplex => + sharedCtx.incrementLead() + var i = 0 + while (i < complex.numChildren) { + primeLeadCounter(sharedCtx, complex.child(i)) + i += 1 + } + case array: DIArray => + var i = 0 + while (i < array.numChildren) { + primeLeadCounter(sharedCtx, array.child(i)) + i += 1 + } + case _ => + sharedCtx.incrementLead() + } + private def runSchemaOnRBC( testSchema: Node, data: ReadableByteChannel, diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/Coroutines.scala b/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/Coroutines.scala index 74d7d3fdb8..ddd4175227 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/Coroutines.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/Coroutines.scala @@ -91,6 +91,9 @@ trait Coroutine[T] { private var thread_ : Option[Future[Unit]] = None + /** True once this coroutine's own thread has actually been created. */ + final def isStarted: Boolean = thread_.isDefined + private final def init(): Unit = { if (!isMain && thread_.isEmpty) { val thr = Future { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/MStack.scala b/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/MStack.scala index 6c25a867bb..f2823f0561 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/MStack.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/MStack.scala @@ -24,6 +24,12 @@ import Maybe.* object MStack { final case class Mark(val v: Int) extends AnyVal val nullMark = Mark(0) + + /** + * Off by default: growing past initialSize isn't itself wrong, so paying + * this bookkeeping cost on every push isn't worth it normally. + */ + final val trackMaxSizeReached: Boolean = false } /** @@ -33,35 +39,25 @@ object MStack { * catches improper initialization. These were not initializing properly, * so the idiom evolved to use the scala initializers. */ -final class MStackOfBoolean private () - extends MStack[Boolean]((n: Int) => new Array[Boolean](n), false) +final class MStackOfBoolean private (initialSize: Int) + extends MStack[Boolean]((n: Int) => new Array[Boolean](n), false, initialSize) object MStackOfBoolean { - def apply() = { - val stk = new MStackOfBoolean() - stk.init() - stk - } + def apply(initialSize: Int = 32) = new MStackOfBoolean(initialSize) } -final class MStackOfInt extends MStack[Int]((n: Int) => new Array[Int](n), 0) +final class MStackOfInt(initialSize: Int) + extends MStack[Int]((n: Int) => new Array[Int](n), 0, initialSize) object MStackOfInt { - def apply() = { - val stk = new MStackOfInt() - stk.init() - stk - } + def apply(initialSize: Int = 32) = new MStackOfInt(initialSize) } -final class MStackOfLong extends MStack[Long]((n: Int) => new Array[Long](n), 0L) +final class MStackOfLong(initialSize: Int) + extends MStack[Long]((n: Int) => new Array[Long](n), 0L, initialSize) object MStackOfLong { - def apply() = { - val stk = new MStackOfLong() - stk.init() - stk - } + def apply(initialSize: Int = 32) = new MStackOfLong(initialSize) } /** @@ -75,11 +71,11 @@ object MStackOfLong { * So we use an Array[AnyRef] as the representation here, and we * convert null to Nope, and an actual object reference to One(x) */ -final class MStackOfMaybe[T <: AnyRef] { +final class MStackOfMaybe[T <: AnyRef](initialSize: Int = 32) { override def toString = delegate.toString - private val delegate = new MStackOf[T] + private val delegate = new MStackOf[T](initialSize) private val nullT = null.asInstanceOf[T] def copyFrom(other: MStackOfMaybe[T]) = delegate.copyFrom(other.delegate) @@ -120,6 +116,7 @@ final class MStackOfMaybe[T <: AnyRef] { def toListMaybe = delegate.toList.map { (x: AnyRef) => Maybe(x) // Scala compiler bug without this cast } + def maxSizeReached = delegate.maxSizeReached } /** @@ -135,7 +132,7 @@ final class MStackOfMaybe[T <: AnyRef] { * an object reference or null, and call Maybe(thing) explicitly outside the * iteration. Maybe(null) is Nope, and Maybe(thing) is One(thing) if thing is not null. */ -final class MStackOf[T <: AnyRef] extends Serializable { +final class MStackOf[T <: AnyRef](initialSize: Int = 32) extends Serializable { override def toString = delegate.toString @@ -143,7 +140,7 @@ final class MStackOf[T <: AnyRef] extends Serializable { @inline final def length = delegate.length - private val delegate = MStackOfAnyRef() + private val delegate = MStackOfAnyRef(initialSize) @inline final def mark = delegate.mark @inline final def reset(m: MStack.Mark) = delegate.reset(m) @@ -156,6 +153,7 @@ final class MStackOf[T <: AnyRef] extends Serializable { @inline final def isEmpty = delegate.isEmpty def clear() = delegate.clear() def toList = delegate.toList + def maxSizeReached = delegate.maxSizeReached def iterator = delegate.iterator.asInstanceOf[ResettableIterator[T]] @@ -163,15 +161,15 @@ final class MStackOf[T <: AnyRef] extends Serializable { } -private[util] final class MStackOfAnyRef private () - extends MStack[AnyRef]((n: Int) => new Array[AnyRef](n), null.asInstanceOf[AnyRef]) +private[util] final class MStackOfAnyRef private (initialSize: Int) + extends MStack[AnyRef]( + (n: Int) => new Array[AnyRef](n), + null.asInstanceOf[AnyRef], + initialSize + ) object MStackOfAnyRef { - def apply() = { - val stk = new MStackOfAnyRef() - stk.init() - stk - } + def apply(initialSize: Int = 32) = new MStackOfAnyRef(initialSize) } /** @@ -184,16 +182,23 @@ object MStackOfAnyRef { */ protected abstract class MStack[@specialized T] private[util] ( arrayAllocator: (Int) => Array[T], - nullValue: T + nullValue: T, + initialSize: Int = 32 ) { private var index = 0 - private var table: Array[T] = null + private var table: Array[T] = arrayAllocator(initialSize) - def init(): Unit = { - index = 0 - table = arrayAllocator(32) - } + private var maxSizeReached_ = 0 + + /** + * The largest this stack's length has ever grown to, across its + * whole lifetime (not just its current length; pops don't reduce + * this). Diagnostic only: useful for profiling to check whether a + * particular use's initialSize is well-chosen, not for any runtime + * decision. Always 0 unless MStack.trackMaxSizeReached is enabled. + */ + final def maxSizeReached: Int = maxSizeReached_ def copyFrom(other: MStack[T]): Unit = { this.index = other.index @@ -212,6 +217,9 @@ protected abstract class MStack[@specialized T] private[util] ( } } + if (MStack.trackMaxSizeReached && other.maxSizeReached_ > this.maxSizeReached_) { + this.maxSizeReached_ = other.maxSizeReached_ + } } // private var currentIteratorIndex = -1 @@ -254,6 +262,9 @@ protected abstract class MStack[@specialized T] private[util] ( if (index == table.length) table = growArray(table) table(index) = x index += 1 + if (MStack.trackMaxSizeReached && index > maxSizeReached_) { + maxSizeReached_ = index + } } /** diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/SuspendableExpression.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/SuspendableExpression.scala index 0d9decbd98..b599aa1215 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/SuspendableExpression.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/SuspendableExpression.scala @@ -32,12 +32,31 @@ import org.apache.daffodil.runtime1.processors.unparsers.UState * dfdl:setVariable expressions (which variables are in-turn used by * dfdl:outputValueCalc. */ +object SuspendableExpression { + + /** True unless expr calls valueLength/contentLength (which needs a + * written DOS position, not merely a known value); other reads + * resolve once the value is known. A static property of expr alone; + * it doesn't know whether the referenced position is already written. */ + def canResolveWithoutWriting(expr: CompiledExpression[AnyRef]): Boolean = + expr.valueReferencedElementInfos.isEmpty && expr.contentReferencedElementInfos.isEmpty + + /** True only for OVCs that can actually create a suspension worth + * prefetching: a constant expression never suspends in the first place, + * so it derives no benefit from racing build ahead of write. */ + def isPrefetchBeneficial(expr: CompiledExpression[AnyRef]): Boolean = + !expr.isConstant && canResolveWithoutWriting(expr) +} + trait SuspendableExpression extends Suspension { override val isReadOnly = true protected def expr: CompiledExpression[AnyRef] + override def canResolveWithoutWriting: Boolean = + SuspendableExpression.canResolveWithoutWriting(expr) + override def toString = "SuspendableExpression(" + rd.diagnosticDebugName + ", expr=" + expr.prettyExpr + ")" diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetImpl.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetImpl.scala index 5edb5196e9..7c83efe866 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetImpl.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetImpl.scala @@ -176,13 +176,13 @@ sealed trait DINode { private var _isFinal: Boolean = false /** - * Use to mark a node as final, indicating that its value will not change or have - * any children added to it. Setting an element as final does not preclude it from - * being discarded by backtracking, i.e. it is only locally final, but might still - * be inside an enclosing PoU. - * - * This cannot be called if an element is already marked as final to help ensure - * correct use. + * Use to mark a node as final, indicating that its value will not change or have + * any children added to it. Setting an element as final does not preclude it from + * being discarded by backtracking, i.e. it is only locally final, but might still + * be inside an enclosing PoU. + * + * This cannot be called if an element is already marked as final to help ensure + * correct use. */ def setFinal(): Unit = { Assert.invariant(!_isFinal) @@ -1334,6 +1334,13 @@ final class DIArray( final def freeChildIfNoLongerNeeded(index: Int, doFree: Boolean): Unit = { val node = _contents(index) + // A null slot here means write's coroutine already freed it before + // build's own (always doFree=false, see UState.releaseUnneededInfoset) + // redundant call arrived; a doFree=true call hitting null is a real bug. + if (node == null) { + Assert.invariant(!doFree) + return + } if (!node.erd.dpathElementCompileInfo.isReferencedByExpressions) { if (doFree) { // set to null so that the garbage collector can free this node @@ -1825,6 +1832,13 @@ sealed class DIComplex(override val erd: ElementRuntimeData) def freeChildIfNoLongerNeeded(index: Int, doFree: Boolean): Unit = { val node = child(index) + // A null slot here means write's coroutine already freed it before + // build's own (always doFree=false, see UState.releaseUnneededInfoset) + // redundant call arrived; a doFree=true call hitting null is a real bug. + if (node == null) { + Assert.invariant(!doFree) + return + } if (!node.erd.dpathElementCompileInfo.isReferencedByExpressions) { if (doFree) { // set to null so that the garbage collector can free this node diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala index 3b16d3680e..15ec0db9b9 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala @@ -68,8 +68,20 @@ import org.apache.daffodil.runtime1.infoset.XMLTextInfosetOutputter import org.apache.daffodil.runtime1.processors.parsers.PState import org.apache.daffodil.runtime1.processors.parsers.ParseError import org.apache.daffodil.runtime1.processors.parsers.Parser +import org.apache.daffodil.runtime1.processors.unparsers.AwaitChildStalledException +import org.apache.daffodil.runtime1.processors.unparsers.BuildCoroutine +import org.apache.daffodil.runtime1.processors.unparsers.BuildFinished +import org.apache.daffodil.runtime1.processors.unparsers.BuildSignal +import org.apache.daffodil.runtime1.processors.unparsers.BuildState +import org.apache.daffodil.runtime1.processors.unparsers.NotUnparsableUnparser +import org.apache.daffodil.runtime1.processors.unparsers.SuspensionCapableUState import org.apache.daffodil.runtime1.processors.unparsers.UState import org.apache.daffodil.runtime1.processors.unparsers.UnparseError +import org.apache.daffodil.runtime1.processors.unparsers.UnparseSharedContext +import org.apache.daffodil.runtime1.processors.unparsers.WriteCoroutine +import org.apache.daffodil.runtime1.processors.unparsers.WriteDone +import org.apache.daffodil.runtime1.processors.unparsers.WriteSignal +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase /** * Implementation mixin - provides simple helper methods @@ -458,6 +470,311 @@ class DataProcessor( } def unparse(actualInputter: api.infoset.InfosetInputter, outStream: java.io.OutputStream) = { + // hasAnyPrefetchBeneficialOVC is false only when every OVC is + // content-length-dependent, which prefetch could never resolve early + // regardless of how far build races ahead; fall back to single-pass + // automatically in that case, regardless of the tunable. + if (tunables.useBuildWritePrefetch && ssrd.hasAnyPrefetchBeneficialOVC) { + unparseViaBuildThenWrite(actualInputter, outStream) + } else { + unparseSinglePass(actualInputter, outStream) + } + } + + /** + * Shared by unparseViaBuildThenWrite and unparseSinglePass's top-level + * catch blocks: maps an exception caught during unparsing to a failed + * `state` plus its `unparseResult`, or rethrows if it's not one of the + * known unparse-error categories. + */ + private def unparseErrorResult(state: UState, t: Throwable): UnparseResult = t match { + case ue: UnparseError => { + state.addUnparseError(ue) + state.unparseResult + } + case procErr: ProcessingError => { + state.setFailed(procErr.toUnparseError) + state.unparseResult + } + case sde: SchemaDefinitionError => { + // A SDE was detected at runtime (perhaps due to a runtime-valued property like byteOrder or encoding) + // These are fatal, and there's no notion of backtracking them, so they propagate to top level + // here. + state.setFailed(sde) + state.unparseResult + } + case sdefw: SchemaDefinitionErrorFromWarning => { + state.setFailed(sdefw) + state.unparseResult + } + case e: ErrorAlreadyHandled => { + state.setFailed(e.th) + state.unparseResult + } + case e: TunableLimitExceededError => { + state.setFailed(e) + state.unparseResult + } + case se: org.xml.sax.SAXException => { + state.setFailed(new UnparseError(None, None, se)) + state.unparseResult + } + case e: scala.xml.parsing.FatalError => { + state.setFailed(new UnparseError(None, None, e)) + state.unparseResult + } + case ie: InfosetException => { + state.setFailed(new UnparseError(None, None, ie)) + state.unparseResult + } + case th: Throwable => throw th + } + + /** + * Build/write-prefetch unparse path (gated on `useBuildWritePrefetch`). + * `BuildState` builds the tree via the ordinary Unparser recursion. + * Only spawns write's own thread, recursing via `writeContent` and + * blocking on `awaitChild` via `Coroutine[T]`, if build's lead ever + * crosses the prefetch limit; otherwise write runs directly on this + * same thread afterward. + */ + private def unparseViaBuildThenWrite( + actualInputter: api.infoset.InfosetInputter, + outStream: java.io.OutputStream + ): UnparseResult = { + // A NotUnparsableUnparser (dfdl:parseUnparsePolicy="parseOnly") can't be + // cast to ElementUnparserBase or driven through the build/write split; + // unparseSinglePass already runs it via unparse1 and gets the correct + // diagnostic, so reuse that instead of a ClassCastException here. + ssrd.unparser match { + case _: NotUnparsableUnparser => return unparseSinglePass(actualInputter, outStream) + case _ => // fall through to the actual build/write-prefetch path below + } + val inputter = new InfosetInputter(actualInputter) + val rootUnparser = ssrd.unparser + var buildState: BuildState = null + // Cleaned up on write's OWN thread (inside runWriteThread below), not + // this method's outer finally block: it's only ever touched from + // write's thread (io.ThreadCheckMixin's affinity invariant), so + // cleaning it up elsewhere would violate that. + var writeState: UState with SuspensionCapableUState = null + // Null until buildState/writeState exists; the catch block below + // constructs a fallback UState to report through only if setup threw + // before either one was assigned here. + var activeState: UState = null + // Hoisted so the outer catch below can reach it (to abort write's + // thread if it's still parked) even when the exception came from + // partway through setup, before the try block's own local val would + // otherwise have gone out of scope. + var sharedCtx: UnparseSharedContext = null + val buildCoroutine = new BuildCoroutine() + + // Constructs write's UState and wires it into sharedCtx. Must run on + // write's own coroutine thread: writeState's DataOutputStream exists + // 1-to-1 with threads (io.ThreadCheckMixin), so constructing it + // eagerly on build's thread would bind it to the wrong one. + def constructWriteSide(): Unit = { + writeState = UState.createInitialUState(outStream, this, inputter, areDebugging) + writeState.setSharedContext(sharedCtx) + if (areDebugging) writeState.notifyDebugging(true) + init(writeState, rootUnparser) + // Forces evaluation of non-constant defineVariable defaults, same + // as single-pass's doUnparse; write is the side that actually + // reads/writes variables here, so it needs this on its own copy. + writeState.initializeVariables() + writeState.getDataOutputStream.setPriorBitOrder(ssrd.elementRuntimeData.defaultBitOrder) + } + + // evalSuspensions(isFinal = true) runs BEFORE the stack-depth + // invariants below: one tripping first could mask the real + // SuspensionDeadlockException diagnostic this ordering exists to + // surface. + def finishWriteSide(): Unit = { + writeState.setProcessor(rootUnparser) + + // Routed via sharedCtx.suspensionTracker, the SAME tracker + // BuildState registered into, so both build-side and write-side + // suspensions get resolved here. + writeState.evalSuspensions(isFinal = true) + + Assert.invariant(writeState.arrayIterationIndexStack.length == 1) + Assert.invariant(writeState.occursIndexStack.length == 1) + Assert.invariant(writeState.groupIndexStack.length == 1) + Assert.invariant(writeState.childIndexStack.length == 1) + Assert.invariant(writeState.currentInfosetNodeMaybe.isEmpty) + Assert.invariant(writeState.escapeSchemeEVCache.isEmpty) + Assert.invariant(writeState.maybeTopTRD().isEmpty) + Assert.invariant(!writeState.withinHiddenNest) + + Assert.invariant(!writeState.getDataOutputStream.isFinished) + try { + writeState.getDataOutputStream.setFinished(writeState) + } catch { + case boc: BitOrderChangeException => + writeState.SDE(boc) + case fio: FileIOException => + writeState.SDE(fio) + } + } + + // The write side's actual work for one unparse call: constructs its + // own UState, walks the tree build has (or will have) produced via + // writeContent, then finalizes. Returns the failure (if any) rather + // than throwing, so both runWriteThread below and the + // no-coroutine-needed inline call further down can report it the + // same way. + def doWriteSide(firstSignal: BuildSignal): Option[Throwable] = { + try { + // firstSignal may already be BuildFinished (build never crossed + // the prefetch-lead threshold) or BuildAborted (build failed + // early); observeBuildSignal handles either before any real work. + sharedCtx.observeBuildSignal(firstSignal) + constructWriteSide() + val rootElemUnp = rootUnparser.asInstanceOf[ElementUnparserBase] + try { + val rootNode = sharedCtx.awaitChild(inputter.documentElement, 0) + rootElemUnp.writeContent(rootNode, writeState) + } catch { + // A genuine deadlock (if any) surfaces via finishWriteSide's + // own evalSuspensions(isFinal = true) call below, which runs + // regardless of how this try block exits. + case _: AwaitChildStalledException => + } + finishWriteSide() + None + } catch { + // Includes BuildAbortedException, when firstSignal is + // BuildAborted: build's own thread has already unwound via its + // own catch below by the time that fires, so this particular + // result is never actually read by anyone (see runWriteThread). + case t: Throwable => Some(t) + } finally { + if (writeState != null) writeState.getDataOutputStream.cleanUp() + } + } + + // Write's entire coroutine-thread body. Reports failure back through + // the coroutine handoff rather than throwing here, since resumeFinal + // must still run. + def runWriteThread(wc: WriteCoroutine, firstSignal: BuildSignal): Unit = { + // Computed before resumeFinal is called below (never in an outer + // finally): resumeFinal requires returning from run() immediately, + // so build's thread mustn't wake until write's cleanup has + // actually finished, or both threads would run at once. doWriteSide's + // own finally block already covers that cleanup before returning. + val result = WriteDone(doWriteSide(firstSignal)) + wc.resumeFinal(buildCoroutine, result) + } + + // Runs build's own structural recursion to completion, asserting its + // stacks end up balanced and the inputter has nothing left unconsumed. + def runBuildPhase(): Unit = { + buildState = new BuildState(inputter, sharedCtx, Nil, areDebugging) + activeState = buildState + if (areDebugging) { + Assert.invariant(optDebugger.isDefined) + addEventHandler(debugger) + buildState.notifyDebugging(true) + } + init(buildState, rootUnparser) + + // The root element always has a builder: it is exactly the case that + // gets ElementBuilder wrapped around it, regardless of schema content. + ssrd.builder.get.build(buildState) + buildState.popTRD(rootUnparser.context.asInstanceOf[TermRuntimeData]) + buildState.setProcessor(rootUnparser) + + Assert.invariant(buildState.arrayIterationIndexStack.length == 1) + Assert.invariant(buildState.occursIndexStack.length == 1) + Assert.invariant(buildState.groupIndexStack.length == 1) + Assert.invariant(buildState.childIndexStack.length == 1) + Assert.invariant(buildState.currentInfosetNodeMaybe.isEmpty) + Assert.invariant(buildState.maybeTopTRD().isEmpty) + + val remainingEvent = buildState.advanceMaybe + if (remainingEvent.isDefined) { + UnparseError( + Nope, + One(buildState.currentLocation), + "Expected no remaining events, but received %s.", + remainingEvent.get + ) + } + } + + try { + inputter.initialize(ssrd.elementRuntimeData, tunables) + + sharedCtx = new UnparseSharedContext( + // Shared by build and write, which together tick this tracker at + // roughly twice the per-node rate a single traversal would; + // doubling both thresholds restores the intended sweep density. + new SuspensionTracker( + tunables.unparseSuspensionWaitYoung * 2, + tunables.unparseSuspensionWaitOld * 2 + ), + this, + tunables, + prefetchLimit = tunables.unparsePrefetchWindowNodes + ) + + val writeCoroutine = new WriteCoroutine(runWriteThread) + sharedCtx.setCoroutines(buildCoroutine, writeCoroutine) + + runBuildPhase() + + // The tree is now entirely built, but some suspensions may still be + // pending. writeCoroutine.isStarted is false whenever build's lead + // never crossed the prefetch threshold, meaning write was never + // needed concurrently at all; run it directly on this thread rather + // than paying to spawn and hand off to a thread purely to give it + // one, first, and only signal. Either way the result is recorded + // into sharedCtx so a later abortWrite/resumeWrite sees write as + // finished, matching what resumeWrite itself already does. + val finalSignal: WriteSignal = + if (writeCoroutine.isStarted) { + sharedCtx.resumeWrite(BuildFinished) + } else { + val result = WriteDone(doWriteSide(BuildFinished)) + sharedCtx.recordWriteDone(result) + result + } + // Only reassign once writeState is known-good: a null here means + // constructWriteSide() itself failed (reported via finalSignal + // below), so keep reporting through buildState instead. + if (writeState != null) activeState = writeState + finalSignal match { + case WriteDone(Some(t)) => throw t + case WriteDone(None) => // writeState.unparseResult below + case other => + Assert.invariantFailed( + s"write coroutine yielded $other instead of signaling completion" + ) + } + + writeState.unparseResult + } catch { + case t: Throwable => + // If write's thread is still parked (this exception came from + // build's own recursion, before the normal resumeWrite(BuildFinished) + // handoff ran), wake it so it can clean up instead of leaking its + // thread and DOS temp files forever. + if (sharedCtx != null) sharedCtx.abortWrite() + // Safe before initialize(): it only copies variableMap, never + // touching inputter.documentElement. + val reportState = + if (activeState != null) activeState + else UState.createInitialUState(outStream, this, inputter, areDebugging) + unparseErrorResult(reportState, t) + } finally { + if (buildState != null) buildState.getDataOutputStream.cleanUp() + } + } + + private def unparseSinglePass( + actualInputter: api.infoset.InfosetInputter, + outStream: java.io.OutputStream + ) = { val inputter = new InfosetInputter(actualInputter) val unparserState = UState.createInitialUState(outStream, this, inputter, areDebugging) @@ -477,47 +794,7 @@ class DataProcessor( unparserState.evalSuspensions(isFinal = true) unparserState.unparseResult } catch { - case ue: UnparseError => { - unparserState.addUnparseError(ue) - unparserState.unparseResult - } - case procErr: ProcessingError => { - val x = procErr - unparserState.setFailed(x.toUnparseError) - unparserState.unparseResult - } - case sde: SchemaDefinitionError => { - // A SDE was detected at runtime (perhaps due to a runtime-valued property like byteOrder or encoding) - // These are fatal, and there's no notion of backtracking them, so they propagate to top level - // here. - unparserState.setFailed(sde) - unparserState.unparseResult - } - case sdefw: SchemaDefinitionErrorFromWarning => { - unparserState.setFailed(sdefw) - unparserState.unparseResult - } - case e: ErrorAlreadyHandled => { - unparserState.setFailed(e.th) - unparserState.unparseResult - } - case e: TunableLimitExceededError => { - unparserState.setFailed(e) - unparserState.unparseResult - } - case se: org.xml.sax.SAXException => { - unparserState.setFailed(new UnparseError(None, None, se)) - unparserState.unparseResult - } - case e: scala.xml.parsing.FatalError => { - unparserState.setFailed(new UnparseError(None, None, e)) - unparserState.unparseResult - } - case ie: InfosetException => { - unparserState.setFailed(new UnparseError(None, None, ie)) - unparserState.unparseResult - } - case th: Throwable => throw th + case t: Throwable => unparseErrorResult(unparserState, t) } finally { unparserState.getDataOutputStream.cleanUp() } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala index e830aead7b..8eb0a6f6c2 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala @@ -18,22 +18,30 @@ package org.apache.daffodil.runtime1.processors import org.apache.daffodil.lib.exceptions.ThrowsSDE +import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.runtime1.layers.LayerRuntimeCompiler import org.apache.daffodil.runtime1.layers.LayerRuntimeData import org.apache.daffodil.runtime1.layers.LayerVarsRuntime import org.apache.daffodil.runtime1.processors.parsers.Parser +import org.apache.daffodil.runtime1.processors.unparsers.Builder import org.apache.daffodil.runtime1.processors.unparsers.Unparser final class SchemaSetRuntimeData( val parser: Parser, val unparser: Unparser, + val builder: Maybe[Builder], val elementRuntimeData: ElementRuntimeData, /* * The original variables determined by the schema compiler. */ variables: VariableMap, allLayers: Seq[LayerRuntimeData], - @transient layerRuntimeCompilerArg: LayerRuntimeCompiler + @transient layerRuntimeCompilerArg: LayerRuntimeCompiler, + /** True if this schema has an outputValueCalc element whose value could + * resolve without writing (see hasAnyPrefetchBeneficialOVC); baked in at + * compile time like the rest of this class. Gates useBuildWritePrefetch - + * zero benefit means automatic fallback to single-pass. */ + val hasAnyPrefetchBeneficialOVC: Boolean ) extends Serializable with ThrowsSDE { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/Suspension.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/Suspension.scala index d41b3accfa..98016cd91d 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/Suspension.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/Suspension.scala @@ -25,8 +25,8 @@ import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.lib.util.MaybeInt import org.apache.daffodil.lib.util.MaybeULong +import org.apache.daffodil.runtime1.processors.unparsers.SuspensionCapableUState import org.apache.daffodil.runtime1.processors.unparsers.UState -import org.apache.daffodil.runtime1.processors.unparsers.UStateMain import org.apache.daffodil.runtime1.processors.unparsers.UnparseError /** @@ -51,6 +51,17 @@ trait Suspension extends Serializable { */ val isReadOnly = false + /** + * True if this suspension might resolve without any bytes written yet + * (e.g. value/variable read); false if it needs a real DOS bit position + * (e.g. valueLength/contentLength, padding/alignment); distinct from + * isReadOnly. A static, direction-blind heuristic (can't tell a + * backward reference to an already-resolved value from a forward + * one), used by build's retries to skip an attempt that's usually, + * but not always, doomed to block. + */ + def canResolveWithoutWriting: Boolean = false + def UE(ustate: UState, s: String, args: Any*) = { UnparseError(One(rd.schemaFileLocation), One(ustate.currentLocation), s, args*) } @@ -111,7 +122,7 @@ trait Suspension extends Serializable { // // As written, we have a bunch of suspensions that occur, but have // specifically known length of zero bits. So nothing being written out. - // In that case, why do we need to split at all? + // TODO: In that case, why do we need to split at all? // val original = ustate.getDataOutputStream if (mkl.isEmpty || (mkl.isDefined && mkl.get > 0)) { @@ -175,18 +186,20 @@ trait Suspension extends Serializable { // // clone the ustate for use when evaluating the expression // - // TODO: Performance - copying this whole state, just for OVC is painful. - // Some sort of copy-on-write scheme would be better. + // This is a targeted partial clone (shallow VariableMap copy, stack + // tops only), not a full deep copy, but still copies the full + // escapeSchemeEVCache/delimiterStack contents unconditionally. + // TODO: Performance: a copy-on-write scheme could avoid that copy. // val didSplit = (ustate.getDataOutputStream ne original) - val cloneUState = ustate.asInstanceOf[UStateMain].cloneForSuspension(original) + val cloneUState = ustate.asInstanceOf[SuspensionCapableUState].cloneForSuspension(original) if (isReadOnly && didSplit) { Assert.invariantFailed("Shouldn't have split. read-only case") } savedUstate_ = cloneUState - ustate.asInstanceOf[UStateMain].addSuspension(this) + ustate.asInstanceOf[SuspensionCapableUState].addSuspension(this) } final def explain(): Unit = { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SuspensionTracker.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SuspensionTracker.scala index b38ba53909..cf7d99f4a8 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SuspensionTracker.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SuspensionTracker.scala @@ -30,6 +30,13 @@ class SuspensionTracker(suspensionWaitYoung: Int, suspensionWaitOld: Int) { def suspensions: Seq[Suspension] = suspensionsYoung.toSeq ++ suspensionsOld.toSeq + /** + * Total not-yet-done suspensions, without allocating (unlike + * `suspensions`, not safe to call once per node). A throttle signal for + * pacing a no-op-sink traversal against the pending backlog. + */ + def pendingCount: Int = suspensionsYoung.length + suspensionsOld.length + private var count: Int = 0 private var suspensionStatTracked: Int = 0 @@ -47,12 +54,24 @@ class SuspensionTracker(suspensionWaitYoung: Int, suspensionWaitOld: Int) { * suspensions, we attempt to evaluate them first, with the hope that their * resolution might make the young suspensions more likely to evaluate. */ - def evalSuspensions(): Unit = { + def evalSuspensions(): Unit = evalSuspensionsThrottled(buildResolvableOnly = false) + + /** + * A no-op-sink sweep variant: same cadence as evalSuspensions, but with + * buildResolvableOnly=true; a suspension whose canResolveWithoutWriting + * is false is skipped-and-requeued instead of genuinely attempted, since + * this heuristic marks it as usually unresolvable here; it stays pending + * for a later unfiltered sweep either way. + */ + def evalBuildResolvableSuspensions(): Unit = + evalSuspensionsThrottled(buildResolvableOnly = true) + + private def evalSuspensionsThrottled(buildResolvableOnly: Boolean): Unit = { if (count % suspensionWaitOld == 0) { - evalSuspensionQueue(suspensionsOld) + evalSuspensionQueue(suspensionsOld, buildResolvableOnly = buildResolvableOnly) } if (count % suspensionWaitYoung == 0) { - evalSuspensionQueue(suspensionsYoung) + evalSuspensionQueue(suspensionsYoung, buildResolvableOnly = buildResolvableOnly) while (suspensionsYoung.nonEmpty) { suspensionsOld.enqueue(suspensionsYoung.dequeue()) } @@ -65,6 +84,20 @@ class SuspensionTracker(suspensionWaitYoung: Int, suspensionWaitOld: Int) { } } + /** + * Attempts every currently-tracked suspension once, bypassing throttling, + * without treating remaining blocks as an error. Intended for a caller + * whose own traversal has fully finished and needs an already-resolvable + * suspension to actually resolve before it can continue. + */ + def evalSuspensionsUnthrottled(): Unit = { + evalSuspensionQueue(suspensionsOld) + evalSuspensionQueue(suspensionsYoung) + while (suspensionsYoung.nonEmpty) { + suspensionsOld.enqueue(suspensionsYoung.dequeue()) + } + } + /** * Evaluates all suspensions until either they are all evaluated or a * deadlock is detected. This moves all young suspensions to the old queue, @@ -94,23 +127,34 @@ class SuspensionTracker(suspensionWaitYoung: Int, suspensionWaitOld: Int) { } /** - * Attempt to evaluate suspensions on the provie queue. Keep repeating the - * evaluates as long as some progress is being made. Suspensions that - * evaluate sucessfully are removed from the queue. Once suspensions make no - * further progress and are all blocked, we return. Blocked suspensions put - * back on the same queue. + * Attempt to evaluate suspensions on the provided queue. Keep repeating + * the evaluates as long as some progress is being made. Suspensions that + * evaluate successfully are removed from the queue. Once suspensions make + * no further progress and are all blocked, we return. Blocked suspensions + * are put back on the same queue. buildResolvableOnly diverts a + * canResolveWithoutWriting=false suspension into skip-and-requeue instead + * of genuinely attempting it, since this heuristic marks it as usually + * unresolvable here. */ - private def evalSuspensionQueue(queue: Queue[Suspension]): Unit = { + private def evalSuspensionQueue( + queue: Queue[Suspension], + buildResolvableOnly: Boolean = false + ): Unit = { var countOfNotMakingProgress = 0 while (!queue.isEmpty && countOfNotMakingProgress < queue.length) { val s = queue.dequeue() - suspensionStatRuns += 1 - s.runSuspension() - if (!s.isDone) queue.enqueue(s) - if (s.isDone || s.isMakingProgress) { - countOfNotMakingProgress = 0 - } else { + if (buildResolvableOnly && !s.canResolveWithoutWriting) { + queue.enqueue(s) countOfNotMakingProgress += 1 + } else { + suspensionStatRuns += 1 + s.runSuspension() + if (!s.isDone) queue.enqueue(s) + if (s.isDone || s.isMakingProgress) { + countOfNotMakingProgress = 0 + } else { + countOfNotMakingProgress += 1 + } } } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildState.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildState.scala new file mode 100644 index 0000000000..be6085490b --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildState.scala @@ -0,0 +1,207 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream + +import org.apache.daffodil.api +import org.apache.daffodil.io.DirectOrBufferedDataOutputStream +import org.apache.daffodil.io.StringDataInputStreamForUnparse +import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.LocalStack +import org.apache.daffodil.lib.util.MStackOfMaybe +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.runtime1.dpath.UnparserBlocking +import org.apache.daffodil.runtime1.infoset.DIDocument +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.infoset.InfosetAccessor +import org.apache.daffodil.runtime1.infoset.InfosetInputter +import org.apache.daffodil.runtime1.processors.DelimiterStackUnparseNode +import org.apache.daffodil.runtime1.processors.EscapeSchemeUnparserHelper +import org.apache.daffodil.runtime1.processors.Suspension +import org.apache.daffodil.runtime1.processors.SuspensionTracker +import org.apache.daffodil.runtime1.processors.TermRuntimeData +import org.apache.daffodil.runtime1.processors.VariableBox +import org.apache.daffodil.runtime1.processors.VariableMap +import org.apache.daffodil.runtime1.processors.dfa.DFADelimiter + +/** + * A `UState` subclass that consumes an actual `InfosetInputter` and provides + * live Cursor/TRD/index-stack behavior; this is the "build" side of the + * build/write unparse split. The write-only surface (delimiter stack, + * escape scheme cache, and the scratch buffers used for measuring/escaping + * text) is stubbed to error, since nothing build does should ever touch it; + * build never writes content. + * + * `getDataOutputStream` is NOT stubbed: generic `UState` utility methods + * (toString, currentLocation, bitPos0b) call into it unconditionally, so + * `BuildState` constructs an actual DOS wrapping a no-op sink purely to + * satisfy that. + * + * Used only when the `useBuildWritePrefetch` tunable is enabled (default + * false); otherwise unused, and unparsing constructs `UStateMain` + * exclusively as before. + */ +final class BuildState( + private val inputter: InfosetInputter, + sharedCtx: UnparseSharedContext, + diagnosticsArg: Seq[api.Diagnostic], + areDebugging: Boolean +) extends UState( + // Build never reads or writes a variable, so an empty map is enough; + // it's just here to satisfy UState's constructor. + new VariableBox(VariableMap()), + diagnosticsArg, + Maybe(sharedCtx.dataProc), + sharedCtx.tunable, + areDebugging + ) + with SuspensionCapableUState + with TraversalIndexStacks { + + dState.setMode(UnparserBlocking) + setSharedContext(sharedCtx) + + // Purely so generic UState utility methods (toString, currentLocation, + // bitPos0b) have something non-null to call into; never actually + // written to for real output. + setDataOutputStream( + DirectOrBufferedDataOutputStream( + new java.io.OutputStream { override def write(b: Int): Unit = () }, + null, + false, + sharedCtx.tunable.outputStreamChunkSizeInBytes, + sharedCtx.tunable.maxByteArrayOutputStreamBufferSizeInBytes, + sharedCtx.tunable.tempFilePath + ) + ) + + // Build runs ahead of write, so freeing a node here would null out a + // child reference write hasn't read yet; write still frees as normal. + override def releaseUnneededInfoset: Boolean = false + + private def writeOnly = + Assert.usageError("BuildState never writes content, so this write-only state doesn't exist") + + override def escapeSchemeEVCache: MStackOfMaybe[EscapeSchemeUnparserHelper] = writeOnly + override def withUnparserDataInputStream: LocalStack[StringDataInputStreamForUnparse] = + writeOnly + override def withByteArrayOutputStream + : LocalStack[(ByteArrayOutputStream, DirectOrBufferedDataOutputStream)] = writeOnly + override def allTerminatingMarkup: List[DFADelimiter] = writeOnly + override def localDelimiters: DelimiterStackUnparseNode = writeOnly + override def pushDelimiters(node: DelimiterStackUnparseNode): Unit = writeOnly + override def popDelimiters(): Unit = writeOnly + + override def advance: Boolean = inputter.advance + override def advanceAccessor: InfosetAccessor = inputter.advanceAccessor + override def inspect: Boolean = inputter.inspect + override def inspectAccessor: InfosetAccessor = inputter.inspectAccessor + override def fini(): Unit = Assert.usageError("Not to be used on UState") + + override def inspectOrError: InfosetAccessor = { + if (inspect) { + inspectAccessor + } else { + Assert.invariantFailed( + "An InfosetEvent was required for building, but no InfosetEvent was available." + ) + } + } + + override def advanceOrError: InfosetAccessor = { + if (advance) { + advanceAccessor + } else { + Assert.invariantFailed( + "An InfosetEvent was required for building, but no InfosetEvent was available." + ) + } + } + + override def isInspectArrayEnd: Boolean = { + if (!inspect) { + false + } else { + inspectAccessor match { + case e if e.isEnd && e.isArray => true + case _ => false + } + } + } + + def currentInfosetNode: DINode = { + if (currentInfosetNodeMaybe.isEmpty) { + null + } else { + currentInfosetNodeMaybe.get + } + } + + def currentInfosetNodeMaybe: Maybe[DINode] = { + if (currentInfosetNodeStack.isEmpty) { + Nope + } else { + currentInfosetNodeStack.top + } + } + + override val currentInfosetNodeStack = new MStackOfMaybe[DINode] + + // Shared, not owned; one SuspensionTracker queue, both build and write + // see the same one via sharedCtx. + def suspensionTracker: SuspensionTracker = sharedCtx.suspensionTracker + + // Build never evaluates an expression or writes a suspendable child, so + // nothing build does can ever suspend, and this is never called: only + // Suspension.suspend calls it, and that always calls cloneForSuspension + // first, which already throws. + def addSuspension(se: Suspension): Unit = writeOnly + + /** + * Uses evalBuildResolvableSuspensions: canResolveWithoutWriting is a + * static, direction-blind heuristic, so a suspension it marks false + * (usually a forward reference, occasionally an already-resolved + * backward one) is skipped here rather than genuinely retried, and + * left pending for a later, unfiltered sweep instead. + */ + def evalSuspensions(isFinal: Boolean): Unit = { + sharedCtx.suspensionTracker.evalBuildResolvableSuspensions() + if (isFinal) sharedCtx.suspensionTracker.requireFinal() + } + def suspensions = sharedCtx.suspensionTracker.suspensions + + /** + * Build never evaluates an expression or writes a suspendable child, so + * nothing build does can ever suspend, and this is never called. + */ + override def cloneForSuspension(suspendedDOS: DirectOrBufferedDataOutputStream): UState = + writeOnly + + final override def pushTRD(trd: TermRuntimeData): Unit = inputter.pushTRD(trd) + final override def maybeTopTRD(): Maybe[TermRuntimeData] = inputter.maybeTopTRD() + final override def popTRD(trd: TermRuntimeData): TermRuntimeData = { + val poppedTRD = inputter.popTRD() + if (poppedTRD ne trd) + Assert.invariantFailed("TRDs do not match. Expected: " + trd + " got " + poppedTRD) + poppedTRD + } + + final override def documentElement: DIDocument = inputter.documentElement +} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildWriteCoroutines.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildWriteCoroutines.scala new file mode 100644 index 0000000000..861fb32301 --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildWriteCoroutines.scala @@ -0,0 +1,100 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.lib.util.Coroutine +import org.apache.daffodil.lib.util.MainCoroutine + +/* + * The build/write handoff for build/write-prefetch: build and write's + * own content dispatch hand off via Coroutine[T], guaranteeing only one + * of the pair ever runs at a time, via a blocking-queue rendezvous + * between two threads. + */ + +/** Signals build sends to write when resuming it. */ +sealed trait BuildSignal + +/** Build made progress since write last ran; try again. */ +case object MoreTreeAvailable extends BuildSignal + +/** + * Build's top-level recursion has fully returned; no further nodes will + * ever be added. Sent exactly once, and may be the very first signal + * write's coroutine ever receives, if build never crossed the + * prefetch-lead threshold. Write must check for this on every resume. + */ +case object BuildFinished extends BuildSignal + +/** + * Build's own thread failed before its normal completion handoff. Write + * is guaranteed to be paused at this exact moment, so this is how + * build's exception wakes it back up; write skips its usual + * finalization and re-throws build's exception instead. + */ +case object BuildAborted extends BuildSignal + +/** Signals write sends to build when yielding control back. */ +sealed trait WriteSignal + +/** + * Write's own recursive dispatch has blocked, needing more tree before + * it can continue; build should keep building, and will be resumed + * again once it does. + */ +case object WriteNeedsMore extends WriteSignal + +/** + * Write has finished all remaining work, or failed trying to. Sent + * exactly once, as write's last act. The captured error, if any, is + * re-thrown on build's thread instead. + */ +final case class WriteDone(error: Option[Throwable]) extends WriteSignal + +/** + * Build's side of the handoff. Runs on the ORIGINAL calling thread + * (`isMain = true`, so no thread is spawned for it); build already drives + * everything from the caller's thread; only write gets a genuinely new + * thread (`WriteCoroutine` below). + */ +final class BuildCoroutine extends MainCoroutine[WriteSignal] + +/** + * Write's side of the handoff. An actual, separate thread, spawned lazily + * (via `Coroutine`'s own `init()`) the first time build resumes it. + * `runBody` (the caller-supplied build/write driver) must construct + * write's own `UState` as its FIRST action on this thread, not before it + * starts and not on the build thread, since `DataOutputStream` is pinned + * to its creating thread; constructing it eagerly on build's thread would + * bind it to the wrong one. + * + * `runBody` receives the signal that started this thread as its second + * argument, rather than `run()` silently discarding it, because that + * first signal might already be `BuildFinished`; `runBody` must treat it + * like any later resume result, not assume the first is always + * `MoreTreeAvailable`. It must end by calling + * `resumeFinal(buildCoroutine, WriteDone(...))` (call it, then return + * from `run()` immediately; its own contract). + */ +final class WriteCoroutine(runBody: (WriteCoroutine, BuildSignal) => Unit) + extends Coroutine[BuildSignal] { + override protected def run(): Unit = { + val firstSignal = waitForResume() + runBody(this, firstSignal) + } +} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Builder.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Builder.scala new file mode 100644 index 0000000000..dc103fad06 --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Builder.scala @@ -0,0 +1,263 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.One +import org.apache.daffodil.runtime1.infoset.ChoiceBranchEndEvent +import org.apache.daffodil.runtime1.infoset.ChoiceBranchEvent +import org.apache.daffodil.runtime1.infoset.ChoiceBranchStartEvent +import org.apache.daffodil.runtime1.processors.ElementRuntimeData +import org.apache.daffodil.runtime1.processors.ModelGroupRuntimeData +import org.apache.daffodil.runtime1.processors.TermRuntimeData +import org.apache.daffodil.unparsers.runtime1.RepeatingChildUnparser +import org.apache.daffodil.unparsers.runtime1.SequenceChildUnparser + +/** + * A node in a much smaller, dedicated tree of Builders that parallels the + * full Unparser tree: only Grams that actually create or select infoset + * content (elements, sequences, choices, hidden groups) contribute one, so + * building the infoset never has to dispatch through the many write-only + * wrapper unparsers (delimiters, escape schemes, layers, padding, + * specified-length) that sit between them in the Unparser tree. Driven + * exclusively from `BuildState`, never from a write-side `UState`. + */ +trait Builder extends Serializable { + def build(state: UState): Unit +} + +/** + * A no-op stand-in for a choice branch or sequence child whose content is + * empty (e.g. an empty sequence), so its build-time presence can still be + * recorded without anything actually needing to happen. + */ +object EmptyBuilder extends Builder { + override def build(state: UState): Unit = () +} + +/** + * Builds each of several sibling Grams' content in order. Used only where a + * `~` composition has more than one child that actually builds infoset + * content; the common case of at most one such child never needs this. + */ +final class SeqCompBuilder(children: Array[Builder]) extends Builder { + override def build(state: UState): Unit = { + var i = 0 + while (i < children.length) { + children(i).build(state) + i += 1 + } + } +} + +/** + * Builds one element's infoset node and, for complex types, recurses into + * contentBuilder to build descendant nodes. unparseBegin/unparseEnd are the + * same element-kind-specific (plain/nillable/OVC/etc.) node-creation logic + * unparse() itself uses, including the bounded-lookahead lead-counter + * hookup and the deferred simple-value finalization; only the "what does + * this element contain" recursion is redirected to the builder tree instead + * of back into the unparser tree. + */ +final class ElementBuilder( + erd: ElementRuntimeData, + unparseBegin: UState => Unit, + unparseEnd: UState => Unit, + contentBuilder: Maybe[Builder] +) extends Builder { + + override def build(state: UState): Unit = { + unparseBegin(state) + + if (erd.isComplexType) { + state.pushTRD(erd.optComplexTypeModelGroupRuntimeData.get) + if (contentBuilder.isDefined) { contentBuilder.get.build(state) } + state.popTRD(erd.optComplexTypeModelGroupRuntimeData.get) + } + + unparseEnd(state) + } +} + +/** + * Pairs a sequence child's existing occurs-count/array bookkeeping (reused + * as-is from the Unparser tree, since it is cheap, pure state bookkeeping + * unrelated to the tree-walking overhead this Builder tree exists to avoid) + * with that same child's own Builder, which SequenceBuilder recurses into + * instead of the child's full Unparser. + */ +final case class SequenceChildBuildInfo( + childUnparser: SequenceChildUnparser, + childBuilder: Builder +) + +/** + * Builds an entire sequence's children, scalar and array/optional alike. + * Mirrors OrderedSequenceUnparserBase's own build loop exactly, except each + * child's recursive build call targets its Builder rather than its full, + * write-only-wrapper-laden Unparser. + */ +final class SequenceBuilder(children: IndexedSeq[SequenceChildBuildInfo]) extends Builder { + + override def build(state: UState): Unit = { + state.groupIndexStack.push(1L) + + var index = 0 + val limit = children.length + while (index < limit) { + val info = children(index) + val cu = info.childUnparser + val trd = cu.trd + state.pushTRD(trd) + cu match { + case rep: RepeatingChildUnparser => { + state.arrayIterationIndexStack.push(1L) + state.occursIndexStack.push(1L) + val erd = rep.erd + var numOccurrences = 0 + val maxReps = rep.maxRepeats(state) + + Assert.invariant(state.inspect, "No event for building.") + val ev = state.inspectAccessor + if (ev.erd eq erd) { + rep.startArrayOrOptional(state) + while (rep.shouldDoUnparser(rep, state)) { + info.childBuilder.build(state) + numOccurrences += 1 + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + state.moveOverOneGroupIndexOnly() + } + rep.checkFinalOccursCountBetweenMinAndMaxOccurs( + state, + rep, + numOccurrences, + maxReps, + state.arrayIterationPos - 1 + ) + rep.endArrayOrOptional(erd, state) + } else { + rep.checkFinalOccursCountBetweenMinAndMaxOccurs( + state, + rep, + numOccurrences, + maxReps, + 0 + ) + } + + state.arrayIterationIndexStack.pop() + state.occursIndexStack.pop() + } + case _ => { + info.childBuilder.build(state) + trd match { + case erd: ElementRuntimeData if !erd.isRepresented => // ok, skip group advance + case _ => state.moveOverOneGroupIndexOnly() + } + } + } + state.popTRD(trd) + index += 1 + } + + state.groupIndexStack.pop() + } +} + +/** + * Builds just the one structurally-present branch of a choice. Mirrors + * ChoiceCombinatorUnparser's own branch resolution exactly, except the + * resolved branch's recursive build call targets its Builder rather than + * its full Unparser. + */ +final class ChoiceBuilder( + mgrd: ModelGroupRuntimeData, + branchMap: Map[ChoiceBranchEvent, (TermRuntimeData, Builder)], + defaultBranch: Maybe[(TermRuntimeData, Builder)] +) extends Builder { + + private def buildBranch(state: UState, branch: (TermRuntimeData, Builder)): Unit = { + val (trd, builder) = branch + state.pushTRD(trd) + builder.build(state) + state.popTRD(trd) + } + + override def build(state: UState): Unit = { + if (state.withinHiddenNest) { + buildBranch(state, defaultBranch.get) + } else { + state.pushTRD(mgrd) + val event = state.inspectOrError + val key: ChoiceBranchEvent = event match { + case e if e.isStart && (e.isElement || e.isArray) => + ChoiceBranchStartEvent(e.erd.namedQName) + case e if e.isEnd && (e.isElement || e.isArray) => + ChoiceBranchEndEvent(e.erd.namedQName) + } + val fromTable = branchMap.get(key) + val resolved = if (fromTable.isDefined) { + fromTable + } else { + defaultBranch.toOption + } + if (resolved.isEmpty) { + UnparseError( + One(mgrd.schemaFileLocation), + One(state.currentLocation), + "Found next element %s, but expected one of %s.", + key.qname.toExtendedSyntax, + branchMap.keys.map { _.qname.toExtendedSyntax }.mkString(", ") + ) + } + state.popTRD(mgrd) + buildBranch(state, resolved.get) + } + } +} + +/** + * Builds the body of a hidden group. withinHiddenNest must stay maintained + * during build too: it is what tells a hidden element's unparseBegin/ + * unparseEnd to manufacture a node instead of consuming an event that will + * never exist. + */ +final class HiddenGroupBuilder(bodyBuilder: Builder) extends Builder { + override def build(state: UState): Unit = { + try { + state.incrementHiddenDef() + bodyBuilder.build(state) + } finally { + state.decrementHiddenDef() + } + } +} + +/** + * A nilled complex element has no children to build; nilled-ness is only + * known once the node exists, so this checks it at build time rather than + * resolving statically to either branch. + */ +final class NilOrContentBuilder(contentBuilder: Builder) extends Builder { + override def build(state: UState): Unit = { + val inode = state.currentInfosetNode.asComplex + if (!inode.isNilled) { contentBuilder.build(state) } + } +} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala index 27a4381877..fb3b205df9 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala @@ -161,6 +161,19 @@ abstract class UState( def moveOverOneGroupIndexOnly(): Unit def moveOverOneElementChildOnly(): Unit + // A dfdl:occursIndex() expression in an occurrence's own content reads + // arrayIterationIndexStack/occursIndexStack's top, which must track the + // occurrence currently being unparsed; both build and write push/pop + // this pair identically around each array/optional occurrence group. + final def pushOccurrenceIndices(): Unit = { + arrayIterationIndexStack.push(1L) + occursIndexStack.push(1L) + } + final def popOccurrenceIndices(): Unit = { + arrayIterationIndexStack.pop() + occursIndexStack.pop() + } + def inspectOrError: InfosetAccessor def advanceOrError: InfosetAccessor def isInspectArrayEnd: Boolean @@ -386,10 +399,10 @@ abstract class UState( // clone the UState so it can no longer change, and pass that clone into // setFinished. val finfo = this match { - case m: UStateMain => m.cloneForSuspension(dos) + case m: SuspensionCapableUState => m.cloneForSuspension(dos) case _ => Assert.invariantFailed( - "State must be a UStateMain when splitting for bit order change" + "State must be SuspensionCapableUState when splitting for bit order change" ) } @@ -404,9 +417,38 @@ abstract class UState( def documentElement: DIDocument - final val releaseUnneededInfoset: Boolean = !areDebugging && tunable.releaseUnneededInfoset + // Build must never free infoset nodes: build runs ahead of write, so + // freeing one would null out a child reference write hasn't read yet. + // Write still frees as normal once done with a node. + def releaseUnneededInfoset: Boolean def delimitedParseResult = Nope + + // UStateMain owns the real one; UStateForSuspension delegates to its + // mainUState so that a Suspension can always reach its tracker via + // savedUstate, even after it's been cloned off for suspension. + def suspensionTracker: SuspensionTracker + + // Optional reference to the shared build/write lead counter. Defaults + // unset (a no-op for every existing call site); only BuildState sets it + // (in its constructor), and only a write-side UState that opts in (via + // setSharedContext) reads it. + private var sharedContextMaybe: Maybe[UnparseSharedContext] = Nope + final def setSharedContext(ctx: UnparseSharedContext): Unit = sharedContextMaybe = One(ctx) + final def sharedContext: Maybe[UnparseSharedContext] = sharedContextMaybe +} + +/** + * Mixed in by any `UState` that tracks `Suspension`s - `UStateMain` and + * `BuildState`. Only write ever creates one; build never does, but + * still needs `evalSuspensions`/`suspensions` to resolve write-created + * suspensions early against the tree it has already built. + */ +trait SuspensionCapableUState { self: UState => + def suspensions: Seq[Suspension] + def addSuspension(se: Suspension): Unit + def evalSuspensions(isFinal: Boolean): Unit + def cloneForSuspension(suspendedDOS: DirectOrBufferedDataOutputStream): UState } /** @@ -418,7 +460,7 @@ abstract class UState( * memory. */ final class UStateForSuspension( - val mainUState: UStateMain, + val mainUState: UState with SuspensionCapableUState, val dataOutputStream: DirectOrBufferedDataOutputStream, vbox: VariableBox, override val currentInfosetNode: DINode, @@ -430,6 +472,10 @@ final class UStateForSuspension( areDebugging: Boolean ) extends UState(vbox, mainUState.diagnostics, mainUState.dataProc, tunable, areDebugging) { + // Always a live write-side UState, since BuildState never creates a + // suspension to clone off of. + override def releaseUnneededInfoset: Boolean = mainUState.releaseUnneededInfoset + _dataOutputStream = dataOutputStream dState.setMode(UnparserBlocking) dState.setCurrentNode(thisElement.asInstanceOf[DINode]) @@ -443,6 +489,11 @@ final class UStateForSuspension( override def getEncoder(cs: BitsCharset): BitsCharsetEncoder = mainUState.getEncoder(cs) override def suspensions = mainUState.suspensions + override val suspensionTracker = mainUState.suspensionTracker + // sharedContext/setSharedContext are final on UState (single mutable + // field, not overridable), so this clone's own copy is primed explicitly + // here rather than delegated, mirroring mainUState's at construction time. + if (mainUState.sharedContext.isDefined) setSharedContext(mainUState.sharedContext.get) // override def charBufferDataOutputStream = mainUState.charBufferDataOutputStream override def withUnparserDataInputStream = mainUState.withUnparserDataInputStream @@ -503,7 +554,44 @@ final class UStateForSuspension( } } -final class UStateMain private ( +/** + * Stack-backed array-iteration/occurs/group/child index tracking, + * shared by UStateMain and BuildState: each stack starts seeded with 1L, + * and moveOverOne*Only bumps its top by one as navigation advances. + * UStateForSuspension needs none of this; it stubs the stacks to die and + * tracks arrayIterationPos/occursPos as plain frozen Longs captured at + * suspension time instead (see its overrides above), so this lives in a + * mixin rather than directly on UState. + */ +trait TraversalIndexStacks { self: UState => + override val arrayIterationIndexStack = MStackOfLong() + arrayIterationIndexStack.push(1L) + override def moveOverOneArrayIterationIndexOnly(): Unit = + arrayIterationIndexStack.setTop(arrayIterationIndexStack.top + 1) + override def arrayIterationPos = arrayIterationIndexStack.top + + override val occursIndexStack = MStackOfLong() + occursIndexStack.push(1L) + override def moveOverOneOccursIndexOnly(): Unit = + occursIndexStack.setTop(occursIndexStack.top + 1) + override def occursPos = occursIndexStack.top + + override val groupIndexStack = MStackOfLong() + groupIndexStack.push(1L) + override def moveOverOneGroupIndexOnly(): Unit = + groupIndexStack.setTop(groupIndexStack.top + 1) + override def groupPos = groupIndexStack.top + + // TODO: it doesn't look anything is actually reading the value of childindex + // stack. Can we get rid of it? + override val childIndexStack = MStackOfLong() + childIndexStack.push(1L) + override def moveOverOneElementChildOnly(): Unit = + childIndexStack.setTop(childIndexStack.top + 1) + override def childPos = childIndexStack.top +} + +final class UStateMain private[unparsers] ( private val inputter: InfosetInputter, outStream: java.io.OutputStream, vbox: VariableBox, @@ -511,7 +599,11 @@ final class UStateMain private ( dataProcArg: DataProcessor, tunable: DaffodilTunables, areDebugging: Boolean -) extends UState(vbox, diagnosticsArg, One(dataProcArg), tunable, areDebugging) { +) extends UState(vbox, diagnosticsArg, One(dataProcArg), tunable, areDebugging) + with SuspensionCapableUState + with TraversalIndexStacks { + + final val releaseUnneededInfoset: Boolean = !areDebugging && tunable.releaseUnneededInfoset dState.setMode(UnparserBlocking) @@ -553,8 +645,10 @@ final class UStateMain private ( // MStack, since the escape scheme cache logic requires an MStack. We // reallyjust need the top for cloning for suspensions, but that // requires changes to how the escape schema cache is accessed, which - // isn't a trivial change. - val esClone = new MStackOfMaybe[EscapeSchemeUnparserHelper]() + // isn't a trivial change. Sized to the source's actual depth since + // nothing ever pushes onto a suspension's cloned escapeSchemeEVCache + // after this point. + val esClone = new MStackOfMaybe[EscapeSchemeUnparserHelper](escapeSchemeEVCache.length) esClone.copyFrom(escapeSchemeEVCache) Maybe(esClone) } else { @@ -563,8 +657,11 @@ final class UStateMain private ( val ds = if (!delimiterStack.isEmpty) { // If there are any delimiters, then we need to clone them all since - // they may be needed for escaping - val dsClone = new MStackOf[DelimiterStackUnparseNode]() + // they may be needed for escaping. Sized to the source's actual + // depth: pushDelimiters/popDelimiters both die on this clone (see + // below), so it's read-only for the rest of the suspension's life + // and can never grow past this depth. + val dsClone = new MStackOf[DelimiterStackUnparseNode](delimiterStack.length) dsClone.copyFrom(delimiterStack) Maybe(dsClone) } else { @@ -665,29 +762,6 @@ final class UStateMain private ( override val currentInfosetNodeStack = new MStackOfMaybe[DINode] - override val arrayIterationIndexStack = MStackOfLong() - arrayIterationIndexStack.push(1L) - override def moveOverOneArrayIterationIndexOnly() = - arrayIterationIndexStack.setTop(arrayIterationIndexStack.top + 1) - override def arrayIterationPos = arrayIterationIndexStack.top - - override val occursIndexStack = MStackOfLong() - occursIndexStack.push(1L) - override def moveOverOneOccursIndexOnly() = occursIndexStack.setTop(occursIndexStack.top + 1) - override def occursPos = occursIndexStack.top - - override val groupIndexStack = MStackOfLong() - groupIndexStack.push(1L) - override def moveOverOneGroupIndexOnly() = groupIndexStack.setTop(groupIndexStack.top + 1) - override def groupPos = groupIndexStack.top - - // TODO: it doesn't look anything is actually reading the value of childindex - // stack. Can we get rid of it? - override val childIndexStack = MStackOfLong() - childIndexStack.push(1L) - override def moveOverOneElementChildOnly() = childIndexStack.setTop(childIndexStack.top + 1) - override def childPos = childIndexStack.top - override lazy val escapeSchemeEVCache = new MStackOfMaybe[EscapeSchemeUnparserHelper] val delimiterStack = new MStackOf[DelimiterStackUnparseNode]() @@ -707,9 +781,20 @@ final class UStateMain private ( * All the other clones used for outputValueCalc, those never * need to add any. */ - private val suspensionTracker = + private val ownSuspensionTracker = new SuspensionTracker(tunable.unparseSuspensionWaitYoung, tunable.unparseSuspensionWaitOld) + // When sharedContext is set, route through the SAME tracker the other + // side of that split uses, or these suspensions would queue where + // nothing ever drains them. + override def suspensionTracker: SuspensionTracker = { + if (sharedContext.isDefined) { + sharedContext.get.suspensionTracker + } else { + ownSuspensionTracker + } + } + def addSuspension(se: Suspension): Unit = { suspensionTracker.trackSuspension(se) } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContext.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContext.scala new file mode 100644 index 0000000000..7ad487859c --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContext.scala @@ -0,0 +1,223 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.iapi.DaffodilTunables +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.* +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.processors.DataProcessor +import org.apache.daffodil.runtime1.processors.SuspensionTracker + +/** + * What build and write genuinely need to share by reference: the + * `SuspensionTracker` (one queue; build opportunistically resolves + * write-created suspensions early against the tree it has already + * built, and write drains whatever remains at the end). + * + * Also owns the lead counter: how far build is ahead of write, + * incremented once per node build constructs and decremented once per + * node write finishes. Once `leadExceedsPrefetchLimit` or the + * pending-suspension backlog exceeds `pendingSuspensionTripLimit` + * (below), build must resume write before continuing its own + * recursion, bounding how far ahead it may run. + */ +final class UnparseSharedContext( + val suspensionTracker: SuspensionTracker, + val dataProc: DataProcessor, + val tunable: DaffodilTunables, + val prefetchLimit: Long +) { + private var buildLead: Long = 0 + + def incrementLead(): Unit = buildLead += 1 + + def decrementLead(): Unit = { + buildLead -= 1 + Assert.invariant(buildLead >= 0) + } + + def currentLead: Long = buildLead + + def leadExceedsPrefetchLimit: Boolean = buildLead > prefetchLimit + + /** + * A second, independent limit on how far build may run ahead of write, + * alongside prefetchLimit: a suspension can be created without moving + * the lead counter, so the lead alone doesn't bound how many pile up + * pending. + */ + def pendingSuspensionTripLimit: Long = tunable.unparsePendingSuspensionTripLimit + + private var buildCoroutine_ : BuildCoroutine = null + private var writeCoroutine_ : WriteCoroutine = null + + def setCoroutines(bc: BuildCoroutine, wc: WriteCoroutine): Unit = { + buildCoroutine_ = bc + writeCoroutine_ = wc + } + def buildCoroutine: BuildCoroutine = buildCoroutine_ + def writeCoroutine: WriteCoroutine = writeCoroutine_ + + private var recordedWriteDone: Maybe[WriteDone] = Nope + + /** + * Resumes writeCoroutine with `signal`, unless write already finished + * (a second resume would park forever), in which case the cached + * WriteDone is returned instead. Always go through this, never resume + * writeCoroutine directly. + */ + def resumeWrite(signal: BuildSignal): WriteSignal = { + if (recordedWriteDone.isDefined) { + recordedWriteDone.get + } else { + val result = buildCoroutine.resume(writeCoroutine, signal) + result match { + case wd: WriteDone => recordedWriteDone = One(wd) + case _ => + } + result + } + } + + /** + * Records write's own result without ever resuming writeCoroutine, + * for when build never needed write started at all and it ran + * directly on build's own thread instead. Still must be recorded + * here so a later abortWrite/resumeWrite sees write as finished. + */ + def recordWriteDone(wd: WriteDone): Unit = { + recordedWriteDone = One(wd) + } + + /** + * Wakes write's thread (spawning it first if never started) after + * build's own thread failed before reaching the normal resumeWrite + * handoff, so it can clean up instead of hanging forever. Blocks until + * that cleanup actually finishes, not just until it starts, so the + * caller never reports an error result while write's cleanup is still + * running on another thread. A no-op if write already finished, or if + * setCoroutines was never reached. + */ + def abortWrite(): Unit = { + if (recordedWriteDone.isEmpty && writeCoroutine_ != null) { + buildCoroutine.resume(writeCoroutine, BuildAborted) + } + } + + /** + * True if `child` may safely be written now: complex/array existing is + * enough; simple needs a value, except hidden/nilled elements (never + * given one) and OVC (deferred via its own Suspension; must not block, + * or an OVC depending on a later sibling's write-time property would deadlock). + */ + private def isChildReady(child: DINode): Boolean = { + if (child.isSimple) { + val s = child.asSimple + child.isHidden || s.isNilled || s.hasValue || s.erd.dpathElementCompileInfo.isOutputValueCalc + } else { + // complex/array, hidden or not; existing is enough + true + } + } + + /** + * True once build's recursion has fully returned. Sticky: once true, + * awaitChild must never again resume buildCoroutine; build's thread has + * moved on to waiting for write's ultimate WriteDone, not another + * WriteNeedsMore cycle, so this can't be a per-call local flag. + */ + private var buildFinished: Boolean = false + + /** + * Records a signal write's coroutine received, without itself resuming + * anyone. Needed for the signal that STARTS write's thread, which can + * legitimately already be BuildFinished; must be recorded before any + * awaitChild call, or tryUnblockWrite would wrongly resume build again. + */ + def observeBuildSignal(signal: BuildSignal): Unit = { + if (signal == BuildFinished) buildFinished = true + else if (signal == BuildAborted) throw new BuildAbortedException + } + + /** + * One attempt at unblocking write: resumes build if not finished yet, + * else retries suspensions directly. False means no progress was made; + * callers must throw AwaitChildStalledException rather than loop again, + * deferring to evalSuspensions(isFinal = true) for the real diagnosis. + */ + private def tryUnblockWrite(): Boolean = { + if (!buildFinished) { + observeBuildSignal(writeCoroutine.resume(buildCoroutine, WriteNeedsMore)) + true + } else { + val before = suspensionTracker.suspensions.length + suspensionTracker.evalSuspensionsUnthrottled() + suspensionTracker.suspensions.length < before + } + } + + /** + * Blocks (parks write's thread) until `parent.child(index)` both exists + * and is ready (isChildReady above), throwing AwaitChildStalledException + * once tryUnblockWrite reports no further progress is possible. + */ + def awaitChild(parent: DINode, index: Int): DINode = { + while (index >= parent.numChildren || !isChildReady(parent.child(index))) { + if (!tryUnblockWrite()) throw new AwaitChildStalledException + } + parent.child(index) + } + + /** + * Blocks until `parent.child(index)` exists, or `parent.isFinal` with no + * child there (none ever will be); for callers asking "is there more + * data at all" rather than awaiting a child known to be coming. A true + * result may still need awaitChild afterward for value readiness. + */ + def childExistsOrFinal(parent: DINode, index: Int): Boolean = { + while (index >= parent.numChildren) { + if (parent.isFinal) return false + if (!tryUnblockWrite()) throw new AwaitChildStalledException + } + true + } +} + +/** + * Thrown when write can make no further progress after build has + * finished and an unthrottled suspension retry made no headway. Caught + * only by the top-level coroutine driver, which still runs its normal + * finalization (the genuine, diagnostic-producing final suspension + * drain) rather than treating this as the final outcome itself. + */ +final class AwaitChildStalledException extends Exception + +/** + * Thrown the moment write's thread observes that build's own thread + * failed with an exception before reaching its normal completion, either + * as the very first signal write's thread ever receives, or as the + * result of a resume from deep inside a block on a pending child. Caught + * only by the top-level coroutine driver, which skips its own normal + * finalization entirely (those invariants and the final suspension drain + * assume a consistently, fully-built tree that a genuine abort may not + * have produced) and just cleans up write's own resources; build's own + * exception, not this one, is what actually gets reported. + */ +final class BuildAbortedException extends Exception diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala index 592544ff0d..7d7d4f8b2f 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala @@ -20,7 +20,9 @@ package org.apache.daffodil.runtime1.processors.unparsers import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.runtime1.dsom.RuntimeSchemaDefinitionError +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.* +import org.apache.daffodil.unparsers.runtime1.WriteUnparser sealed trait Unparser extends Processor { @@ -48,7 +50,6 @@ sealed trait Unparser extends Processor { // keeping track of prior bit order. Finding those has been problematic. // // So this is a temporary fix, until we can figure out where else to do this. - // this match { // bit order only applies to primitives, not combinators, nor "noData" unparsers. case af: AlignmentPrimUnparser => // ok. Don't check bitOrder before Aligning. @@ -72,7 +73,11 @@ sealed trait Unparser extends Processor { ustate.resetFormatInfoCaches() } if (ustate.dataProc.isDefined) ustate.dataProc.get.after(ustate, this) - ustate.setMaybeProcessor(savedProc) + // Restore the prior processor only if one existed. Nope means this is + // the first unparse1 call on a freshly cloned suspension UState, which + // starts with none; resetting to Nope would discard the only context + // it will ever have, which is still needed once the suspension completes. + if (savedProc.isDefined) ustate.setMaybeProcessor(savedProc) } def UE(ustate: UState, s: String, args: Any*) = { @@ -147,7 +152,8 @@ final class ErrorUnparser(override val context: TermRuntimeData = null) final class SeqCompUnparser(context: RuntimeData, val childUnparsers: Array[Unparser]) extends CombinatorUnparser(context) - with ToBriefXMLImpl { + with ToBriefXMLImpl + with WriteUnparser { override val runtimeDependencies = Array() @@ -158,9 +164,27 @@ final class SeqCompUnparser(context: RuntimeData, val childUnparsers: Array[Unpa def unparse(ustate: UState): Unit = { var i = 0 while (i < childUnparsers.length) { - val unparser = childUnparsers(i) + childUnparsers(i).unparse1(ustate) + i += 1 + } + } + + /** + * SeqCompUnparser can wrap any `WriteUnparser` (sequence/choice/ + * hidden-group/delimiter-stack) alongside plain prims: a `WriteUnparser` + * recurses into its own writeContent; everything else (a bare element + * never appears here directly, only wrapped by one) runs via unparse1. + */ + override def writeContent(containerNode: DINode, ustate: UState): Unit = { + var i = 0 + while (i < childUnparsers.length) { + childUnparsers(i) match { + case wu: WriteUnparser => + wu.writeContent(containerNode, ustate) + case cu => + cu.unparse1(ustate) + } i += 1 - unparser.unparse1(ustate) } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala index 9cfff9e970..57ea053ab8 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala @@ -19,6 +19,7 @@ package org.apache.daffodil.unparsers.runtime1 import scala.jdk.CollectionConverters.* +import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.lib.util.MaybeInt @@ -78,13 +79,124 @@ class ChoiceCombinatorUnparser( choiceBranchMap: ChoiceBranchMap, choiceLengthInBits: MaybeInt ) extends CombinatorUnparser(mgrd) - with ToBriefXMLImpl { + with ToBriefXMLImpl + with WriteUnparser { override def nom = "Choice" override val runtimeDependencies = Array() override def childProcessors = choiceBranchMap.childProcessors + /** + * Resolves which branch an already-built child belongs to by keying off + * the child's element identity, blocking until that's known, then + * recurses into the resolved branch. Self-manages advancing/freeing its + * own resolved tree-child position, and applies the same choice-length + * filling as the structural pass. Covers the visible-choice path only; + * the hidden-choice default branch is handled separately below. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = { + val sharedCtx = state.sharedContext.get + val complex = containerNode.asComplex + val childIndex = state.childIndexStack.top.toInt + + val (maybeChildUnparser, resolvedChildIndex): (Maybe[Unparser], Int) = + if (state.withinHiddenNest) { + // A hidden choice's branch is always the single deterministic + // default one (DFDL requires its outcome to be schema-determined, + // not data-driven), so no key/event peek is needed. The tree child at this position, if any, IS this default branch's own content, not something to skip past. + val idx = if (sharedCtx.childExistsOrFinal(complex, childIndex)) { + childIndex + } else { + -1 + } + (Maybe.toMaybe(choiceBranchMap.defaultUnparser), idx) + } else { + // Hidden elements never produce infoset events, so the branch + // lookup keys are built from the first represented child. Build + // still materializes a branch's leading hidden group as actual + // tree children, so skip past those first. + var idx = childIndex + while (sharedCtx.childExistsOrFinal(complex, idx) && complex.child(idx).isHidden) { + idx += 1 + } + if (idx >= complex.numChildren) { + // Build is done and no child ever showed up: the choice resolved + // to a branch with no infoset footprint at all (e.g. an empty + // sequence, or an absent defaultable element); fall back to the + // default/unmapped branch. + (Maybe.toMaybe(choiceBranchMap.defaultUnparser), -1) + } else { + val child = sharedCtx.awaitChild(complex, idx) + val key: ChoiceBranchEvent = ChoiceBranchStartEvent(child.erd.namedQName) + val fromTable = choiceBranchMap.lookupTable.get(key) + if (fromTable != null) { + // An actual match; this tree position genuinely belongs to this + // choice. + (One(fromTable), idx) + } else { + // No branch key matches this child, so it must belong to a + // sibling term after this choice: this choice resolved to a + // branch with no infoset footprint here and consumes no + // tree position. + (Maybe.toMaybe(choiceBranchMap.defaultUnparser), -1) + } + } + } + if (maybeChildUnparser.isEmpty) { + // A real UnparseError, not an internal assertion: a choice with no + // default and no match for what's actually in the tree is + // malformed input, not an invariant violation. + UnparseError( + One(mgrd.schemaFileLocation), + One(state.currentLocation), + "No matching or default choice branch found." + ) + } + val child = if (resolvedChildIndex >= 0) { + complex.child(resolvedChildIndex) + } else { + // A resumable group or ChoiceBranchEmptyUnparser needs no tree child, + // so null is fine, but the unmapped default can also be a bare + // ElementUnparserBase, which DOES need one: writeContent(null, state) + // on that would NPE deep inside it instead of failing clearly here. + if (maybeChildUnparser.get.isInstanceOf[ElementUnparserBase]) { + Assert.invariantFailed( + "Choice resolved to a simple-element default branch with no resolved tree position." + ) + } + null + } + + // True when the resolved branch is itself a resumable group, whose own + // dispatch already advances/frees its tree positions; this choice must + // not also advance/free then, or it double-advances past the branch's + // last child, skipping the following sibling term. + var innerSelfManagesPosition = false + withChoiceLengthFiller(state) { + maybeChildUnparser.get match { + case elemUnp: ElementUnparserBase => elemUnp.writeContent(child, state) + case wu: WriteUnparser => + innerSelfManagesPosition = true + wu.writeContent(containerNode, state) + case emptyUnp: ChoiceBranchEmptyUnparser => + // A branch that optimized to nothing (e.g. a sequence containing + // only an assert); runs its (no-op) unparse, same as any other + // synchronous branch. + emptyUnp.unparse1(state) + case other => + Assert.usageError(s"unhandled choice branch unparser type: $other") + } + } + + // Nothing to advance/free when the branch had no infoset footprint + // (resolvedChildIndex == -1), nor when it's itself a resumable group (innerSelfManagesPosition), which already did so for its own positions. + if (resolvedChildIndex >= 0 && !innerSelfManagesPosition) { + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(resolvedChildIndex, state.releaseUnneededInfoset) + } + } + def unparse(state: UState): Unit = { if (state.withinHiddenNest) { val branchForUnparseIfHidden = choiceBranchMap.defaultUnparser @@ -121,22 +233,36 @@ class ChoiceCombinatorUnparser( val childUnparser = maybeChildUnparser.get state.popTRD(mgrd) state.pushTRD(childUnparser.context.asInstanceOf[TermRuntimeData]) - if (choiceLengthInBits.isDefined) { - val suspendableOp = - new ChoiceUnusedUnparserSuspendableOperation(mgrd, choiceLengthInBits.get) - val choiceUnusedUnparser = - new ChoiceUnusedUnparser(mgrd, choiceLengthInBits.get, suspendableOp) - - suspendableOp.captureDOSStartForChoiceUnused(state) - childUnparser.unparse1(state) - suspendableOp.captureDOSEndForChoiceUnused(state) - choiceUnusedUnparser.unparse(state) - } else { + withChoiceLengthFiller(state) { childUnparser.unparse1(state) } state.popTRD(childUnparser.context.asInstanceOf[TermRuntimeData]) } } + + /** + * Wraps runChosenBranch with the dfdl:choiceLength "unused region" + * filler (no-op if choiceLengthInBits isn't set), capturing DOS + * position before/after via ChoiceUnusedUnparser. The + * state.setProcessor calls are redundant for unparse()'s call site but + * required for writeContent's, which bypasses unparse1 (and its + * equivalent setProcessor) entirely. + */ + private def withChoiceLengthFiller(state: UState)(runChosenBranch: => Unit): Unit = { + if (choiceLengthInBits.isEmpty) { + runChosenBranch + } else { + val suspendableOp = + new ChoiceUnusedUnparserSuspendableOperation(mgrd, choiceLengthInBits.get) + val unusedUnparser = new ChoiceUnusedUnparser(mgrd, choiceLengthInBits.get, suspendableOp) + state.setProcessor(ChoiceCombinatorUnparser.this) + suspendableOp.captureDOSStartForChoiceUnused(state) + runChosenBranch + state.setProcessor(ChoiceCombinatorUnparser.this) + suspendableOp.captureDOSEndForChoiceUnused(state) + unusedUnparser.unparse(state) + } + } } class DelimiterStackUnparser( @@ -145,7 +271,30 @@ class DelimiterStackUnparser( terminatorOpt: Maybe[TerminatorUnparseEv], ctxt: TermRuntimeData, bodyUnparser: Unparser -) extends CombinatorUnparser(ctxt) { +) extends CombinatorUnparser(ctxt) + with WriteUnparser { + + // Hoisted once per instance rather than passed as `pushDelimiterScope`/ + // `bodyUnparser.unparse1` at each call site: an eta-expansion of an + // instance method (or one of its fields) closes over `this`, so it + // allocates a fresh closure on every call otherwise, and both + // writeContent and unparse run once per matching element in the infoset. + private val funcPushDelimiterScope: UState => Unit = pushDelimiterScope + private val funcBodyUnparserUnparse1: UState => Unit = bodyUnparser.unparse1 + + /** + * Pushes the delimiter scope, recurses into the body, and pops only + * once the body's own writeContent (if any) returns, since a pending + * pause still needs the stack for separator writing. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop( + containerNode, + bodyUnparser, + state, + setup = funcPushDelimiterScope, + teardown = (state, _) => state.popDelimiters() + ) override def nom = "DelimiterStack" override def toBriefXML(depthLimit: Int = -1): String = { @@ -163,8 +312,15 @@ class DelimiterStackUnparser( override val runtimeDependencies = (initiatorOpt.toList ++ separatorOpt.toList ++ terminatorOpt.toList).toArray - def unparse(state: UState): Unit = { - // Evaluate Delimiters + def unparse(state: UState): Unit = + withPushPop( + state, + setup = funcPushDelimiterScope, + dispatch = funcBodyUnparserUnparse1, + teardown = (state, _) => state.popDelimiters() + ) + + private def pushDelimiterScope(state: UState): Unit = { val init = if (initiatorOpt.isDefined) initiatorOpt.get.evaluate(state) else EmptyDelimiterStackUnparseNode.empty @@ -174,14 +330,7 @@ class DelimiterStackUnparser( val term = if (terminatorOpt.isDefined) terminatorOpt.get.evaluate(state) else EmptyDelimiterStackUnparseNode.empty - - val node = DelimiterStackUnparseNode(init, sep, term) - - state.pushDelimiters(node) - - bodyUnparser.unparse1(state) - - state.popDelimiters() + state.pushDelimiters(DelimiterStackUnparseNode(init, sep, term)) } } @@ -189,25 +338,52 @@ class DynamicEscapeSchemeUnparser( escapeScheme: EscapeSchemeUnparseEv, ctxt: TermRuntimeData, bodyUnparser: Unparser -) extends CombinatorUnparser(ctxt) { +) extends CombinatorUnparser(ctxt) + with WriteUnparser { override def nom = "EscapeSchemeStack" override def childProcessors = Vector(bodyUnparser) override val runtimeDependencies = Array(escapeScheme) - def unparse(state: UState): Unit = { - // evaluate the dynamic escape scheme in the correct scope. the resulting - // value is cached in the Evaluatable (since it is manually cached) and - // future parsers/evaluatables that use this escape scheme will use that - // cached value. + // Hoisted once per instance rather than passed at each call site: an + // eta-expansion of an instance method, or a lambda referencing an + // instance field like `escapeScheme`, closes over `this`, so it + // allocates a fresh closure on every call otherwise, and both + // writeContent and unparse run once per matching element in the infoset. + private val funcCacheEscapeScheme: UState => Unit = cacheEscapeScheme + private val funcBodyUnparserUnparse1: UState => Unit = bodyUnparser.unparse1 + private val funcInvalidateCache: (UState, Unit) => Unit = + (state, _) => escapeScheme.invalidateCache(state) + + /** + * Caches the escape scheme, recurses into the body, and invalidates + * the cache only once the body's own writeContent (if any) returns, + * since a pending pause still needs the cache for delimiter writing. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop( + containerNode, + bodyUnparser, + state, + setup = funcCacheEscapeScheme, + teardown = funcInvalidateCache + ) + + def unparse(state: UState): Unit = + withPushPop( + state, + setup = funcCacheEscapeScheme, + dispatch = funcBodyUnparserUnparse1, + teardown = funcInvalidateCache + ) + + // Evaluates the dynamic escape scheme in the correct scope; the result is + // cached in the Evaluatable (since it is manually cached), so future + // unparsers/evaluatables that use this escape scheme reuse that cached + // value. + private def cacheEscapeScheme(state: UState): Unit = { escapeScheme.newCache(state) escapeScheme.evaluate(state) - - // Unparse - bodyUnparser.unparse1(state) - - // invalidate the escape scheme cache - escapeScheme.invalidateCache(state) } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala index c1343654ba..b067b76181 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala @@ -24,6 +24,7 @@ import org.apache.daffodil.lib.util.MaybeULong import org.apache.daffodil.runtime1.dpath.SuspendableExpression import org.apache.daffodil.runtime1.dsom.CompiledExpression import org.apache.daffodil.runtime1.infoset.DIComplex +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.infoset.DISimple import org.apache.daffodil.runtime1.infoset.DataValue.DataValuePrimitive import org.apache.daffodil.runtime1.infoset.RetryableException @@ -31,6 +32,7 @@ import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.Evaluatable import org.apache.daffodil.runtime1.processors.UnparseTargetLengthInBitsEv import org.apache.daffodil.runtime1.processors.unparsers.* +import org.apache.daffodil.runtime1.processors.unparsers.MoreTreeAvailable /** * Elements that, when unparsing, have no length specified. @@ -65,6 +67,42 @@ sealed trait RepMoveMixin { } } +/** + * The build-side hookup for bounded lookahead: called once per node, only + * from the Builder tree (via unparseBeginForBuild below), never from + * write's or single-pass's own unparseBegin call. Increments the shared + * lead counter, and once it exceeds the prefetch limit, or too many + * suspensions are pending, resumes write's coroutine and blocks until it + * yields back before continuing build's own recursion. + */ +private object BuildWriteLeadHookup { + + /** + * Must be called after unparseBegin, not unparseEnd: an ancestor's + * increment must happen before any descendant's content is built. A + * write cascade triggered here can reach that ancestor before + * build's own recursive call into it returns; incrementing in + * unparseEnd instead would underflow the counter. + */ + def afterNodeAdded(state: UState): Unit = { + val ctx = state.sharedContext.get + ctx.incrementLead() + // leadExceedsPrefetchLimit alone doesn't bound the retained + // memory of pending suspensions, so pendingSuspensionTripLimit + // separately caps total pendingCount; resuming write here can + // also resolve suspensions immediately. + if ( + ctx.leadExceedsPrefetchLimit || + ctx.suspensionTracker.pendingCount > ctx.pendingSuspensionTripLimit + ) { + // resumeWrite may legitimately send WriteDone here, not just + // WriteNeedsMore, and records that so nothing tries to resume an + // already-finished coroutine again later. + ctx.resumeWrite(MoreTreeAvailable) + } + } +} + /** * The unparser used for an element that has inputValueCalc. * @@ -126,7 +164,8 @@ sealed abstract class ElementUnparserBase( val eReptypeUnparser: Maybe[Unparser] ) extends CombinatorUnparser(erd) with RepMoveMixin - with ElementUnparserStartEndStrategy { + with ElementUnparserStartEndStrategy + with WriteUnparser { final override def childProcessors = (eBeforeUnparser.toList ++ eUnparser.toList ++ eAfterUnparser.toList ++ eReptypeUnparser.toList ++ setVarUnparsers.toList).toVector @@ -161,51 +200,143 @@ sealed abstract class ElementUnparserBase( } } - protected def doBeforeContentUnparser(state: UState): Unit = { + private[runtime1] def doBeforeContentUnparser(state: UState): Unit = { if (eBeforeUnparser.isDefined) eBeforeUnparser.get.unparse1(state) } - protected def doAfterContentUnparser(state: UState): Unit = { + private[runtime1] def doAfterContentUnparser(state: UState): Unit = { if (eAfterUnparser.isDefined) eAfterUnparser.get.unparse1(state) } - protected def runContentUnparser(state: UState): Unit = { - if (eReptypeUnparser.isDefined) { - eReptypeUnparser.get.unparse1(state) - } else if (eUnparser.isDefined) - eUnparser.get.unparse1(state) + /** + * Registers whatever suspensions this element's content depends on + * (dfdl:length, dfdl:outputValueCalc) before content bytes get + * written. No-op by default; overridden by the SpecifiedLength + * unparsers. Called unconditionally from writeContent too, before + * its group-unparser check, so this isn't skipped for + * delimited/escape-scheme-wrapped elements. + */ + private[runtime1] def contentSetup(state: UState): Unit = () + + private[runtime1] def dispatchContentUnparser(state: UState): Unit = { + (eReptypeUnparser.toOption, eUnparser.toOption) match { + case (Some(rep), _) => + rep.unparse1(state) + case (None, Some(eu)) => + eu.unparse1(state) + case _ => // nothing to do: no content unparser applies + } } - override def unparse(state: UState): Unit = { + private[runtime1] def runContentUnparser(state: UState): Unit = { + contentSetup(state) + dispatchContentUnparser(state) + } - if (state.dataProc.isDefined) state.dataProc.value.startElement(state, this) + // Hoisted once per instance rather than passed inline at the unparse() + // call site below: a closure referencing instance fields/methods (erd, + // runContentUnparser) closes over `this`, so it allocates a fresh + // closure on every call otherwise, and unparse runs once per matching + // element in the infoset. writeContent's own dispatch closure is not + // hoisted the same way: it also captures containerNode, a per-call + // parameter, so a fresh closure there is unavoidable regardless. + private val funcUnparseDispatch: UState => Unit = { s => + // We must push the TermRuntimeData for all model-groups, starting + // from the complex type's model-group; simple types have none to + // push. Only unparse's own event-driven dispatch needs this: + // nextElement, its sole reader, resolves a raw event's tag name + // against it, and writeContent never consumes events at all. + if (erd.isComplexType) + s.pushTRD(erd.optComplexTypeModelGroupRuntimeData.get) - unparseBegin(state) + runContentUnparser(s) - captureRuntimeValuedExpressionValues(state) + if (erd.isComplexType) + s.popTRD(erd.optComplexTypeModelGroupRuntimeData.get) + } + /** + * The steps identical whether reached via writeContent's already-built + * containerNode or unparse's own event-consuming attach: before-content, + * the content dispatch itself, after-content, and setVariables. dispatch + * abstracts writeContent's WriteUnparser-bypass check (needed to reach a + * nested group's own writeContent) from unparse's plain, purely + * event-driven runContentUnparser. + */ + private[runtime1] def runElementContent(state: UState, dispatch: UState => Unit): Unit = { + captureRuntimeValuedExpressionValues(state) doBeforeContentUnparser(state) + dispatch(state) + doAfterContentUnparser(state) + computeSetVariables(state) + } - // - // We must push the TermRuntimeData for all model-groups. - // The starting point for this is the model-group of a complex type. - // Simple types don't have model groups, so no pushing those. - // - if (erd.isComplexType) - state.pushTRD(erd.optComplexTypeModelGroupRuntimeData.get) + /** + * Writes this element's content against an already-built + * `containerNode`, without consuming any InfosetInputter events: + * dispatches to whatever `eUnparser` turns out to be, a group + * unparser or a plain simple-element value-writer. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = { + state.currentInfosetNodeStack.push(One(containerNode)) + state.childIndexStack.push(0L) + try { + // contentSetup can suspend, and suspending reads state.processor. + // That's normally set by the ordinary unparse dispatch, which this + // call bypasses entirely, so it's set explicitly here to match. + state.setProcessor(this) + runElementContent( + state, + dispatch = { s => + contentSetup(s) + // eReptypeUnparser takes priority here too, matching + // dispatchContentUnparser: a repType'd element's raw eUnparser can + // itself be group-wrapped, and without this check would wrongly + // delegate to that raw content instead of converting via repType. + eUnparser.toOption match { + case Some(wu: WriteUnparser) if eReptypeUnparser.isEmpty => + wu.writeContent(containerNode, s) + case _ => + dispatchContentUnparser(s) + } + } + ) + // Only a simple node needs finalizing here (complex/array nodes were + // already finalized in build's unparseEnd; re-finalizing trips + // setFinal()'s !isFinal assert). An OVC node may still be valueless + // here, so check hasValue rather than assert it. + if (containerNode.isSimple && !containerNode.isFinal && containerNode.asSimple.hasValue) + containerNode.setFinal() + // Write-side half of the shared lead counter, matching build's own + // once-per-node increment; no-op unless a write-side UState has + // opted in via setSharedContext. + if (state.sharedContext.isDefined) state.sharedContext.get.decrementLead() + // Content/value-length suspensions can only unblock once write has + // actually written the bytes, so this must run here too, not just + // build's unparseEnd, or they pile up unresolved until the final + // isFinal=true call instead of resolving as data becomes available. + state.asInstanceOf[SuspensionCapableUState].evalSuspensions(isFinal = false) + } finally { + // A stall (AwaitChildStalledException) or other exception mid-recursion + // must still unwind this node's own push, or these stacks end up + // unbalanced and later invariant checks fail with a confusing internal + // assertion instead of the real diagnostic. + state.childIndexStack.pop() + state.currentInfosetNodeStack.pop + } + } - runContentUnparser(state) + override def unparse(state: UState): Unit = { - if (erd.isComplexType) - state.popTRD(erd.optComplexTypeModelGroupRuntimeData.get) + if (state.dataProc.isDefined) state.dataProc.value.startElement(state, this) - doAfterContentUnparser(state) + unparseBegin(state) - computeSetVariables(state) + runElementContent(state, dispatch = funcUnparseDispatch) - unparseEnd(state) + unparseEnd(state, isBuild = false) if (state.dataProc.isDefined) state.dataProc.value.endElement(state, this) @@ -298,11 +429,10 @@ class ElementSpecifiedLengthUnparser( override val runtimeDependencies = maybeTargetLengthEv.toArray - override def runContentUnparser(state: UState): Unit = { - computeTargetLength( - state - ) // must happen before run() so that we can take advantage of knowing the length - super.runContentUnparser(state) // setup unparsing, which will block for no valu + // Must happen before dispatchContentUnparser so we can take + // advantage of knowing the length. + override private[runtime1] def contentSetup(state: UState): Unit = { + computeTargetLength(state) } } @@ -324,12 +454,10 @@ class ElementOVCSpecifiedLengthUnparserSuspendableExpression( val diSimple = state.currentInfosetNode.asSimple diSimple.setDataValue(v) - - // - // These are now done in the main unparse, but they will - // suspend if they cannot be evaluated because there is not data value yet. - // - // callingUnparser.computeSetVariables(state) + // Do NOT setFinal here: a chained retry for this value's own + // conversion may still be pending, and finalizing now would trip + // that retry's own not-yet-final assertion; writeContent's own + // hasValue-guarded setFinal covers the synchronous common case. } override protected def maybeKnownLengthInBits(ustate: UState): MaybeULong = MaybeULong(0L) @@ -362,12 +490,14 @@ class ElementOVCSpecifiedLengthUnparser( Assert.invariant(context.dpathElementCompileInfo.isOutputValueCalc) - override def runContentUnparser(state: UState): Unit = { - computeTargetLength( - state - ) // must happen before run() so that we can take advantage of knowing the length - suspendableExpression.run(state) // run the expression. It might or might not have a value. - super.runContentUnparser(state) // setup unparsing, which will block for no valu + override private[runtime1] def contentSetup(state: UState): Unit = { + // Must happen before dispatchContentUnparser so we can take + // advantage of knowing the length. + computeTargetLength(state) + if (!state.currentInfosetNode.asSimple.hasValue) { + // run the expression. It might or might not have a value. + suspendableExpression.run(state) + } } } @@ -381,12 +511,25 @@ sealed trait ElementUnparserStartEndStrategy { * Consumes the required infoset events and changes context so that the * element's DIElement node is the context element. */ - protected def unparseBegin(state: UState): Unit + def unparseBegin(state: UState): Unit /** - * Restores prior context. Consumes end-element event. + * Restores prior context. Consumes end-element event. A freshly-built + * simple node has no value yet (write still has to set it), so isBuild + * defers finalizing it; a complex/array node is always final here. + */ + def unparseEnd(state: UState, isBuild: Boolean): Unit + + /** + * The Builder tree's entry points: same node-creation logic as + * unparseBegin/unparseEnd, plus the build-side lead-counter hookup that + * only ever applies on this side. */ - protected def unparseEnd(state: UState): Unit + final def unparseBeginForBuild(state: UState): Unit = { + unparseBegin(state) + BuildWriteLeadHookup.afterNodeAdded(state) + } + final def unparseEndForBuild(state: UState): Unit = unparseEnd(state, isBuild = true) protected def captureRuntimeValuedExpressionValues(ustate: UState): Unit @@ -403,7 +546,7 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart * Consumes the required infoset events and changes context so that the * element's DIElement node is the context element. */ - final override protected def unparseBegin(state: UState): Unit = { + final override def unparseBegin(state: UState): Unit = { if (erd.isQuasiElement) { // Quasi elements are used for RepType and PrefixedLength, and have no corresponding // events in the infoset inputter. The parent parser will push a DIElement for us to @@ -490,7 +633,7 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart /** * Restores prior context. Consumes end-element event. */ - final override protected def unparseEnd(state: UState): Unit = { + final override def unparseEnd(state: UState, isBuild: Boolean): Unit = { if (erd.isQuasiElement) { // Quasi elements are used for TypeValueCalc, and have no corresponding events in the infoset inputter // The parent parser will handle pushing and poping the Infoset, so we do not need to do anything here. @@ -530,15 +673,14 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart } } - // cur is finished, mark it as final and free if possible. Note that we - // need the container and not the parent of the current element to free - // it. This way if this element is in an array, we free this element - // from the array. We also do not set hidden IVC elements as - // final--although we allow hidden IVC elements when unparsing, they - // never get a value so we can't set them as final without breaking - // assertions. Nothing can access hidden IVC elements, so this should - // not break anything - if (!state.withinHiddenNest || erd.isRepresented) cur.setFinal() + // cur is finished: mark it final and free via its container (not + // parent, so an array-member frees from the array), except hidden + // IVC elements (never get a value) and, for build, SIMPLE elements + // (write still needs to set their actual value afterward). + if ( + (!state.withinHiddenNest || erd.isRepresented) && + !(isBuild && cur.isSimple) + ) cur.setFinal() val curContainer = if (cur.erd.isArray) cur.diParent.maybeLastChild.get else cur.diParent @@ -558,7 +700,7 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart move(state) - state.asInstanceOf[UStateMain].evalSuspensions(isFinal = false) + state.asInstanceOf[SuspensionCapableUState].evalSuspensions(isFinal = false) } } @@ -573,7 +715,7 @@ trait OVCStartEndStrategy extends ElementUnparserStartEndStrategy { /** * For OVC, the behavior w.r.t. consuming infoset events is different. */ - protected final override def unparseBegin(state: UState): Unit = { + final override def unparseBegin(state: UState): Unit = { val ovcElem = if (!state.withinHiddenNest) { // outputValueCalc elements are optional in the infoset. If the next event @@ -628,7 +770,7 @@ trait OVCStartEndStrategy extends ElementUnparserStartEndStrategy { state.currentInfosetNodeStack.push(One(ovcElem)) } - protected final override def unparseEnd(state: UState): Unit = { + final override def unparseEnd(state: UState, isBuild: Boolean): Unit = { // if an OVC element existed, the start AND end events were consumed in // unparseBegin. No need to advance the cursor here. diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala index 070e787290..076b89eb26 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala @@ -17,6 +17,7 @@ package org.apache.daffodil.unparsers.runtime1 +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.ModelGroupRuntimeData import org.apache.daffodil.runtime1.processors.unparsers.* @@ -27,19 +28,38 @@ import org.apache.daffodil.runtime1.processors.unparsers.* * we unwind from the refs, we'll decrement. */ class HiddenGroupCombinatorUnparser(ctxt: ModelGroupRuntimeData, bodyUnparser: Unparser) - extends CombinatorUnparser(ctxt) { + extends CombinatorUnparser(ctxt) + with WriteUnparser { override def childProcessors = Vector(bodyUnparser) override val runtimeDependencies = Array() - def unparse(start: UState): Unit = { - try { - start.incrementHiddenDef() - // unparse - bodyUnparser.unparse1(start) - } finally { - start.decrementHiddenDef() - } - } + // Hoisted once per instance rather than passed as `bodyUnparser.unparse1` + // at the unparse() call site below: an eta-expansion of an instance + // field's method closes over `this`, so it allocates a fresh closure on + // every call otherwise, and unparse runs once per matching element in + // the infoset. + private val funcBodyUnparserUnparse1: UState => Unit = bodyUnparser.unparse1 + + // The hidden-depth counter must stay incremented across any pauses, since + // anything the body writes needs it (e.g. choice-branch resolution + // branches on state.withinHiddenNest, and RepType conversion asserts + // it's never true). + override def writeContent(containerNode: DINode, start: UState): Unit = + writeWithPushPop( + containerNode, + bodyUnparser, + start, + setup = _.incrementHiddenDef(), + teardown = (start, _) => start.decrementHiddenDef() + ) + + def unparse(start: UState): Unit = + withPushPop( + start, + setup = _.incrementHiddenDef(), + dispatch = funcBodyUnparserUnparse1, + teardown = (start, _) => start.decrementHiddenDef() + ) } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala index 3856c0b272..373c56b94d 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala @@ -17,6 +17,7 @@ package org.apache.daffodil.unparsers.runtime1 +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.layers.LayerDriver import org.apache.daffodil.runtime1.processors.SequenceRuntimeData import org.apache.daffodil.runtime1.processors.unparsers.* @@ -28,8 +29,40 @@ class LayeredSequenceUnparser( override def nom = "LayeredSequence" + private def handleLayerThrowable(layerDriver: LayerDriver, t: Throwable): Unit = { + if (layerDriver ne null) { + layerDriver.handleThrowable(t) + } else { + LayerDriver.handleThrowableWithoutLayer(t) + } + } + + // Same setup/teardown as unparse() below, via withLayerTransform. Without + // this override, write-side dispatch would treat this as a plain + // WriteUnparser (inherited), bypassing the layer transform entirely: raw + // bytes, no compression/checksum, no layer error handling. + override def writeContent(containerNode: DINode, state: UState): Unit = { + // Needed for the same reason as ChoiceCombinatorUnparser's writeContent: + // setFinished/cloneForSuspension reach state.bitOrder/state.processor, + // normally set by Unparser.unparse1's wrapper, which this + // recursive-dispatch code bypasses. + state.setProcessor(LayeredSequenceUnparser.this) + withLayerTransform(state) { + LayeredSequenceUnparser.super.writeContent(containerNode, state) + } + } + override def unparse(state: UState): Unit = { + withLayerTransform(state) { + super.unparse(state) + } + } + // Splits off a buffered DOS for the layer to flush through, so fragment + // bits/bitOrder on the original DOS can't affect how the layer flushes + // bytes, then runs the layer driver's transform around runBody. The + // `finally` restoration ensures later writes always reach `layerFollowingDOS`, win or lose. + private def withLayerTransform(state: UState)(runBody: => Unit): Unit = { val originalDOS = state.getDataOutputStream // create a new buffered DOS that this layer will flush to when the layer @@ -49,7 +82,8 @@ class LayeredSequenceUnparser( // TODO: we're not unparsing here, just writing bytes, so perhaps we do not // need this cloned state? Everything in layers is byte-centric, so there is // no issue of fragment bytes. - val formatInfoPre = state.asInstanceOf[UStateMain].cloneForSuspension(layerUnderlyingDOS) + val formatInfoPre = + state.asInstanceOf[SuspensionCapableUState].cloneForSuspension(layerUnderlyingDOS) // mark the original DOS as finished--no more data will be unparsed to it. // If known, this will carry bit position forward to the layerUnderlyingDOS, @@ -74,7 +108,7 @@ class LayeredSequenceUnparser( // unparse the layer body into layerDOS state.setDataOutputStream(layerDOS) - super.unparse(state) + runBody // now we're done unparsing the layer recursively. // While doing that unparsing, the data output stream may have been split, so the // DOS in the state may no longer be the layerDOS. @@ -85,7 +119,7 @@ class LayeredSequenceUnparser( // val endOfLayerUnparseDOS = state.getDataOutputStream val formatInfoPost = - state.asInstanceOf[UStateMain].cloneForSuspension(endOfLayerUnparseDOS) + state.asInstanceOf[SuspensionCapableUState].cloneForSuspension(endOfLayerUnparseDOS) // setFinished on this end-of-layer-unparse data-output-stream ensures // that the layerDOS gets close() called on it. @@ -95,8 +129,13 @@ class LayeredSequenceUnparser( // layer stack is potentially still needed, so // nothing can be cleaned up at this point. } catch { - case t: Throwable if (layerDriver ne null) => layerDriver.handleThrowable(t) - case t: Throwable => LayerDriver.handleThrowableWithoutLayer(t) + // Pure write-side control-flow signals, unrelated to the layer + // itself; rewrapping either would defeat write's own handling + // (a stall diagnostic, or build's abort cleanup) with a raw + // "layer failed" exception. + case e: AwaitChildStalledException => throw e + case e: BuildAbortedException => throw e + case t: Throwable => handleLayerThrowable(layerDriver, t) // otherwise we have no layer driver, so we were unable to load the layer. // just let that propagate. } finally { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala index 1d31e39d1e..6bc692202c 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala @@ -19,6 +19,7 @@ package org.apache.daffodil.unparsers.runtime1 import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.unparsers.* @@ -52,18 +53,25 @@ case class ComplexNilOrContentUnparser( ctxt: ElementRuntimeData, nilUnparser: Unparser, contentUnparser: Unparser -) extends CombinatorUnparser(ctxt) { +) extends CombinatorUnparser(ctxt) + with WriteUnparser { override val runtimeDependencies = Array() override def childProcessors = Vector(nilUnparser, contentUnparser) + private def chooseBodyUnparser(node: DINode): Unparser = + if (node.asComplex.isNilled) nilUnparser else contentUnparser + def unparse(state: UState): Unit = { Assert.invariant(Maybe.WithNulls.isDefined(state.currentInfosetNode)) - val inode = state.currentInfosetNode.asComplex - if (inode.isNilled) - nilUnparser.unparse1(state) - else - contentUnparser.unparse1(state) + chooseBodyUnparser(state.currentInfosetNode).unparse1(state) } + + // Without this override, WriteUnparser dispatch would call + // contentUnparser.unparse1 synchronously on write's state, but it can + // itself be a resumable group unparser expecting live InfosetInputter + // events that don't exist yet (see SpecifiedLengthExplicitImplicitUnparser). + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop(containerNode, chooseBodyUnparser(containerNode), state) } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala index 7c451f1ddb..e20d040c92 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala @@ -26,6 +26,9 @@ import org.apache.daffodil.lib.schema.annotation.props.gen.SeparatorPosition import org.apache.daffodil.lib.schema.annotation.props.gen.SeparatorPosition.* import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.MaybeInt +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.runtime1.infoset.DIComplex +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.ModelGroupRuntimeData import org.apache.daffodil.runtime1.processors.SequenceRuntimeData @@ -120,7 +123,8 @@ class OrderedSeparatedSequenceUnparser( sepMtaUnparserMaybe: Maybe[Unparser], sep: Unparser, childUnparsers: Array[SequenceChildUnparser with Separated] -) extends OrderedSequenceUnparserBase(rd) { +) extends OrderedSequenceUnparserBase(rd) + with WriteUnparser { // Sequences of nothing (no initiator, no terminator, nothing at all) should // have been optimized away Assert.invariant(childUnparsers.length > 0) @@ -129,6 +133,428 @@ class OrderedSeparatedSequenceUnparser( override def childProcessors = childUnparsers.toVector + /** + * In-flight separator-suppression state for one writeContent call: + * whether any term has already written something (`wroteAny`), whichever + * separator is currently deferred pending its term's content + * (`pendingPostfixSeparatorAtTerm` for ssp Never, + * `pendingSuppressibleOp` for AnyEmpty/TrailingEmpty(Strict)), and the + * end-of-sequence TrailingEmpty(Strict) queue (`trailingSuspendedOps`). + */ + private class SeparatorSuppressionState(state: UState) { + + private var wroteAny = false + + // for spos == Postfix, the separator for a represented term must come + // AFTER its content is written, not before. Set by beforeSeparator, + // cleared by afterSeparator once that term's content has been written. + private var pendingPostfixSeparatorAtTerm = false + + // For ssp AnyEmpty/TrailingEmpty(Strict): the in-flight suspension + // beforeSeparator started for the current term, completed by + // afterSeparator. Mirrors pendingPostfixSeparatorAtTerm's role for ssp + // Never, but as a suspension object rather than a flag. + private var pendingSuppressibleOp: SuppressableSeparatorUnparserSuspendableOperation = null + + // For ssp TrailingEmpty/TrailingEmptyStrict: separators deferred until + // the sequence's own end (DFDL requires trailing through the whole + // sequence, not just the local group), matching + // unparseWithSuppression's end-of-loop resolution. + private val trailingSuspendedOps = + scala.collection.mutable.Buffer[SuppressableSeparatorUnparserSuspendableOperation]() + + /** + * ssp=never never omits separators based on content, so a term with + * fewer than maxOccurs actual occurrences still gets one separator per + * missing occurrence. Other policies decide via content length + * instead; irrelevant here. + */ + def writeExtraSeparatorsIfNeverSuppressed( + rep: RepeatingChildUnparser, + numOccurrences: Long + ): Unit = { + if (ssp != Never) return + // Uses erd.maxOccurs, not rep.maxRepeats(state): for + // occursCountKind="expression"/"parsed", maxRepeats(state) is + // Long.MaxValue, which would loop until OOM. An unbounded array's + // maxOccurs is -1, so sepsNeeded goes negative and this loop no-ops. + val sepsNeeded = rep.erd.maxOccurs - numOccurrences + if (sepsNeeded <= 0) return + val numExtraSeps = if ((spos eq Infix) && !wroteAny) { + sepsNeeded - 1 + } else { + sepsNeeded + } + var n = numExtraSeps + while (n > 0) { + unparseJustSeparator(state) + n -= 1 + } + // Deliberately NOT wroteAny = true: this never advances + // state.groupPos, so a following Infix term must still see this as + // unwritten, or it would wrongly get its own separator too. + } + + /** + * occursCountKind="implicit" is positional, so a non-trailing bounded + * array/optional still needs speculative missing-occurrence separators, + * or a later term shifts position. stacksAlreadyPushed: true only right + * after an actual occurrence loop already positioned the stacks. + */ + def writePositionallyRequiredSepsIfSuppressed( + rep: RepeatingChildUnparser with Separated, + numOccurrences: Long, + stacksAlreadyPushed: Boolean + ): Unit = { + if (ssp == Never) return + if ( + (rep.ock ne OccursCountKind.Implicit) || + !rep.isPositional || !rep.isBoundedMax || + (rep.isDeclaredLast && rep.isPotentiallyTrailing) + ) return + // safe: isBoundedMax being true guarantees maxRepeats(state) is the + // actual, finite erd.maxOccurs, never Long.MaxValue. + val maxReps = rep.maxRepeats(state) + if (numOccurrences >= maxReps) return + if (!stacksAlreadyPushed) { + state.pushOccurrenceIndices() + } + var n = numOccurrences + while (n < maxReps) { + beforeSeparator(rep.erd, rep.isKnownStaticallyNotToSuppressSeparator) + afterSeparator() + n += 1 + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + } + if (!stacksAlreadyPushed) { + state.popOccurrenceIndices() + } + } + + /** + * Called before a term/occurrence's content is written, once known + * to actually be written. For ssp Never (or staticallyNotSuppressible), + * Prefix/Infix write immediately and Postfix defers; otherwise + * Prefix/Infix speculatively unparse a suppressible separator. + */ + def beforeSeparator( + trd: TermRuntimeData, + staticallyNotSuppressible: Boolean + ): Unit = { + if (staticallyNotSuppressible || (ssp eq Never)) { + spos match { + case Prefix => unparseJustSeparator(state) + case Infix => if (wroteAny) unparseJustSeparator(state) + case Postfix => pendingPostfixSeparatorAtTerm = true + } + wroteAny = true + return + } + ssp match { + case Never => + Assert.invariantFailed("handled above") + case AnyEmpty | TrailingEmpty | TrailingEmptyStrict => + spos match { + case Prefix | Infix => + if ((spos eq Infix) && !wroteAny) { + // no separator possible; hence, no suppression + } else { + val suspendableOp = + new SuppressableSeparatorUnparserSuspendableOperation( + sepMtaAlignmentMaybe, + sep, + trd + ) + val suppressableSep = SuppressableSeparatorUnparser(sep, trd, suspendableOp) + suppressableSep.unparse1(state) + pendingSuppressibleOp = suspendableOp + } + case Postfix => + val suspendableOp = + new SuppressableSeparatorUnparserSuspendableOperation( + sepMtaAlignmentMaybe, + sep, + trd + ) + suspendableOp.captureDOSForStartOfSeparatedRegionBeforePostfixSeparator(state) + pendingSuppressibleOp = suspendableOp + } + } + wroteAny = true + } + + /** + * Completes whatever beforeSeparator started for a term. Handles ssp + * Never (pendingPostfixSeparatorAtTerm, immediate write) and + * AnyEmpty/TrailingEmpty(Strict) (pendingSuppressibleOp); exactly one + * is ever set, matching beforeSeparator's own ssp branch. + */ + def afterSeparator(): Unit = { + if (pendingPostfixSeparatorAtTerm) { + pendingPostfixSeparatorAtTerm = false + unparseJustSeparator(state) + } + if (pendingSuppressibleOp ne null) { + val op = pendingSuppressibleOp + pendingSuppressibleOp = null + ssp match { + case AnyEmpty => + spos match { + case Prefix | Infix => + op.captureStateAtEndOfPotentiallyZeroLengthRegionFollowingTheSeparator(state) + case Postfix => + op.captureDOSForEndOfSeparatedRegionBeforePostfixSeparator(state) + SuppressableSeparatorUnparser(sep, op.rd, op).unparse1(state) + op.captureStateAtEndOfPotentiallyZeroLengthRegionFollowingTheSeparator(state) + } + case TrailingEmpty | TrailingEmptyStrict => + spos match { + case Prefix | Infix => + trailingSuspendedOps += op + case Postfix => + op.captureDOSForEndOfSeparatedRegionBeforePostfixSeparator(state) + SuppressableSeparatorUnparser(sep, op.rd, op).unparse1(state) + trailingSuspendedOps += op + } + case Never => + Assert.invariantFailed("pendingSuppressibleOp should never be set for ssp Never") + } + } + } + + // The term produced zero occurrences (a different term's child took + // this position, or build finished with nothing more coming); no + // tree node was added, so position doesn't move, but it may still owe + // separators, like an actual occurrence loop's own end-of-term handling. + def zeroOccurrences(rep: RepeatingChildUnparser with Separated): Unit = { + if (ssp eq Never) { + writeExtraSeparatorsIfNeverSuppressed(rep, 0) + } else { + writePositionallyRequiredSepsIfSuppressed(rep, 0, stacksAlreadyPushed = false) + } + } + + // ssp TrailingEmpty(Strict): now that nothing at all remains in this + // sequence, every deferred separator's "after" boundary is this exact + // point. Resolve them all. + def resolveTrailingSuspended(): Unit = { + if ((ssp eq TrailingEmpty) || (ssp eq TrailingEmptyStrict)) { + trailingSuspendedOps.foreach { + _.captureStateAtEndOfPotentiallyZeroLengthRegionFollowingTheSeparator(state) + } + trailingSuspendedOps.clear() + } + } + } + + /** + * Walks childUnparsers positionally against an already-built + * containerNode, writing each term's content/separator per spos; blocks + * via childExistsOrFinal/awaitChild wherever a needed child isn't ready. + * childIndexStack.top tracks position (an array is ONE slot). + */ + override def writeContent(containerNode: DINode, state: UState): Unit = { + val sharedCtx = state.sharedContext.get + val complex = containerNode.asComplex + val sepState = new SeparatorSuppressionState(state) + + var index = 0 + while (index < childUnparsers.length) { + childUnparsers(index) match { + case rep: RepeatingChildUnparser => + writeRepeatingTerm(rep, complex, sharedCtx, state, sepState) + case cu => + writeRequiredTerm(cu, complex, sharedCtx, state, sepState) + } + index += 1 + } + + sepState.resolveTrailingSuspended() + } + + /** + * Writes an array/optional term: either its full occurrence loop (an + * actual `DIArray`, or a scalar optional's single occurrence) or, if the + * term produced no occurrences at all, its zero-occurrences separator + * bookkeeping. + */ + private def writeRepeatingTerm( + rep: RepeatingChildUnparser, + complex: DIComplex, + sharedCtx: UnparseSharedContext, + state: UState, + sepState: SeparatorSuppressionState + ): Unit = { + val idx = state.childIndexStack.top.toInt + if (sharedCtx.childExistsOrFinal(complex, idx)) { + val next = complex.child(idx) + if (next.erd eq rep.erd) { + next match { + case arrayNode: DIArray => + val repSep = rep.asInstanceOf[RepeatingChildUnparser with Separated] + // A dfdl:occursIndex() expression in the occurrence's content + // reads state.occursIndexStack.top, which must track it or + // every occurrence would evaluate as if it were the first. + state.pushOccurrenceIndices() + try { + var arrayOcc = 0 + while (sharedCtx.childExistsOrFinal(arrayNode, arrayOcc)) { + val occNode = sharedCtx.awaitChild(arrayNode, arrayOcc) + // Postfix's separator comes after this occurrence's + // content; afterSeparator runs once it's written. + sepState.beforeSeparator( + repSep.erd, + repSep.isKnownStaticallyNotToSuppressSeparator + ) + repSep.childUnparser + .asInstanceOf[ElementUnparserBase] + .writeContent(occNode, state) + sepState.afterSeparator() + arrayNode.freeChildIfNoLongerNeeded(arrayOcc, state.releaseUnneededInfoset) + arrayOcc += 1 + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + } + // this whole array term is done; it occupies exactly one + // slot among containerNode's own children, regardless of + // how many occurrences it held + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + state.moveOverOneElementChildOnly() + if (ssp eq Never) { + sepState.writeExtraSeparatorsIfNeverSuppressed(repSep, arrayOcc) + } else { + sepState.writePositionallyRequiredSepsIfSuppressed( + repSep, + arrayOcc, + stacksAlreadyPushed = true + ) + } + } finally { + state.popOccurrenceIndices() + } + case scalarOptional => + val repSep = rep.asInstanceOf[RepeatingChildUnparser with Separated] + val readyChild = sharedCtx.awaitChild(complex, idx) + // Same push as a true array's entry (see above); a + // scalar optional is still a RepeatingChildUnparser (just + // one that can never hold more than one occurrence). + state.pushOccurrenceIndices() + try { + // Postfix's separator comes after this element's + // content; afterSeparator runs once it's written. + sepState.beforeSeparator( + repSep.erd, + repSep.isKnownStaticallyNotToSuppressSeparator + ) + repSep.childUnparser + .asInstanceOf[ElementUnparserBase] + .writeContent(readyChild, state) + sepState.afterSeparator() + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + // Matches the DIArray arm's own per-occurrence advancement: + // occursIndexStack.top must reflect the next (missing) + // occurrence's index before writePositionallyRequiredSepsIfSuppressed's + // loop runs, or its first separator would see index 1, not 2. + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + if (ssp eq Never) { + sepState.writeExtraSeparatorsIfNeverSuppressed(repSep, 1) + } else { + sepState.writePositionallyRequiredSepsIfSuppressed( + repSep, + 1, + stacksAlreadyPushed = true + ) + } + } finally { + state.popOccurrenceIndices() + } + } + } else { + // this term produced zero occurrences: a different term's + // child appeared in this tree position instead + sepState.zeroOccurrences(rep.asInstanceOf[RepeatingChildUnparser with Separated]) + } + } else { + // no more children will ever come; this optional/array term is absent + sepState.zeroOccurrences(rep.asInstanceOf[RepeatingChildUnparser with Separated]) + } + } + + /** + * Writes a required term (exactly one occurrence): a simple/complex + * element, a nested bare group, or a statement-only term with no tree + * child of its own. Includes its own separator bookkeeping if represented. + */ + private def writeRequiredTerm( + cu: SequenceChildUnparser with Separated, + complex: DIComplex, + sharedCtx: UnparseSharedContext, + state: UState, + sepState: SeparatorSuppressionState + ): Unit = { + cu.childUnparser match { + case nvi: NewVariableInstanceStartUnparser => + nvi.unparse1(state) + case sv: SetVariableUnparser => + sv.unparse1(state) + case nvi: NewVariableInstanceEndUnparser => + nvi.unparse1(state) + case align: AlignmentPrimUnparser => + // Padding has no non-idempotent side effect (unlike assert/discriminator + // below), so re-running it here is required, not forbidden: build's + // own recursion only ran it against build's no-op-sink DOS. Always + // represented, and never suppressible since its content is deterministic. + if (cu.trd.isRepresented) { + sepState.beforeSeparator(cu.trd, staticallyNotSuppressible = true) + } + align.unparse1(state) + if (cu.trd.isRepresented) { + sepState.afterSeparator() + } + case statementOnly if !statementOnly.isInstanceOf[WriteUnparser] => + // No tree child, so no separator, and nothing to wait on. + () + case _ => + val idx = state.childIndexStack.top.toInt + // A nested bare group (e.g. a choice) may resolve to a branch with no + // infoset footprint at all; once build is done and no child showed up, + // let the group's own dispatch decide. A plain element term always + // needs an actual child, so it always waits unconditionally. + val isGroupTerm = !cu.childUnparser.isInstanceOf[ElementUnparserBase] && + cu.childUnparser.isInstanceOf[WriteUnparser] + if (isGroupTerm) { + if (sharedCtx.childExistsOrFinal(complex, idx)) { + sharedCtx.awaitChild(complex, idx) + } + } else { + sharedCtx.awaitChild(complex, idx) + } + // A non-represented term gets no separator, so wroteAny must not + // flip true. Suppression only applies to terms whose presence is + // uncertain (array/optional, or a bare group not statically known + // to need it), never to a plain element's own content length. + val useSuppression = isGroupTerm && !cu.isKnownStaticallyNotToSuppressSeparator + if (cu.trd.isRepresented) { + sepState.beforeSeparator(cu.trd, staticallyNotSuppressible = !useSuppression) + } + cu.childUnparser match { + case elemUnp: ElementUnparserBase => + elemUnp.writeContent(complex.child(idx), state) + sepState.afterSeparator() + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + case wu: WriteUnparser => + wu.writeContent(complex, state) + sepState.afterSeparator() + case other => + Assert.usageError(s"unhandled sequence term unparser type: $other") + } + } + } + /** * Unparses one occurrence with associated separator (non-suppressable). */ @@ -315,8 +741,7 @@ class OrderedSeparatedSequenceUnparser( val zlDetector = childUnparser.zeroLengthDetector childUnparser match { case unparser: RepOrderedSeparatedSequenceChildUnparser => { - state.arrayIterationIndexStack.push(1L) - state.occursIndexStack.push(1L) + state.pushOccurrenceIndices() val erd = unparser.erd var numOccurrences = 0 val maxReps = unparser.maxRepeats(state) @@ -476,8 +901,7 @@ class OrderedSeparatedSequenceUnparser( // no event (state.inspect returned false) Assert.invariantFailed("No event for unparsing.") } - state.arrayIterationIndexStack.pop() - state.occursIndexStack.pop() + state.popOccurrenceIndices() } case scalarUnparser => trd match { @@ -606,8 +1030,7 @@ class OrderedSeparatedSequenceUnparser( // childUnparser match { case unparser: RepOrderedSeparatedSequenceChildUnparser => { - state.arrayIterationIndexStack.push(1L) - state.occursIndexStack.push(1L) + state.pushOccurrenceIndices() val erd = unparser.erd Assert.invariant(erd.isArray || erd.isOptional) Assert.invariant(erd.isRepresented) // arrays/optionals cannot have inputValueCalc @@ -698,8 +1121,7 @@ class OrderedSeparatedSequenceUnparser( unparser.endArrayOrOptional(erd, state) } - state.arrayIterationIndexStack.pop() - state.occursIndexStack.pop() + state.popOccurrenceIndices() } case scalarUnparser => { unparseOne(scalarUnparser, trd, state) diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala index 0f9b8c0eae..013803b89a 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala @@ -22,6 +22,7 @@ import org.apache.daffodil.lib.schema.annotation.props.gen.LengthUnits import org.apache.daffodil.lib.schema.annotation.props.gen.Representation import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.runtime1.infoset.DIElement +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.infoset.DISimple import org.apache.daffodil.runtime1.infoset.Infoset import org.apache.daffodil.runtime1.processors.CharsetEv @@ -35,7 +36,8 @@ final class SpecifiedLengthExplicitImplicitUnparser( eUnparser: Unparser, erd: ElementRuntimeData, targetLengthInBitsEv: UnparseTargetLengthInBitsEv -) extends CombinatorUnparser(erd) { +) extends CombinatorUnparser(erd) + with WriteUnparser { override val runtimeDependencies = Array() @@ -52,7 +54,7 @@ final class SpecifiedLengthExplicitImplicitUnparser( dcs } - override final def unparse(state: UState): Unit = { + private def checkVariableWidthComplexType(state: UState): Unit = { lazy val dcs = getCharset(state) if ( erd.impliedRepresentation == Representation.Text && @@ -66,10 +68,36 @@ final class SpecifiedLengthExplicitImplicitUnparser( lengthKind.toString, lengthUnits.toString ) - } else { - eUnparser.unparse1(state) } } + + // Hoisted once per instance rather than passed at each call site below: + // an eta-expansion of an instance method (or field) closes over `this`, + // so it allocates a fresh closure on every call otherwise, and both + // writeContent and unparse run once per matching element in the infoset. + private val funcCheckVariableWidthComplexType: UState => Unit = checkVariableWidthComplexType + private val funcEUnparserUnparse1: UState => Unit = eUnparser.unparse1 + + override final def unparse(state: UState): Unit = + withPushPop( + state, + setup = funcCheckVariableWidthComplexType, + dispatch = funcEUnparserUnparse1, + teardown = (_, _) => () + ) + + // Without this, a SeqCompUnparser wrapping this class would treat + // eUnparser as a synchronous call via its generic fallback, but + // eUnparser can itself be a resumable group unparser expecting live + // InfosetInputter events, desyncing build's event stream entirely. + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop( + containerNode, + eUnparser, + state, + setup = funcCheckVariableWidthComplexType, + teardown = (_, _) => () + ) } /** @@ -139,13 +167,55 @@ class SpecifiedLengthPrefixedUnparser( override val lengthUnits: LengthUnits, override val prefixedLengthAdjustmentInUnits: Long ) extends CombinatorUnparser(erd) - with CalculatedPrefixedLengthUnparserMixin { + with CalculatedPrefixedLengthUnparserMixin + with WriteUnparser { override val runtimeDependencies = Array() override def childProcessors = Vector(prefixedLengthUnparser, eUnparser) - override def unparse(state: UState): Unit = { + // Hoisted once per instance rather than passed at the unparse() call + // site below: an eta-expansion of an instance method (or field) closes + // over `this`, so it allocates a fresh closure on every call otherwise, + // and unparse runs once per matching element in the infoset. + // writeContent's own setup/teardown below are NOT hoisted the same way: + // its teardown also captures containerNode, a per-call parameter, so a + // fresh closure there is unavoidable regardless. + private val funcPushDetachedPrefixLengthElement: UState => DISimple = + pushDetachedPrefixLengthElement + private val funcEUnparserUnparse1: UState => Unit = eUnparser.unparse1 + private val funcUnparseTeardown: (UState, DISimple) => Unit = { (state, plElem) => + resolvePrefixLength(state, state.currentInfosetNode.asInstanceOf[DIElement], plElem) + } + + override def unparse(state: UState): Unit = + withPushPop( + state, + setup = funcPushDetachedPrefixLengthElement, + dispatch = funcEUnparserUnparse1, + teardown = funcUnparseTeardown + ) + + // Without this, WriteUnparser dispatch (a plain recursive-dispatch + // fallback for a group-wrapped eUnparser) would call eUnparser.unparse1 + // synchronously, but it can itself be a resumable group unparser + // expecting live InfosetInputter events. + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop( + containerNode, + eUnparser, + state, + setup = funcPushDetachedPrefixLengthElement, + teardown = { (state, plElem) => + // resolvePrefixLength (via assignPrefixLength/suspension.run) + // expects state.processor to already be set, normally done by + // Unparser.unparse1, which this recursive-dispatch path bypasses. + state.setProcessor(SpecifiedLengthPrefixedUnparser.this) + resolvePrefixLength(state, containerNode.asInstanceOf[DIElement], plElem) + } + ) + + private def pushDetachedPrefixLengthElement(state: UState): DISimple = { // Create a "detached" DIDocument with a single child element that the // prefix length will be parsed to. This creates a completely new // infoset and parses to that, so care is taken to ensure this infoset @@ -160,10 +230,10 @@ class SpecifiedLengthPrefixedUnparser( state.currentInfosetNodeStack.push(One(plElem)) prefixedLengthUnparser.unparse1(state) state.currentInfosetNodeStack.pop + plElem + } - val elem = state.currentInfosetNode.asInstanceOf[DIElement] - eUnparser.unparse1(state) - + private def resolvePrefixLength(state: UState, elem: DIElement, plElem: DISimple): Unit = { if (elem.contentLength.maybeLengthInBits().isDefined) { // If we were able to immediately calculate the length of the element, // then just set it as the value of the detached element created above so diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala index 8b4488a884..cf84670fa2 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala @@ -18,6 +18,9 @@ package org.apache.daffodil.unparsers.runtime1 import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.schema.annotation.props.gen.OccursCountKind +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.runtime1.infoset.DIComplex +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.SequenceRuntimeData import org.apache.daffodil.runtime1.processors.TermRuntimeData @@ -56,7 +59,8 @@ class RepOrderedUnseparatedSequenceChildUnparser( class OrderedUnseparatedSequenceUnparser( rd: SequenceRuntimeData, childUnparsers: Array[SequenceChildUnparser] -) extends OrderedSequenceUnparserBase(rd) { +) extends OrderedSequenceUnparserBase(rd) + with WriteUnparser { // Sequences of nothing (no initiator, no terminator, nothing at all) should // have been optimized away @@ -66,6 +70,141 @@ class OrderedUnseparatedSequenceUnparser( override def childProcessors = childUnparsers.toVector + /** + * Walks childUnparsers positionally against an already-built + * containerNode, same shape as a separated sequence but without any + * separator writing. Without a dedicated writeContent override here, + * the outer dispatch's WriteUnparser check would be false, falling + * through to the single-pass-style dispatch path against an untouched + * inputter. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = { + val sharedCtx = state.sharedContext.get + val complex = containerNode.asComplex + + var index = 0 + while (index < childUnparsers.length) { + childUnparsers(index) match { + case rep: RepeatingChildUnparser => + writeRepeatingTerm(rep, complex, sharedCtx, state) + case cu => + writeRequiredTerm(cu, complex, sharedCtx, state) + } + index += 1 + } + } + + /** + * Writes an array/optional term's full occurrence loop, if it produced + * any occurrences; nothing to do otherwise. There's no separator + * bookkeeping to account for either way, since this sequence has none. + */ + private def writeRepeatingTerm( + rep: RepeatingChildUnparser, + complex: DIComplex, + sharedCtx: UnparseSharedContext, + state: UState + ): Unit = { + val idx = state.childIndexStack.top.toInt + if (sharedCtx.childExistsOrFinal(complex, idx)) { + val next = complex.child(idx) + if (next.erd eq rep.erd) { + next match { + case arrayNode: DIArray => + // A dfdl:occursIndex() expression in the occurrence's own + // content reads state.occursIndexStack.top, which must track + // the actual occurrence being written. + state.pushOccurrenceIndices() + try { + var arrayOcc = 0 + while (sharedCtx.childExistsOrFinal(arrayNode, arrayOcc)) { + val occNode = sharedCtx.awaitChild(arrayNode, arrayOcc) + rep.childUnparser + .asInstanceOf[ElementUnparserBase] + .writeContent(occNode, state) + arrayNode.freeChildIfNoLongerNeeded(arrayOcc, state.releaseUnneededInfoset) + arrayOcc += 1 + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + } + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + state.moveOverOneElementChildOnly() + } finally { + state.popOccurrenceIndices() + } + case scalarOptional => + val readyChild = sharedCtx.awaitChild(complex, idx) + // Same reason as the array case above: a dfdl:occursIndex() + // expression in the occurrence's own content reads + // state.occursIndexStack.top. + state.pushOccurrenceIndices() + try { + rep.childUnparser + .asInstanceOf[ElementUnparserBase] + .writeContent(readyChild, state) + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + } finally { + state.popOccurrenceIndices() + } + } + } + // else: this term produced zero occurrences; a different term's + // child appeared in this tree position instead, and there's no + // separator bookkeeping to resolve for it here, unlike the + // separated sequence's own zeroOccurrences. + } + // else: no more children will ever come; this optional/array term is + // absent, again with nothing further to do about it here. + } + + /** + * Writes a required term: a simple or complex element, a nested bare + * group, or a statement-only term with no tree child of its own. + */ + private def writeRequiredTerm( + cu: SequenceChildUnparser, + complex: DIComplex, + sharedCtx: UnparseSharedContext, + state: UState + ): Unit = { + cu.childUnparser match { + // A plain scalar element term: await its one tree child, write + // its content, then advance the element-child position. + case elemUnp: ElementUnparserBase => + val idx = state.childIndexStack.top.toInt + val child = sharedCtx.awaitChild(complex, idx) + elemUnp.writeContent(child, state) + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + case wu: WriteUnparser => + val idx = state.childIndexStack.top.toInt + // This nested group (e.g. a choice) may have resolved to a branch + // with no infoset footprint at all; only wait for a child's + // readiness when childExistsOrFinal says one genuinely exists. + if (sharedCtx.childExistsOrFinal(complex, idx)) { + sharedCtx.awaitChild(complex, idx) + } + wu.writeContent(complex, state) + case nvi: NewVariableInstanceStartUnparser => + nvi.unparse1(state) + case sv: SetVariableUnparser => + sv.unparse1(state) + case nvi: NewVariableInstanceEndUnparser => + nvi.unparse1(state) + case align: AlignmentPrimUnparser => + // Padding has no non-idempotent side effect (unlike assert/discriminator + // below), so re-running it here is required, not forbidden: build's + // own recursion only ran it against build's no-op-sink DOS. + align.unparse1(state) + case _ => + // No tree child, so nothing to wait on. + () + } + } + /** * Unparses one iteration of an array/optional element */ @@ -101,8 +240,7 @@ class OrderedUnseparatedSequenceUnparser( // childUnparser match { case unparser: RepeatingChildUnparser => { - state.arrayIterationIndexStack.push(1L) - state.occursIndexStack.push(1L) + state.pushOccurrenceIndices() val erd = unparser.erd var numOccurrences = 0 val maxReps = unparser.maxRepeats(state) @@ -177,8 +315,7 @@ class OrderedUnseparatedSequenceUnparser( ) } - state.arrayIterationIndexStack.pop() - state.occursIndexStack.pop() + state.popOccurrenceIndices() } // case scalarUnparser => { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala new file mode 100644 index 0000000000..1a68ed6d4c --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala @@ -0,0 +1,95 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.unparsers.runtime1 + +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.processors.unparsers.* + +/** + * Write-side dispatch for the build/write-prefetch unparse path, by + * ordinary recursive calls, using `UnparseSharedContext.awaitChild` to + * block (park write's coroutine thread, see `BuildWriteCoroutines.scala`) + * wherever a needed child doesn't yet exist or isn't ready. + */ +trait WriteUnparser { + + // Writes containerNode's content via direct recursive calls on write's + // own coroutine thread - "where we are" is just the JVM call stack, not + // a return-value state machine. May block (via awaitChild) until a + // needed child exists and is ready, then resumes where it left off. + def writeContent(containerNode: DINode, state: UState): Unit + + // Shared push-once/pop-once skeleton, used by both unparse() and + // writeContent() implementations that need one: setup runs before + // dispatch, teardown runs once dispatch returns (even on exception, + // e.g. a stall caught higher up and followed by finishWriteSide's + // invariant checks against this same state) with setup's result + // threaded through (e.g. a detached element from setup to teardown). + protected def withPushPop[A]( + state: UState, + setup: UState => A, + dispatch: UState => Unit, + teardown: (UState, A) => Unit + ): Unit = { + val setupResult = setup(state) + try { + dispatch(state) + } finally { + teardown(state, setupResult) + } + } + + // withPushPop specialized for writeContent's own dispatch: bodyUnparser + // is dispatched to writeContent if it's a WriteUnparser, else plain + // unparse1. + protected def writeWithPushPop[A]( + containerNode: DINode, + bodyUnparser: Unparser, + state: UState, + setup: UState => A, + teardown: (UState, A) => Unit + ): Unit = + withPushPop( + state, + setup = setup, + dispatch = { s => + bodyUnparser match { + case wu: WriteUnparser => wu.writeContent(containerNode, s) + case _ => bodyUnparser.unparse1(s) + } + }, + teardown = teardown + ) + + /** + * `writeWithPushPop` for a combinator with nothing to push or pop; just + * the dispatch-to-`writeContent`-or-`unparse1` part. + */ + protected def writeWithPushPop( + containerNode: DINode, + bodyUnparser: Unparser, + state: UState + ): Unit = + writeWithPushPop( + containerNode, + bodyUnparser, + state, + setup = (_: UState) => (), + teardown = (_: UState, _: Unit) => () + ) +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/core/outputValueCalc/TestOutputValueCalcSuspensionRegressions.scala b/daffodil-core/src/test/scala/org/apache/daffodil/core/outputValueCalc/TestOutputValueCalcSuspensionRegressions.scala new file mode 100644 index 0000000000..527fc5bf8c --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/core/outputValueCalc/TestOutputValueCalcSuspensionRegressions.scala @@ -0,0 +1,230 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.core.outputValueCalc + +import java.nio.channels.Channels + +import org.apache.daffodil.core.compiler.Compiler +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter +import org.apache.daffodil.runtime1.processors.DataProcessor + +import org.junit.Assert.* +import org.junit.Test + +/** + * Regression guards for evalSuspensionQueue's buildResolvableOnly + * restructuring: single-pass unparse doesn't use that mode, but must + * still resolve these scenarios exactly as before. + */ +class TestOutputValueCalcSuspensionRegressions { + + private val example = XMLUtils.EXAMPLE_NAMESPACE + + private val readsOvcRecordCount = 3 + + private def readsOvcSchema = { + SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + } + + private def readsOvcInfoset = { + val recordXml = (1 to readsOvcRecordCount).map { i => + + {f"T$i%03d"} + {f"data$i%04d"} + + } + +
+ {recordXml} + + } + + @Test def testOvcReadsOvcResolvesViaFinalDrain(): Unit = { + val compiler = Compiler().withTunables(Map("unparseSuspensionWaitOld" -> "1000000")) + val pf = compiler.compileNode(readsOvcSchema) + if (pf.isError) fail(pf.getDiagnostics.toString) + val dp = pf.onPath("/").asInstanceOf[DataProcessor] + if (dp.isError) fail(dp.getDiagnostics.toString) + + val outputStream = new java.io.ByteArrayOutputStream() + val out = Channels.newChannel(outputStream) + val inputter = new ScalaXMLInfosetInputter(readsOvcInfoset) + val actual = dp.unparse(inputter, out) + out.close() + assertFalse(actual.getDiagnostics.toString, actual.isProcessingError) + + val unparsed = outputStream.toString + // lenA reads lenB's own value (not its length): both should equal + // record[readsOvcRecordCount]/data's length, 8. + assertEquals(" 8 8", unparsed.substring(0, 8)) + + val recordsPart = unparsed.substring(8) + val expectedRecords = + (1 to readsOvcRecordCount).map(i => f"T$i%03d" + f"data$i%04d").mkString + assertEquals(expectedRecords, recordsPart) + } + + private val pendingRetryRecordCount = 12 + + private def pendingRetrySchema = { + SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + } + + private def pendingRetryInfoset = { + val recordXml = (1 to pendingRetryRecordCount).map { i => + + {f"T$i%03d"} + {f"data$i%04d"} + + } + +
+ {recordXml} + + } + + // Every len field is a fixed-length "data" element (8 bytes), so all + // four should compute to 8 regardless of which record they target. + private def pendingRetryExpectedOutput = { + val header = " 8 8 8 8" + val records = (1 to pendingRetryRecordCount).map(i => f"T$i%03d" + f"data$i%04d").mkString + header + records + } + + @Test def testManyForceRetryCyclesResolveCorrectly(): Unit = { + TestUtils.testUnparsing( + pendingRetrySchema, + pendingRetryInfoset, + pendingRetryExpectedOutput, + tunables = Map("unparseSuspensionWaitOld" -> "1", "unparseSuspensionWaitYoung" -> "1") + ) + } + + @Test def testManyForceRetryCyclesDefaultTunablesStillMatch(): Unit = { + // Same schema at default tunables: confirms the pass above isn't + // vacuous (e.g. a schema mistake unrelated to the tunable). + TestUtils.testUnparsing(pendingRetrySchema, pendingRetryInfoset, pendingRetryExpectedOutput) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/lib/util/TestMStack.scala b/daffodil-core/src/test/scala/org/apache/daffodil/lib/util/TestMStack.scala index b4cebcdbcb..5905e6deb3 100644 --- a/daffodil-core/src/test/scala/org/apache/daffodil/lib/util/TestMStack.scala +++ b/daffodil-core/src/test/scala/org/apache/daffodil/lib/util/TestMStack.scala @@ -17,6 +17,11 @@ package org.apache.daffodil.lib.util +import org.apache.daffodil.lib.util.Maybe.* + +import org.junit.Assert.* +import org.junit.Test + /** * Compare MStack performance to ArrayStack. It should be faster for primitives */ @@ -24,6 +29,107 @@ class TestMStack { var junk: Long = 0 + // Copying into a sized destination must match copying into a default one. + @Test def testMStackOfCopyFromSizedDestinationMatchesDefault(): Unit = { + val source = new MStackOf[String] + source.push("a") + source.push("b") + source.push("c") + + val sizedDest = new MStackOf[String](source.length) + sizedDest.copyFrom(source) + + val defaultDest = new MStackOf[String] + defaultDest.copyFrom(source) + + assertEquals(defaultDest.toList, sizedDest.toList) + assertEquals(3, sizedDest.length) + assertEquals("c", sizedDest.top) + } + + // A sized-to-exact-depth destination must still grow past capacity. + @Test def testMStackOfSizedDestinationStillGrowsCorrectly(): Unit = { + val source = new MStackOf[String] + source.push("a") + source.push("b") + + val sizedDest = new MStackOf[String](source.length) + sizedDest.copyFrom(source) + + sizedDest.push("c") + sizedDest.push("d") + sizedDest.push("e") + + assertEquals(5, sizedDest.length) + assertEquals("e", sizedDest.top) + assertEquals(List("e", "d", "c", "b", "a"), sizedDest.toList) + } + + /** Same sized-clone pattern, but for MStackOfMaybe. */ + @Test def testMStackOfMaybeCopyFromSizedDestinationMatchesDefault(): Unit = { + val source = new MStackOfMaybe[String] + source.push(One("x")) + source.push(Nope) + source.push(One("z")) + + val sizedDest = new MStackOfMaybe[String](source.length) + sizedDest.copyFrom(source) + + val defaultDest = new MStackOfMaybe[String] + defaultDest.copyFrom(source) + + assertEquals(defaultDest.toListMaybe, sizedDest.toListMaybe) + assertEquals(3, sizedDest.length) + assertEquals(One("z"), sizedDest.top) + assertEquals(One("z"), sizedDest.pop) + assertEquals(Nope, sizedDest.pop) + assertEquals(One("x"), sizedDest.pop) + assertTrue(sizedDest.isEmpty) + } + + /** Same sized-construction pattern, for the primitive-specialized variants. */ + @Test def testMStackOfBooleanSizedConstructionWorks(): Unit = { + val stk = MStackOfBoolean(3) + stk.push(true) + stk.push(false) + stk.push(true) + assertEquals(3, stk.length) + assertEquals(true, stk.pop()) + assertEquals(false, stk.pop()) + assertEquals(true, stk.pop()) + } + + @Test def testMStackOfIntSizedConstructionWorks(): Unit = { + val stk = MStackOfInt(2) + stk.push(1) + stk.push(2) + stk.push(3) + assertEquals(3, stk.length) + assertEquals(3, stk.pop()) + assertEquals(2, stk.pop()) + assertEquals(1, stk.pop()) + } + + @Test def testMStackOfLongSizedConstructionWorks(): Unit = { + val stk = MStackOfLong(1) + stk.push(10L) + stk.push(20L) + assertEquals(2, stk.length) + assertEquals(20L, stk.pop()) + assertEquals(10L, stk.pop()) + } + + // trackMaxSizeReached is a `final val`; confirms it's off by default + // and maxSizeReached stays 0 while it is. + @Test def testMaxSizeReachedDisabledByDefault(): Unit = { + assertEquals(false, MStack.trackMaxSizeReached) + val stk = new MStackOf[String] + stk.push("a") + stk.push("b") + stk.push("c") + assertEquals(0, stk.maxSizeReached) + } + /** * This test compares MStackOfLong to ArrayStack[Long]. * diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/TestBuildWritePrefetchDataProcessor.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/TestBuildWritePrefetchDataProcessor.scala new file mode 100644 index 0000000000..c94be64fa5 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/TestBuildWritePrefetchDataProcessor.scala @@ -0,0 +1,1580 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors + +import java.io.ByteArrayOutputStream +import java.nio.charset.StandardCharsets +import scala.jdk.CollectionConverters.* + +import org.apache.daffodil.api +import org.apache.daffodil.core.compiler.Compiler +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.externalvars.ExternalVariablesLoader +import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter + +import org.junit.Assert.* +import org.junit.Test + +/** + * Exercises DataProcessor.unparse with useBuildWritePrefetch enabled, + * confirming byte-identical output vs. single-pass across many schema shapes. + */ +class TestBuildWritePrefetchDataProcessor { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + /** A hidden, constant-valued OVC probe wired in via `dfdl:hiddenGroupRef="ex:ovcProbe"`. + * Needed because `hasAnyPrefetchBeneficialOVC` is false for a schema with no resolvable + * OVC, which would silently fall back to single-pass regardless of the + * `useBuildWritePrefetch` tunable; contributes a leading "Z" byte to expected output. */ + val ovcProbeGroup: scala.xml.Elem = + + + + + + + @Test def testOVCSuspensionSchemaMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = 005 + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + // dfdl:textNumberPadCharacter="0"/textNumberJustification="right" produces + // actual, schema-configured zero-padding, not generic fillByte-based padding. + assertEquals("006005", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + @Test def testArrayChoiceSeparatorSchemaMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + +
H
+ a + b + c + X +
+ + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("ZH,a,b,c,X", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A choice reached while withinHiddenNest: both sides pick + // defaultUnparser unconditionally, since the outcome is schema-determined. + @Test def testHiddenChoiceMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = 3 + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("[hello,3", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A postfix-separated array as the last term: the separator must + // still be written after the final occurrence, not dropped. + @Test def testTrailingArrayPostfixSeparatorMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + a + b + c + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Za\nb\nc\n", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A choice branch with zero tree footprint (absent optional element) + // must still write its own initiator, or write's dispatch can hang. + @Test def testChoiceBranchWithAbsentOptionalElementMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + , + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = 1 + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Zfirst_defaultable1", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + /** Same schema as above, but with "opt" PRESENT (an actual match, not the empty-branch fallback). */ + @Test def testChoiceBranchWithPresentOptionalElementMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + , + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + 0 + 1 + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Zfirst_defaultable01", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + @Test def testManyOccurrenceArrayWithSmallPrefetchLimitMatchesSinglePass(): Unit = { + val numItems = 40 + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val items = (0 until numItems).map(i => {s"i$i"}) + val infoset = + + {items} + + + // A tiny lookahead window forces many build<->write handoffs across this array's 40 + // occurrences. BoundedPrefetchTest proves the lead counter stays bounded; this test + // just confirms the tunable threads through DataProcessor.unparse correctly and the + // output is still byte-perfect. + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes( + sch, + infoset, + extraTunables = Map("unparsePrefetchWindowNodes" -> "3") + ) + assertEquals( + "Z" + (0 until numItems).map(i => s"i$i").mkString(","), + new String(singlePassBytes, StandardCharsets.US_ASCII) + ) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A bare nested xs:sequence has no tree node of its own; the enclosing + // writeContent must not advance/free a tree-child position recursing into it. + @Test def testNestedBareSequenceMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + B + I1 + I2 + A + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + // The nested xs:sequence has no dfdl:separator of its own and doesn't inherit the + // outer's "," (a local property of the outer xs:sequence's annotation, not pushed + // down to nested groups), so it compiles as unseparated: no separator between + // inner1/inner2, but the outer's separator still appears before/after the nested group. + assertEquals("ZB,I1I2,A", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A fixed-length choice must pad unused space after a shorter branch, + // not just write the branch content. + @Test def testFixedLengthChoicePaddingMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + // typeB is 2 bytes; the choice's declared length is 5 bytes, so 3 + // bytes of ChoiceUnusedUnparser filler should follow it. Plus the + // leading 1-byte ovcProbe. + val infoset = XY + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals(6, singlePassBytes.length) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A hidden group's elements never get hasValue=true; write's dispatch + // must special-case that instead of blocking on it forever. + @Test def testHiddenGroupMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + + + + + , + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = V + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("HV", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Initiator/terminator with no separator: same delimiter-frame code + // path as the separator tests above, via a different property combination. + @Test def testInitiatorTerminatorMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + A + B + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z[AB]", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Escape-scheme state is confined entirely to write-only unparsers; + // confirms that end to end with no frame changes needed. + @Test def testEscapeSchemeMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + + + + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + one, two + three + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Zone#, two,three", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A dfdlx:layer-wrapped sequence: confirms the layer's DOS-splitting + // setup produces byte-identical output under prefetch. + @Test def testLayeredSequenceMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + // s1 is exactly 8 bytes (the layer's declared fixedLength), long enough that + // FixedLengthOutputStream.write's accumulate-then-auto-close-on-count-==fixedLength + // logic runs across several bytes, not just one or two; a short value could pass + // while still masking an off-by-one in the accumulation. + val infoset = + + ABCDEFGH + Q + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("ZABCDEFGHQ", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // FixedLengthLayer's length-exceeded error must surface the same way + // under prefetch, not hang write's coroutine or leak a raw exception. + @Test def testLayeredSequenceLengthMismatchErrorMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + // s1 is 10 bytes, 2 more than the layer's declared fixedLength=8. + val infoset = + + ABCDEFGHIJ + Q + + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassOut = new ByteArrayOutputStream() + val singlePassRes = + singlePassDp.unparse(new ScalaXMLInfosetInputter(infoset), singlePassOut) + assertTrue("expected a failed UnparseResult, not a successful one", singlePassRes.isError) + assertTrue( + singlePassRes.getDiagnostics + .get(0) + .getMessage + .contains("exceeded fixed layer length of 8") + ) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchOut = new ByteArrayOutputStream() + val prefetchRes = prefetchDp.unparse(new ScalaXMLInfosetInputter(infoset), prefetchOut) + assertTrue("expected a failed UnparseResult, not a successful one", prefetchRes.isError) + assertTrue( + prefetchRes.getDiagnostics + .get(0) + .getMessage + .contains("exceeded fixed layer length of 8") + ) + } + + // A chained suspension: computed1's OVC references computed2 (itself + // an OVC forward reference), two levels deep instead of one. + @Test def testNestedOVCSuspensionsMatchSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = 005 + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("007006005", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Binary integers (every other test here is textual), through an + // array to also exercise the repeating-child frame path. + @Test def testBinaryIntArrayMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + + + + + , + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + 1 + 2 + 3 + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertArrayEquals( + Array[Byte]('Z'.toByte, 0, 0, 0, 1, 0, 0, 0, 2, 0, 0, 0, 3), + singlePassBytes + ) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A variable-length dfdl:length expression (every other test here is + // constant), exercising computeTargetLength's non-constant branch. + @Test def testVariableLengthExpressionMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + 5 + hello + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z05hello", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A prefixed-length element: the prefix depends on the content's + // unparsed length, a forward dependency like OVC but length-driven. + @Test def testPrefixedLengthMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + , + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = hello + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z05hello", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A prefixed-length element with complex (not simple) content, so + // write's dispatch must recurse into the wrapped content unparser. + @Test def testPrefixedLengthComplexContentMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + , + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + + hi + bye + + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z06hi,bye", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // An OVC referencing fn:count of a preceding array, resolvable + // directly against the tree: the ordinary non-suspending OVC path. + @Test def testOVCCountOfPrecedingArrayMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = + + 1 + 2 + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("122", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A large array where build fully finishes before write's coroutine + // starts, so write's first signal is BuildFinished, not a live coroutine. + @Test def testOVCCountOfPrecedingArrayAfterBuildFullyFinishesMatchesSinglePass(): Unit = { + val numItems = 500 + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val items = (0 until numItems).map(i => {i % 10}) + val infoset = + + {items} + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A dynamic dfdl:terminator expression referencing a preceding + // array's count: navigation-only, resolvable directly against the tree. + @Test def testDynamicTerminatorReferencingArrayCountMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Zy", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // An IVC field (no unparse effect) referencing fn:exists over a + // nested array: exercises ordinary navigation past a nested array. + @Test def testIVCExistsOverNestedArrayMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + + 1 + 2 + 3 + 4 + + true + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z1,2,3,4", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Aggressive throttling forces every "len" suspension to be retried + // repeatedly; guards against re-running an already-resolved one. + @Test def testManyValueLengthForwardReferencesWithAggressiveThrottlingMatchesSinglePass() + : Unit = { + val numRecords = 25 + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val records = (0 until numRecords).map(i => {s"value$i"}) + val infoset = + + {records} + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes( + sch, + infoset, + extraTunables = Map( + "unparsePrefetchWindowNodes" -> "2", + "unparseSuspensionWaitYoung" -> "1", + "unparseSuspensionWaitOld" -> "1" + ) + ) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A repeating NVI scope with a valueLength OVC reading it, letting + // build race many iterations ahead: guards against a stale iteration's value. + @Test def testNVIScopedVariableWithValueLengthOVCMatchesSinglePass(): Unit = { + val numRecords = 25 + val sch = SchemaUtils.dfdlTestSchema( + , { + + + }, + + + + + + + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val records = + (0 until numRecords).map(i => + {f"$i%02d"} + {s"value$i"} + ) + val infoset = + + {records} + + + // Small window forces build to race many iterations ahead of write, + // each pushing its own runningVar instance, before write catches up. + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes( + sch, + infoset, + extraTunables = Map("unparsePrefetchWindowNodes" -> "2") + ) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Simpler variant of the repeating-NVI test above: a single + // (non-repeating) scope, as a distinct data point. + @Test def testSingleNVIScopedVariableWithValueLengthOVCMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , { + + + }, + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = + + 42 + hello + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Regression guard for NVI-scope setVariable resolution: no forward + // OVC or separator, the minimal shape that exposes the race. + @Test def testNVIScopedSetVariableWithNoForwardReferenceMatchesSinglePass(): Unit = { + val numRecords = 25 + val sch = SchemaUtils.dfdlTestSchema( + , { + + + }, + + + + + + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val records = (0 until numRecords).map(i => {f"$i%02d"}) + val infoset = + + {records} + + + val extVars = + ExternalVariablesLoader.mapToBindings(Map(s"{$example}runningVar" -> "-1").asJava) + + val singlePassDp = Compiler() + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + .withExternalVariables(extVars) + val singlePassBytes = TestUtils.unparseToBytes(singlePassDp, infoset) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .withTunable("unparsePrefetchWindowNodes", "2") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + .withExternalVariables(extVars) + val prefetchBytes = TestUtils.unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A setup failure (inputter never produces StartDocument) must yield + // a failed UnparseResult, not an NPE from a null error-path state. + @Test def testMalformedInfosetInputterGetsCleanErrorNotNPE(): Unit = { + // Needs at least one prefetch-beneficial (value-only, not length-dependent) OVC, or + // DataProcessor.unparse's dispatch (ssrd.hasAnyPrefetchBeneficialOVC) falls back to + // single-pass regardless of the tunable, and unparseViaBuildThenWrite (the method + // under test) would never run. + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + + // hasNext() = false immediately means initialize()'s + // "!delegate.hasNext" check fires straight away, before any actual + // infoset event is produced; exactly the "never starts with + // StartDocument" failure this guards. + val neverStartsInputter = new api.infoset.InfosetInputter { + override def getEventType() = null + override def getLocalName() = null + override def getNamespaceURI() = null + override def getSimpleText( + primType: org.apache.daffodil.runtime1.dpath.NodeInfo.Kind, + runtimeProperties: java.util.Map[String, String] + ) = null + override def isNilled(): java.lang.Boolean = null + override def hasNext() = false + override def next(): Unit = () + override def fini(): Unit = () + } + + val out = new ByteArrayOutputStream() + val res = dp.unparse(neverStartsInputter, out) + + assertTrue("expected a failed UnparseResult, not a successful one", res.isError) + assertTrue( + res.getDiagnostics.get(0).getMessage.contains("does not start with StartDocument") + ) + } + + // A variable-length dfdl:length expression inside a separated + // sequence must also call computeTargetLength from write's dispatch. + @Test def testDelimitedVariableLengthExpressionMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + 5 + hello + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z05|hello", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Same dispatch path as the test above, but for a complex element + // whose expression-based dfdl:length wraps a whole group's content chain. + @Test def testDelimitedComplexVariableLengthExpressionMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + 3 + xyz + + + val (singlePassBytes, prefetchBytes) = TestUtils.getSinglePassAndPrefetchBytes(sch, infoset) + assertEquals("Z03|xyz", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A schema where every OVC is content-length-dependent (never + // resolvable early) must fall back to single-pass automatically. + @Test def testPurelyContentLengthOVCSchemaFallsBackAutomatically(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = xyz + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = TestUtils.unparseToBytes(singlePassDp, infoset) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertFalse( + "schema has only a content-length-dependent OVC; should never report prefetch-beneficial", + prefetchDp.ssrd.hasAnyPrefetchBeneficialOVC + ) + val prefetchBytes = TestUtils.unparseToBytes(prefetchDp, infoset) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A schema mixing a resolvable OVC with an unrelated + // content-length-dependent one must still use the prefetch path. + @Test def testMixedSchemaWithResolvableAndContentLengthOVCStillUsesPrefetchPath(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = + + 005 + xyz + + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = TestUtils.unparseToBytes(singlePassDp, infoset) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertTrue( + "schema has a resolvable-without-writing OVC alongside a content-length one; " + + "must still report prefetch-beneficial", + prefetchDp.ssrd.hasAnyPrefetchBeneficialOVC + ) + val prefetchBytes = TestUtils.unparseToBytes(prefetchDp, infoset) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A schema with no OVC at all: hasAnyPrefetchBeneficialOVC must not + // throw, and simply reports false. + @Test def testSchemaWithNoOVCAtAllReportsNotBeneficial(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertFalse(dp.ssrd.hasAnyPrefetchBeneficialOVC) + } + + // A schema where every OVC is resolvable-without-writing must report + // true: the common case prefetch already handles. + @Test def testSchemaWithOnlyResolvableOVCReportsBeneficial(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertTrue(dp.ssrd.hasAnyPrefetchBeneficialOVC) + } + + // hasAnyPrefetchBeneficialOVC is baked in at compile time; confirms + // the fallback still applies after a save/reload round trip. + @Test def testHasAnyPrefetchBeneficialOVCSurvivesSaveReload(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertFalse(dp.ssrd.hasAnyPrefetchBeneficialOVC) + + val os = new ByteArrayOutputStream() + dp.save(java.nio.channels.Channels.newChannel(os)) + val reloadedDp = Compiler() + .reload(new java.io.ByteArrayInputStream(os.toByteArray)) + .asInstanceOf[DataProcessor] + assertFalse(reloadedDp.ssrd.hasAnyPrefetchBeneficialOVC) + + val infoset = xyz + assertArrayEquals( + TestUtils.unparseToBytes(dp, infoset), + TestUtils.unparseToBytes(reloadedDp, infoset) + ) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBoundedPrefetch.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBoundedPrefetch.scala new file mode 100644 index 0000000000..607f5f2995 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBoundedPrefetch.scala @@ -0,0 +1,132 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream +import java.nio.charset.StandardCharsets + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase + +import org.junit.Assert.* +import org.junit.Test + +/** + * Proves build pauses to let write catch up: with a small prefetchLimit, + * the lead counter stays bounded mid-recursion, not equal to the full count. + */ +class TestBoundedPrefetch { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testBuildLeadStaysBoundedDuringRecursion(): Unit = { + val numItems = 40 + val prefetchLimit = 3L + + val sch = SchemaUtils.dfdlTestSchema( + , + { + + + }, + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + val items = (0 until numItems).map(i => {s"i$i"}) + val infoset = + + {items} + + val expectedBytes = (0 until numItems).map(i => s"i$i").mkString(",") + ",M" + + val dp = TestUtils.compileForUnparse( + sch, + Map("releaseUnneededInfoset" -> "false", "useBuildWritePrefetch" -> "true") + ) + + val buildInputter = TestUtils.newInitializedInputter(infoset, dp) + + val sharedCtx = + UnparseSharedContextTestFixture.build(dp, prefetchLimit)() + + val walkerOut = new ByteArrayOutputStream() + val writeInputter = TestUtils.newInitializedInputter(infoset, dp) + val writeState = UState.createInitialUState(walkerOut, dp, writeInputter, false) + writeState.setSharedContext(sharedCtx) + writeState.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + val rootUnparser = dp.ssrd.unparser.asInstanceOf[ElementUnparserBase] + UnparseSharedContextTestFixture.wireCoroutines( + sharedCtx, + buildInputter.documentElement, + rootUnparser, + writeState + ) + + val buildState = new BuildState(buildInputter, sharedCtx, Nil, false) + + dp.ssrd.builder.get.build(buildState) + + // See class doc above for why this proves interleaving; numItems + 2 + // (row + all items + marker) is what currentLead would equal here if + // resumeWrite never fired mid-recursion. + assertTrue( + s"expected lead close to prefetchLimit=$prefetchLimit after build, but was ${sharedCtx.currentLead} " + + s"(numItems=$numItems); build did not actually pause for write", + sharedCtx.currentLead <= prefetchLimit + 1 + ) + assertTrue( + "expected build to have gotten ahead of write by at least one node", + sharedCtx.currentLead > 0 + ) + + // Drain the rest and confirm the output is byte-for-byte correct + // despite having been produced across many separate resumeWrite calls + // rather than a single one-shot write pass. + val finalSignal = sharedCtx.resumeWrite(BuildFinished) + finalSignal match { + case WriteDone(Some(t)) => throw t + case WriteDone(None) => // continue below + case other => fail(s"unexpected final signal: $other") + } + writeState.evalSuspensions(isFinal = true) + writeState.getDataOutputStream.setFinished(writeState) + + assertEquals(expectedBytes, new String(walkerOut.toByteArray, StandardCharsets.US_ASCII)) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildState.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildState.scala new file mode 100644 index 0000000000..aed0e1dde4 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildState.scala @@ -0,0 +1,100 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils + +import org.junit.Assert.* +import org.junit.Test + +/** + * Validates BuildState in isolation: confirms it surfaces the same + * event sequence a real UStateMain would, for a separated schema. + */ +class TestBuildState { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testBuildStateSurfacesCorrectEventSequence(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + { + + + }, + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + val infoset = + + Alice + 30 + Boston + + + val dp = TestUtils.compileForUnparse( + sch, + Map("releaseUnneededInfoset" -> "false", "useBuildWritePrefetch" -> "true") + ) + + val inputter = TestUtils.newInitializedInputter(infoset, dp) + // initialize() pushes the root TRD as its last step - the same + // setup a real unparse's invariant check relies on. + + val sharedCtx = + UnparseSharedContextTestFixture.build(dp, prefetchLimit = 100)() + val buildState = new BuildState(inputter, sharedCtx, Nil, false) + + // Drives through the ACTUAL Builder recursion, not hand-driven + // advance() calls, since next-element resolution depends on the same + // TRD push/pop dance ElementBuilder.build performs. This schema's + // separator never reaches BuildState at all: the Builder tree skips + // straight past the delimiter-stack wrapper unparser entirely. + dp.ssrd.builder.get.build(buildState) + + assertEquals(5L, sharedCtx.currentLead) // row, name, age, city, marker + + val rootNode = inputter.documentElement.child(0).asComplex + assertEquals(4, rootNode.numChildren) + assertEquals("name", rootNode.child(0).erd.name) + assertEquals("age", rootNode.child(1).erd.name) + assertEquals("city", rootNode.child(2).erd.name) + assertEquals("marker", rootNode.child(3).erd.name) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildWriteArrayChoice.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildWriteArrayChoice.scala new file mode 100644 index 0000000000..3722cb7523 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestBuildWriteArrayChoice.scala @@ -0,0 +1,222 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream +import java.nio.charset.StandardCharsets + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase + +import org.junit.Assert.* +import org.junit.Test + +/** + * Validates write-side dispatch against a single-pass tree, and a + * standalone BuildState run, for both scalar and array/choice content. + */ +class TestBuildWriteArrayChoice { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testWriteWalkerMatchesActualUnparse(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + val infoset = + + Alice + 30 + Boston + + + val dp = TestUtils.compileForUnparse( + sch, + Map("releaseUnneededInfoset" -> "false", "useBuildWritePrefetch" -> "true") + ) + + val (singlePassBytes, walkerBytes) = + TestUtils.getSinglePassAndWriteContentBytes(dp, infoset) + + assertArrayEquals(singlePassBytes, walkerBytes) + } + + @Test def testArrayAndChoiceWriteContentMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + // Three "item" occurrences exercise the array loop; typeB (not typeA) + // exercises actual choice resolution. lengthKind=delimited (not + // explicit) avoids double-firing CaptureStartOfContentLengthUnparser's + // non-idempotent marker, since this tree is reused from a completed unparse. + val infoset = + +
H
+ a + b + c + X +
+ + val dp = TestUtils.compileForUnparse( + sch, + Map("releaseUnneededInfoset" -> "false", "useBuildWritePrefetch" -> "true") + ) + + val (singlePassBytes, walkerBytes) = + TestUtils.getSinglePassAndWriteContentBytes(dp, infoset) + + assertEquals("H,a,b,c,X", new String(singlePassBytes, StandardCharsets.US_ASCII)) + assertArrayEquals(singlePassBytes, walkerBytes) + } + + // Standalone-build regression: drives BuildState directly, then feeds + // its tree into write's writeContent (end-to-end build-then-write). + @Test def testStandaloneBuildStateNavigatesArrayChoiceSeparator(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + { + + + }, + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + val infoset = + +
H
+ a + b + c + X +
+ + val dp = TestUtils.compileForUnparse( + sch, + Map("releaseUnneededInfoset" -> "false", "useBuildWritePrefetch" -> "true") + ) + + // Build phase: standalone BuildState drives the actual Unparser recursion, + // navigating past the sequence's separator and through the + // array/choice content, purely to build the tree. + val buildInputter = TestUtils.newInitializedInputter(infoset, dp) + + val sharedCtx = + UnparseSharedContextTestFixture.build(dp, prefetchLimit = 100)() + val buildState = new BuildState(buildInputter, sharedCtx, Nil, false) + + dp.ssrd.builder.get.build(buildState) + + // row, header, item x3, typeB, marker = 7 elements total. + assertEquals(7L, sharedCtx.currentLead) + + val rootNode = buildInputter.documentElement.child(0).asComplex + assertEquals(4, rootNode.numChildren) + assertEquals("header", rootNode.child(0).erd.name) + assertEquals("item", rootNode.child(1).erd.name) + assertEquals(3, rootNode.child(1).asInstanceOf[DIArray].numChildren) + assertEquals("typeB", rootNode.child(2).erd.name) + assertEquals("marker", rootNode.child(3).erd.name) + + // Build was driven directly (no coroutine handoff), so sharedCtx can't + // know build is done; tell it so write's awaitChild call takes the + // post-BuildFinished (suspension-retry) path instead of resuming a + // coroutine that was never set up. + sharedCtx.observeBuildSignal(BuildFinished) + + // Write phase: write the tree BuildState just constructed, confirming + // it's a usable, fully-built tree, not just a navigation exercise. + val walkerOut = new ByteArrayOutputStream() + val writeInputter = TestUtils.newInitializedInputter(infoset, dp) + val writeState = UState.createInitialUState(walkerOut, dp, writeInputter, false) + writeState.setSharedContext(sharedCtx) + writeState.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + val rootUnparser = dp.ssrd.unparser.asInstanceOf[ElementUnparserBase] + val rootWriteNode = sharedCtx.awaitChild(buildInputter.documentElement, 0) + rootUnparser.writeContent(rootWriteNode, writeState) + // The separator (default separatorSuppressionPolicy "anyEmpty") is + // written speculatively via a suspension deciding, once known, if the + // region it precedes is zero-length; this drains that chain before the + // DOS is finalized, mirroring DataProcessor's finishWriteSide. + writeState.evalSuspensions(isFinal = true) + writeState.getDataOutputStream.setFinished(writeState) + + assertEquals("H,a,b,c,X,M", new String(walkerOut.toByteArray, StandardCharsets.US_ASCII)) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestLeadCounter.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestLeadCounter.scala new file mode 100644 index 0000000000..a26da08f31 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestLeadCounter.scala @@ -0,0 +1,125 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream +import java.nio.charset.StandardCharsets + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase + +import org.junit.Assert.* +import org.junit.Test + +/** + * Validates the shared build/write lead counter end to end: build + * increments and write decrements against the same shared instance. + */ +class TestLeadCounter { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testLeadCounterIncrementsOnBuildAndDecrementsOnWrite(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + { + + + }, + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + // Fixed dfdl:length is safe here because this tree comes from + // BuildState, which never runs content-writing (including + // CaptureStartOfContentLengthUnparser). + val infoset = + + Alice + 30 + Boston + + + val dp = TestUtils.compileForUnparse( + sch, + Map("releaseUnneededInfoset" -> "false", "useBuildWritePrefetch" -> "true") + ) + + // Build phase: drive BuildState through the actual recursion, incrementing + // the shared lead counter via the actual unparseBegin hookup, as in + // BuildStateTest. + val buildInputter = TestUtils.newInitializedInputter(infoset, dp) + + val sharedCtx = + UnparseSharedContextTestFixture.build(dp, prefetchLimit = 100)() + val buildState = new BuildState(buildInputter, sharedCtx, Nil, false) + + assertEquals(0L, sharedCtx.currentLead) + dp.ssrd.builder.get.build(buildState) + + // row itself, name, age, city, marker = 5 elements total, each + // incrementing once via unparseBegin's actual hookup. + assertEquals(5L, sharedCtx.currentLead) + + // Build was driven directly (no coroutine handoff), so sharedCtx can't + // know build is done; tell it so write's awaitChild calls take the + // post-BuildFinished (suspension-retry) path instead of resuming a + // coroutine that was never set up. + sharedCtx.observeBuildSignal(BuildFinished) + + // Write phase: write the SAME already-built tree + // (buildInputter.documentElement) against the SAME sharedCtx, + // decrementing the lead counter as it goes. + val walkerOut = new ByteArrayOutputStream() + val writeInputter = TestUtils.newInitializedInputter(infoset, dp) + val writeState = UState.createInitialUState(walkerOut, dp, writeInputter, false) + writeState.setSharedContext(sharedCtx) + writeState.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + val rootUnparser = dp.ssrd.unparser.asInstanceOf[ElementUnparserBase] + val rootNode = sharedCtx.awaitChild(buildInputter.documentElement, 0) + rootUnparser.writeContent(rootNode, writeState) + writeState.getDataOutputStream.setFinished(writeState) + + assertEquals("Alice30BostonM", new String(walkerOut.toByteArray, StandardCharsets.US_ASCII)) + + // Write decremented once per element too, so the counter is back to 0: + // build and write agree on how many nodes exist, coordinated through + // the shared UnparseSharedContext. + assertEquals(0L, sharedCtx.currentLead) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContextTestFixture.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContextTestFixture.scala new file mode 100644 index 0000000000..29d1a9c899 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContextTestFixture.scala @@ -0,0 +1,70 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.runtime1.infoset.DIDocument +import org.apache.daffodil.runtime1.processors.DataProcessor +import org.apache.daffodil.runtime1.processors.SuspensionTracker +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase + +/** + * Shared BuildState construction for tests, mirroring production's `* 2` + * doubling of tunable-derived suspension-wait thresholds (default only). + */ +object UnparseSharedContextTestFixture { + def build(dp: DataProcessor, prefetchLimit: Long)( + suspensionWaitYoung: Int = dp.tunables.unparseSuspensionWaitYoung * 2, + suspensionWaitOld: Int = dp.tunables.unparseSuspensionWaitOld * 2 + ): UnparseSharedContext = { + new UnparseSharedContext( + new SuspensionTracker(suspensionWaitYoung, suspensionWaitOld), + dp, + dp.tunables, + prefetchLimit + ) + } + + /** Wires a BuildCoroutine/WriteCoroutine pair onto sharedCtx, mirroring + * production's WriteCoroutine, so mid-recursion resumeWrite calls do + * something; caller must still perform the final handoff afterward. + */ + def wireCoroutines( + sharedCtx: UnparseSharedContext, + documentElement: DIDocument, + rootUnparser: ElementUnparserBase, + writeState: UState + ): Unit = { + val buildCoroutine = new BuildCoroutine() + val writeCoroutine = new WriteCoroutine({ (wc, firstSignal) => + try { + sharedCtx.observeBuildSignal(firstSignal) + try { + val rootNode = sharedCtx.awaitChild(documentElement, 0) + rootUnparser.writeContent(rootNode, writeState) + } catch { + case _: AwaitChildStalledException => + } + wc.resumeFinal(buildCoroutine, WriteDone(None)) + } catch { + case t: Throwable => + wc.resumeFinal(buildCoroutine, WriteDone(Some(t))) + } + }) + sharedCtx.setCoroutines(buildCoroutine, writeCoroutine) + } +} diff --git a/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd b/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd index db5568846a..532d1e0c29 100644 --- a/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd +++ b/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd @@ -534,6 +534,50 @@ + + + + If true, unparsing uses a build/write-prefetch path instead of the + default single-pass unparse: a build pass runs ahead of a write pass + over the same infoset tree, resolving forward references against the + tree directly instead of via Suspensions where possible. How far + build may run ahead of write is bounded by + unparsePrefetchWindowNodes. + + + + + + + Only consulted when useBuildWritePrefetch is true. The maximum + number of infoset nodes build may construct ahead of write before + build pauses to let write catch up. Bounds peak memory use to + roughly this many resident nodes rather than the whole infoset. + + + + + + + Only consulted when useBuildWritePrefetch is true. A second, + independent limit on how far build may run ahead of write, + alongside unparsePrefetchWindowNodes: the maximum number of + pending suspensions (forward references that cannot resolve + until write itself writes the referenced bytes, e.g. + dfdl:valueLength/dfdl:contentLength) build may accumulate + before it pauses to let write catch up, regardless of how + large unparsePrefetchWindowNodes is set. Without this, + unparsePrefetchWindowNodes alone does not bound this + backlog, since a node's lead-counter contribution is + released once write's structural pass reaches it, whether or + not that node's own suspension has resolved yet, so a large + window can otherwise let many resolved-but-still- + pending nodes accumulate unboundedly before anything + forces write to actually catch up to what they're + waiting on. + + + @@ -644,7 +688,11 @@ evaluating suspensions that are likely to fail. The unparseSuspensionWaitYoung and unparseSuspensionWaitOld values determine how many elements are unparsed before evaluating - young and old suspensions, respectively. + young and old suspensions, respectively. When useBuildWritePrefetch + is true, both values are used internally at double what is + configured here, since build and write independently advance this + same counter, so it would otherwise tick twice as fast as a single + traversal. diff --git a/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala b/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala index d6632308b8..3e44d57583 100644 --- a/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala +++ b/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala @@ -79,7 +79,8 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm | if (configOpt.isDefined) { | val loader = new DaffodilXMLLoader() | val node = loader.load(URISchemaSource(Paths.get(configPath).toFile, configOpt.get), Some(XMLUtils.dafextURI)) - | tunablesMap(node) + | val optTunablesNode = (node \ "tunables").headOption + | optTunablesNode.map(tunablesMap(_)).getOrElse(Map.empty) | } else { | Map.empty | } @@ -98,15 +99,17 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm | tunables.foldLeft(this) { case (dafTuns, (tunable, value)) => dafTuns.withTunable(tunable, value) } | } | + | // One large match over every tunable name would exceed the JVM's + | // 64KB-per-method bytecode limit as tunables accumulate over time, so + | // this is split into a chain of smaller private methods instead, each + | // falling through to the next on no match. | def withTunable(tunable: String, value: String): DaffodilTunables = { - | tunable match { + | withTunablePart0(tunable, value) + | } + | """.trim.stripMargin val bottom = """ - | case _ => throw new IllegalArgumentException("Unknown tunable: " + tunable) - | } - | } - | | private def throwInvalidTunableValue(tunable: String, value: String) = { | throw new IllegalArgumentException("Invalid value for tunable " + tunable + ": " + value) | } @@ -114,6 +117,11 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm |} """.trim.stripMargin + // How many tunables' worth of case-match bytecode go into each + // withTunablePartN method; small enough to leave headroom against the + // JVM's 64KB-per-method limit as more tunables are added over time. + private val tunablesPerPart = 8 + val tunablesRoot = (schemaRootConfig \ "element").find(_ \@ "name" == "tunables").get val tunableNodes = tunablesRoot \\ "all" \ "element" @@ -165,15 +173,39 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm .map(_.scalaDefinition) .mkString(" ", ",\n ", ")") - val conversionString = - tunables - .map { tunable => - tunable.scalaConversion - .split("\n") - .filter(_.trim.length > 0) - .mkString(" ", "\n ", "") + // Split across several withTunablePartN methods rather than one large + // match (see middle's own comment): each chunk's case bodies, followed + // by a fallthrough to the next chunk's method, or to the final + // "unknown tunable" error on the last chunk. + val tunableChunks = tunables.grouped(tunablesPerPart).toIndexedSeq + val numParts = tunableChunks.length + val partsString = + tunableChunks.zipWithIndex + .map { case (chunk, idx) => + val casesString = + chunk + .map { tunable => + tunable.scalaConversion + .split("\n") + .filter(_.trim.length > 0) + .mkString(" ", "\n ", "") + } + .mkString("\n") + val fallthrough = + if (idx == numParts - 1) + """ case _ => throw new IllegalArgumentException("Unknown tunable: " + tunable)""" + else + s" case _ => withTunablePart${idx + 1}(tunable, value)" + s""" + | private def withTunablePart${idx}(tunable: String, value: String): DaffodilTunables = { + | tunable match { + |${casesString} + |${fallthrough} + | } + | } + """.trim.stripMargin } - .mkString("\n") + .mkString("\n\n") w.write(top) w.write("\n") @@ -181,7 +213,7 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm w.write("\n") w.write(middle) w.write("\n") - w.write(conversionString) + w.write(partsString) w.write("\n") w.write(bottom) w.write("\n")