diff --git a/.github/workflows/main.yml b/.github/workflows/main.yml index 47c904a8fd..7a123afe2f 100644 --- a/.github/workflows/main.yml +++ b/.github/workflows/main.yml @@ -197,6 +197,14 @@ jobs: - name: Run Unit Tests run: $SBT test + # Build ahead is the default, so run the suite once more event driven to + # keep that unparse path tested. + - name: Run Unit Tests (event-driven unparse) + if: matrix.os == 'ubuntu-22.04' && matrix.java_version == '17' + env: + DAFFODIL_TDML_TUNABLES: infosetBuilderMode=eventDriven + run: $SBT test + - name: Run Integration Tests run: $SBT daffodil-test-integration/test diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Grammar.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Grammar.scala index 47f1f3740f..ded35b7704 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Grammar.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Grammar.scala @@ -21,6 +21,8 @@ import org.apache.daffodil.core.compiler.ForParser import org.apache.daffodil.core.compiler.ForUnparser import org.apache.daffodil.core.dsom.* import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.runtime1.infoset.InfosetBuilder +import org.apache.daffodil.runtime1.infoset.SeqCompInfosetBuilder import org.apache.daffodil.runtime1.processors.parsers.AssertExpressionEvaluationParser import org.apache.daffodil.runtime1.processors.parsers.NadaParser import org.apache.daffodil.runtime1.processors.parsers.SeqCompParser @@ -124,6 +126,19 @@ class SeqComp private (context: SchemaComponent, children: Seq[Gram]) else if (unparserChildren.length == 1) unparserChildren.head else new SeqCompUnparser(context.runtimeData, unparserChildren.toArray) } + + // Only the (usually at most one) child(ren) that actually build infoset + // content contribute here; children that build no infoset events + // (delimiters, padding, etc.) simply have no builder to collect. + lazy val builderChildren: Array[InfosetBuilder] = { + children + .filter(x => !x.isEmpty && (x.forWhat != ForParser)) + .map(_.builder) + .filterNot(_.isEmpty) + .toArray + } + + final override lazy val builder: InfosetBuilder = SeqCompInfosetBuilder(builderChildren) } object EmptyGram extends Gram(null) { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Production.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Production.scala index 833b1f71f2..aa108b6099 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Production.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/Production.scala @@ -22,6 +22,8 @@ import org.apache.daffodil.core.compiler.ForUnparser import org.apache.daffodil.core.compiler.ParserOrUnparser import org.apache.daffodil.core.dsom.SchemaComponent import org.apache.daffodil.lib.util.Logger +import org.apache.daffodil.runtime1.infoset.InfosetBuilder +import org.apache.daffodil.runtime1.infoset.NadaInfosetBuilder import org.apache.daffodil.runtime1.processors.parsers.NadaParser import org.apache.daffodil.unparsers.runtime1.NadaUnparser @@ -107,4 +109,16 @@ final class Prod( else unp } + + final override lazy val builder: InfosetBuilder = { + if (gram.isEmpty) { + NadaInfosetBuilder + } else { + (forWhat, gram.forWhat) match { + case (ForParser, _) => NadaInfosetBuilder + case (_, ForParser) => NadaInfosetBuilder + case _ => gram.builder + } + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ChoiceCombinator.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ChoiceCombinator.scala index b91428e4ca..fb79fc89a6 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ChoiceCombinator.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ChoiceCombinator.scala @@ -29,10 +29,17 @@ import org.apache.daffodil.lib.cookers.ChoiceBranchKeyCooker import org.apache.daffodil.lib.cookers.IntRangeCooker import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.schema.annotation.props.gen.ChoiceLengthKind +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One import org.apache.daffodil.lib.util.MaybeInt import org.apache.daffodil.lib.util.ProperlySerializableMap.* -import org.apache.daffodil.runtime1.infoset.ChoiceBranchEvent +import org.apache.daffodil.lib.xml.NamedQName +import org.apache.daffodil.runtime1.infoset.ChoiceInfosetBuilder +import org.apache.daffodil.runtime1.infoset.InfosetBuilder +import org.apache.daffodil.runtime1.infoset.NadaInfosetBuilder import org.apache.daffodil.runtime1.processors.RangeBound +import org.apache.daffodil.runtime1.processors.TermRuntimeData import org.apache.daffodil.runtime1.processors.parsers.* import org.apache.daffodil.runtime1.processors.unparsers.* import org.apache.daffodil.unparsers.runtime1.* @@ -266,8 +273,15 @@ case class ChoiceCombinator(ch: ChoiceTermBase, alternatives: Seq[Gram]) } } + private lazy val eventUnparserMap = ch.choiceBranchMap._1.map { case (qname, branchTerm) => + (qname, branchTerm.termContentBody.unparser) + } + + private lazy val hasEventBranchUnparser: Boolean = + eventUnparserMap.exists { case (_, branchUnparser) => !branchUnparser.isEmpty } + override lazy val unparser: Unparser = { - val (eventRDMap, optDefaultBranch) = ch.choiceBranchMap + val optDefaultBranch = ch.choiceBranchMap._2 /* * Since it's impossible to know the hiddenness for terms at this level (unless * they're a hiddenGroupRef), we always attempt to find a defaultable unparser. @@ -313,11 +327,7 @@ case class ChoiceCombinator(ch: ChoiceTermBase, alternatives: Seq[Gram]) optDefaultUnparser } - val eventUnparserMap = eventRDMap.map { case (cbe, branchTerm) => - (cbe, branchTerm.termContentBody.unparser) - } - val mapValues = eventUnparserMap.map { case (k, v) => v }.filterNot(_.isEmpty) - if (mapValues.isEmpty) { + if (!hasEventBranchUnparser) { if (branchForUnparse.isEmpty) { new NadaUnparser(null) } else { @@ -326,10 +336,28 @@ case class ChoiceCombinator(ch: ChoiceTermBase, alternatives: Seq[Gram]) branchForUnparse.get } } else { - val serializableMap: ProperlySerializableMap[ChoiceBranchEvent, Unparser] = + val serializableMap: ProperlySerializableMap[NamedQName, Unparser] = eventUnparserMap.toProperlySerializableMap val cbm = ChoiceBranchMap(serializableMap, branchForUnparse) new ChoiceCombinatorUnparser(ch.modelGroupRuntimeData, cbm, choiceLengthInBits) } } + + override lazy val builder: InfosetBuilder = { + val (eventRDMap, optDefaultBranch) = ch.choiceBranchMap + + if (!hasEventBranchUnparser && optDefaultBranch.isEmpty) { + NadaInfosetBuilder + } else { + val branchMap: Map[NamedQName, (TermRuntimeData, InfosetBuilder)] = + eventRDMap.map { case (qname, branchTerm) => + (qname, (branchTerm.termRuntimeData, branchTerm.termContentBody.builder)) + } + val defaultBranch: Maybe[(TermRuntimeData, InfosetBuilder)] = optDefaultBranch match { + case Some(term) => One((term.termRuntimeData, term.termContentBody.builder)) + case None => Nope + } + new ChoiceInfosetBuilder(ch.modelGroupRuntimeData, branchMap, defaultBranch) + } + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/DelimiterAndEscapeRelated.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/DelimiterAndEscapeRelated.scala index 6bf946a91d..a17900ab78 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/DelimiterAndEscapeRelated.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/DelimiterAndEscapeRelated.scala @@ -21,9 +21,11 @@ import org.apache.daffodil.core.dsom.* import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.core.grammar.Terminal import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.lib.util.Misc import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.infoset.InfosetBuilder import org.apache.daffodil.runtime1.processors.parsers.{ Parser as DaffodilParser, * } import org.apache.daffodil.runtime1.processors.unparsers.Unparser as DaffodilUnparser import org.apache.daffodil.unparsers.runtime1.* @@ -55,6 +57,10 @@ case class DelimiterStackCombinatorSequence(sq: SequenceTermBase, body: Gram) override lazy val unparser: DaffodilUnparser = new DelimiterStackUnparser(uInit, uSep, uTerm, sq.termRuntimeData, body.unparser) + + // Delimiters build no infoset events; the builder tree skips straight to + // whatever this sequence's body itself builds, if anything. + override lazy val builder: InfosetBuilder = body.builder } case class DelimiterStackCombinatorChoice(ch: ChoiceTermBase, body: Gram) @@ -77,6 +83,10 @@ case class DelimiterStackCombinatorChoice(ch: ChoiceTermBase, body: Gram) override lazy val unparser: DaffodilUnparser = new DelimiterStackUnparser(uInit, None, uTerm, ch.termRuntimeData, body.unparser) + + // Delimiters build no infoset events; the builder tree skips straight to + // whatever this choice's body itself builds, if anything. + override lazy val builder: InfosetBuilder = body.builder } case class DelimiterStackCombinatorElement(e: ElementBase, body: Gram) @@ -111,6 +121,10 @@ case class DelimiterStackCombinatorElement(e: ElementBase, body: Gram) if (u.isEmpty) u else new DelimiterStackUnparser(uInit, None, uTerm, e.termRuntimeData, u) } + + // Delimiters build no infoset events; the builder tree skips straight to + // whatever this element's body itself builds, if anything. + override lazy val builder: InfosetBuilder = body.builder } case class DynamicEscapeSchemeCombinatorElement(e: ElementBase, body: Gram) @@ -135,4 +149,9 @@ case class DynamicEscapeSchemeCombinatorElement(e: ElementBase, body: Gram) if (u.isEmpty || schemeUnparseIsConstant) u else new DynamicEscapeSchemeUnparser(schemeUnparseOpt.get, e.termRuntimeData, u) } + + // The escape scheme only governs delimiter matching in written bytes; the + // builder tree skips straight to whatever this element's body itself + // builds, if anything. + override lazy val builder: InfosetBuilder = body.builder } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ElementCombinator.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ElementCombinator.scala index 0dcc7ad5cd..841d5fbb13 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ElementCombinator.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/ElementCombinator.scala @@ -28,6 +28,9 @@ import org.apache.daffodil.lib.schema.annotation.props.gen.LengthKind import org.apache.daffodil.lib.schema.annotation.props.gen.Representation import org.apache.daffodil.lib.schema.annotation.props.gen.TestKind import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.runtime1.infoset.ElementInfosetBuilder +import org.apache.daffodil.runtime1.infoset.InfosetBuilder +import org.apache.daffodil.runtime1.infoset.NadaInfosetBuilder import org.apache.daffodil.runtime1.processors.parsers.CaptureEndOfContentLengthParser import org.apache.daffodil.runtime1.processors.parsers.CaptureEndOfValueLengthParser import org.apache.daffodil.runtime1.processors.parsers.CaptureStartOfContentLengthParser @@ -44,6 +47,7 @@ import org.apache.daffodil.unparsers.runtime1.CaptureStartOfValueLengthUnparser import org.apache.daffodil.unparsers.runtime1.ElementOVCSpecifiedLengthUnparser import org.apache.daffodil.unparsers.runtime1.ElementOVCUnspecifiedLengthUnparser import org.apache.daffodil.unparsers.runtime1.ElementSpecifiedLengthUnparser +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase import org.apache.daffodil.unparsers.runtime1.ElementUnparserInputValueCalc import org.apache.daffodil.unparsers.runtime1.ElementUnspecifiedLengthUnparser import org.apache.daffodil.unparsers.runtime1.ElementUnusedUnparser @@ -109,7 +113,13 @@ class ElementCombinator( if (eAfterValue.isEmpty) Maybe.Nope else Maybe(eAfterValue.unparser) - private lazy val eReptypeUnparser: Maybe[Unparser] = repTypeElementGram.maybeUnparser + private lazy val eRepTypeUnparser: Maybe[Unparser] = repTypeElementGram.maybeUnparser + + private lazy val isSpecifiedLength: Boolean = + (context.lengthKind._eq_(LengthKind.Explicit)) || + (context.isSimpleType && + (context.lengthKind._eq_(LengthKind.Implicit)) && + (context.impliedRepresentation._eq_(Representation.Text))) override lazy val unparser: Unparser = { if (context.isOutputValueCalc) { @@ -122,12 +132,7 @@ class ElementCombinator( eAfterUnparser, context.ovcCompiledExpression ) - } else if ( - (context.lengthKind._eq_(LengthKind.Explicit)) || - (context.isSimpleType && - (context.lengthKind._eq_(LengthKind.Implicit)) && - (context.impliedRepresentation._eq_(Representation.Text))) - ) { + } else if (isSpecifiedLength) { new ElementSpecifiedLengthUnparser( context.erd, @@ -136,13 +141,33 @@ class ElementCombinator( eBeforeUnparser, eUnparser, eAfterUnparser, - eReptypeUnparser + eRepTypeUnparser ) } else { subComb.unparser } } + private lazy val eBuilder: InfosetBuilder = { + if (eValue.isEmpty) { + NadaInfosetBuilder + } else { + eValue.builder + } + } + private lazy val eRepTypeBuilder: InfosetBuilder = repTypeElementGram.builder + + // Shares the memoized unparser above for unparseBegin/unparseEnd, so + // build and the unparse see identical node-creation behavior. + override lazy val builder: InfosetBuilder = { + if (context.isOutputValueCalc || isSpecifiedLength) { + val eu = unparser.asInstanceOf[ElementUnparserBase] + val contentBuilder = eRepTypeBuilder.orElse(eBuilder) + new ElementInfosetBuilder(context.erd, eu, contentBuilder) + } else { + subComb.builder + } + } } case class ElementUnused(ctxt: ElementBase) @@ -374,6 +399,14 @@ class ElementParseAndUnspecifiedLength( new ElementUnparserInputValueCalc(context.erd, uSetVar) } } + + // Shares the memoized unparser above for unparseBegin/unparseEnd, so + // build and the unparse see identical nilled/OVC/IVC node-creation behavior. + override lazy val builder: InfosetBuilder = { + val eu = unparser.asInstanceOf[ElementUnparserBase] + val contentBuilder = eRepTypeBuilder.orElse(eBuilder) + new ElementInfosetBuilder(context.erd, eu, contentBuilder) + } } abstract class ElementCombinatorBase( @@ -449,4 +482,8 @@ abstract class ElementCombinatorBase( def unparser: Unparser + lazy val eBuilder: InfosetBuilder = eGram.builder + + lazy val eRepTypeBuilder: InfosetBuilder = repTypeElementGram.builder + } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/HiddenGroupCombinator.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/HiddenGroupCombinator.scala index a993c98cfe..06b31c841b 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/HiddenGroupCombinator.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/HiddenGroupCombinator.scala @@ -20,6 +20,8 @@ package org.apache.daffodil.core.grammar.primitives import org.apache.daffodil.core.dsom.ModelGroup import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.core.grammar.Terminal +import org.apache.daffodil.runtime1.infoset.HiddenGroupInfosetBuilder +import org.apache.daffodil.runtime1.infoset.InfosetBuilder import org.apache.daffodil.runtime1.processors.parsers.HiddenGroupCombinatorParser import org.apache.daffodil.runtime1.processors.parsers.Parser import org.apache.daffodil.runtime1.processors.unparsers.Unparser @@ -34,4 +36,5 @@ final class HiddenGroupCombinator(ctxt: ModelGroup, body: Gram) override lazy val unparser: Unparser = new HiddenGroupCombinatorUnparser(ctxt.modelGroupRuntimeData, body.unparser) + override lazy val builder: InfosetBuilder = HiddenGroupInfosetBuilder(body.builder) } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/LayeredSequence.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/LayeredSequence.scala index 2b75da3f4b..3157ce71f2 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/LayeredSequence.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/LayeredSequence.scala @@ -21,6 +21,8 @@ import org.apache.daffodil.core.dsom.* import org.apache.daffodil.core.grammar.Terminal import org.apache.daffodil.core.layers.LayerSchemaCompiler import org.apache.daffodil.lib.util.Misc +import org.apache.daffodil.runtime1.infoset.InfosetBuilder +import org.apache.daffodil.runtime1.infoset.SequenceInfosetBuilder import org.apache.daffodil.runtime1.processors.parsers.LayeredSequenceParser import org.apache.daffodil.runtime1.processors.parsers.Parser as DaffodilParser import org.apache.daffodil.runtime1.processors.unparsers.Unparser as DaffodilUnparser @@ -47,4 +49,11 @@ case class LayeredSequence(sq: SequenceGroupTermBase, bodyTerm: SequenceChild) override lazy val unparser: DaffodilUnparser = new LayeredSequenceUnparser(srd, bodyUnparser) + + // The layer transform builds no infoset events, but this is still a one-child + // sequence position: it must push/pop bodyTerm's TRD and advance the + // group index like any sequence child, or next-element resolution on + // the shared InfosetInputter breaks. + override lazy val builder: InfosetBuilder = + SequenceInfosetBuilder(Array(bodyTerm.sequenceChildBuildInfo)) } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/NilEmptyCombinators.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/NilEmptyCombinators.scala index 1ed7a027bc..3ea97248bc 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/NilEmptyCombinators.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/NilEmptyCombinators.scala @@ -21,6 +21,8 @@ import org.apache.daffodil.core.dsom.ElementBase import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.core.grammar.Terminal import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.runtime1.infoset.InfosetBuilder +import org.apache.daffodil.runtime1.infoset.NilOrContentInfosetBuilder import org.apache.daffodil.runtime1.processors.parsers.ComplexNilOrContentParser import org.apache.daffodil.runtime1.processors.parsers.SimpleNilOrValueParser import org.apache.daffodil.unparsers.runtime1.ComplexNilOrContentUnparser @@ -59,4 +61,8 @@ case class ComplexNilOrContent(ctxt: ElementBase, nilGram: Gram, contentGram: Gr override lazy val unparser = ComplexNilOrContentUnparser(ctxt.erd, nilUnparser, contentUnparser) + // A nilled complex element has no children to build; nilled-ness is only + // known once the node exists, so this needs a real runtime check, not a + // static pass-through. + override lazy val builder: InfosetBuilder = NilOrContentInfosetBuilder(contentGram.builder) } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceChild.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceChild.scala index 725eff5363..418b7eb373 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceChild.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceChild.scala @@ -25,6 +25,8 @@ import org.apache.daffodil.lib.schema.annotation.props.gen.LengthKind import org.apache.daffodil.lib.schema.annotation.props.gen.OccursCountKind import org.apache.daffodil.lib.schema.annotation.props.gen.Representation import org.apache.daffodil.runtime1.dpath.NodeInfo +import org.apache.daffodil.runtime1.infoset.InfosetBuilder +import org.apache.daffodil.runtime1.infoset.SequenceChildInfosetBuildInfo import org.apache.daffodil.runtime1.processors.parsers.* import org.apache.daffodil.unparsers.runtime1.* @@ -69,6 +71,7 @@ abstract class SequenceChild(protected val sq: SequenceTermBase, child: Term, gr protected lazy val childParser = child.termContentBody.parser protected lazy val childUnparser = child.termContentBody.unparser + protected lazy val childBuilder: InfosetBuilder = child.termContentBody.builder final override lazy val parser = sequenceChildParser final override lazy val unparser = sequenceChildUnparser @@ -82,6 +85,9 @@ abstract class SequenceChild(protected val sq: SequenceTermBase, child: Term, gr final lazy val optSequenceChildUnparser: Option[SequenceChildUnparser] = if (childUnparser.isEmpty) None else Some(unparser) + final lazy val sequenceChildBuildInfo: SequenceChildInfosetBuildInfo = + SequenceChildInfosetBuildInfo(unparser, childBuilder) + /** * There's only parse result helpers here, so let's abbreviate */ diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceCombinator.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceCombinator.scala index aab310aa2c..6dc8c8e269 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceCombinator.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SequenceCombinator.scala @@ -24,6 +24,9 @@ import org.apache.daffodil.lib.schema.annotation.props.SeparatorSuppressionPolic import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.MaybeInt import org.apache.daffodil.lib.util.Misc +import org.apache.daffodil.runtime1.infoset.InfosetBuilder +import org.apache.daffodil.runtime1.infoset.SequenceChildInfosetBuildInfo +import org.apache.daffodil.runtime1.infoset.SequenceInfosetBuilder import org.apache.daffodil.runtime1.processors.parsers.* import org.apache.daffodil.runtime1.processors.unparsers.* import org.apache.daffodil.unparsers.runtime1.{ Separated as SeparatedUnparser, * } @@ -124,6 +127,9 @@ class OrderedSequence(sq: SequenceTermBase, sequenceChildrenArg: Seq[SequenceChi } } } + + override lazy val builder: InfosetBuilder = + SequenceInfosetBuilder(sequenceChildren.map { _.sequenceChildBuildInfo }) } class UnorderedSequence( @@ -220,4 +226,7 @@ class UnorderedSequence( } } } + + override lazy val builder: InfosetBuilder = + SequenceInfosetBuilder(sequenceChildren.map { _.sequenceChildBuildInfo }) } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SpecifiedLength.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SpecifiedLength.scala index 0ae3e6a4d9..c5b758b0aa 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SpecifiedLength.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/grammar/primitives/SpecifiedLength.scala @@ -25,6 +25,7 @@ import org.apache.daffodil.core.grammar.Terminal import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.schema.annotation.props.gen.LengthUnits import org.apache.daffodil.runtime1.dpath.NodeInfo.PrimType +import org.apache.daffodil.runtime1.infoset.InfosetBuilder import org.apache.daffodil.runtime1.processors.parsers.* import org.apache.daffodil.runtime1.processors.unparsers.* import org.apache.daffodil.unparsers.runtime1.* @@ -44,6 +45,11 @@ abstract class SpecifiedLengthCombinatorBase(val e: ElementBase, eGramArg: => Gr u } + // None of the length-kind wrapping below (explicit/implicit/prefixed + // lengths, pattern matching) creates infoset nodes; the builder tree skips + // straight to whatever the wrapped element content itself builds. + override lazy val builder: InfosetBuilder = eGram.builder + def kind: String def toBriefXML(depthLimit: Int = -1): String = { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ChoiceTermRuntime1Mixin.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ChoiceTermRuntime1Mixin.scala index 9aa556a86b..8309b2fad9 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ChoiceTermRuntime1Mixin.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ChoiceTermRuntime1Mixin.scala @@ -25,10 +25,8 @@ import org.apache.daffodil.core.dsom.Term import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.iapi.WarnID import org.apache.daffodil.lib.util.Delay +import org.apache.daffodil.lib.xml.NamedQName import org.apache.daffodil.runtime1.dpath.NodeInfo -import org.apache.daffodil.runtime1.infoset.ChoiceBranchEndEvent -import org.apache.daffodil.runtime1.infoset.ChoiceBranchEvent -import org.apache.daffodil.runtime1.infoset.ChoiceBranchStartEvent import org.apache.daffodil.runtime1.processors.ChoiceDispatchKeyEv import org.apache.daffodil.runtime1.processors.ChoiceRuntimeData @@ -58,7 +56,7 @@ trait ChoiceTermRuntime1Mixin { self: ChoiceTermBase => private lazy val allBranchesClosed = identifyingEventsForAllChoiceBranches.forall { _.isClosed } - final lazy val choiceBranchMap: (Map[ChoiceBranchEvent, Term], Option[Term]) = { + final lazy val choiceBranchMap: (Map[NamedQName, Term], Option[Term]) = { import PossibleNextElements.* @@ -67,7 +65,7 @@ trait ChoiceTermRuntime1Mixin { self: ChoiceTermBase => case t: Term => { val poss = t.identifyingEventsForChoiceBranch poss.pnes.flatMap { case PNE(e, ovr) => - Seq((ChoiceBranchStartEvent(e.namedQName).asInstanceOf[ChoiceBranchEvent], t)) + Seq((e.namedQName, t)) } } case _ => Assert.invariantFailed("must be a term") @@ -103,7 +101,7 @@ trait ChoiceTermRuntime1Mixin { self: ChoiceTermBase => // more than one branch. val noDupes = eventMap.map { - case (event, terms) => { + case (qname, terms) => { Assert.invariant(terms.length > 0) if (terms.length > 1) { if ( @@ -131,7 +129,7 @@ trait ChoiceTermRuntime1Mixin { self: ChoiceTermBase => "Note that elements with dfdl:outputValueCalc cannot be used to distinguish choice branches.\n" + "Note that choice branches with entirely optional content are not allowed.\n" + "The offending choice branches are:\n%s", - event.qname, + qname, terms .map { trd => "%s at %s".format(trd.diagnosticDebugName, trd.locationDescription) @@ -139,20 +137,15 @@ trait ChoiceTermRuntime1Mixin { self: ChoiceTermBase => .mkString("\n") ) } else { - val eventType = event match { - case _: ChoiceBranchEndEvent => "end" - case _: ChoiceBranchStartEvent => "start" - } // there are no element children in any of the branches. SDW( WarnID.MultipleChoiceBranches, - "Multiple choice branches are associated with the %s of element %s.\n" + + "Multiple choice branches are associated with the start of element %s.\n" + "Note that elements with dfdl:outputValueCalc cannot be used to distinguish choice branches.\n" + "Note that choice branches with entirely optional content are not allowed.\n" + "The offending choice branches are:\n%s\n" + "The first branch will be used during unparsing when an infoset ambiguity exists.", - eventType, - event.qname, + qname, terms .map { trd => "%s at %s".format(trd.diagnosticDebugName, trd.locationDescription) @@ -161,7 +154,7 @@ trait ChoiceTermRuntime1Mixin { self: ChoiceTermBase => ) } } - (event, terms(0)) + (qname, terms(0)) } } (noDupes, optDefaultBranch) diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/GramRuntime1Mixin.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/GramRuntime1Mixin.scala index 54b45c8148..b7e52b1ad1 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/GramRuntime1Mixin.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/GramRuntime1Mixin.scala @@ -20,6 +20,8 @@ package org.apache.daffodil.core.runtime1 import org.apache.daffodil.core.grammar.Gram import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.runtime1.infoset.InfosetBuilder +import org.apache.daffodil.runtime1.infoset.NadaInfosetBuilder import org.apache.daffodil.runtime1.processors.parsers.Parser import org.apache.daffodil.runtime1.processors.unparsers.Unparser @@ -59,4 +61,11 @@ trait GramRuntime1Mixin { self: Gram => else Maybe(u) } } + + /** + * Provides this Gram's node in the dedicated InfosetBuilder tree that parallels + * the Unparser tree. Most Grams create or select no infoset content and + * inherit this Nada default; only those that do override it. + */ + def builder: InfosetBuilder = NadaInfosetBuilder } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala index 3e9c5634fe..989629ee51 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala @@ -22,6 +22,8 @@ import org.apache.daffodil.core.dsom.SequenceTermBase import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Logger import org.apache.daffodil.runtime1.iapi.DFDL +import org.apache.daffodil.runtime1.infoset.InfosetBuilder +import org.apache.daffodil.runtime1.infoset.NadaInfosetBuilder import org.apache.daffodil.runtime1.layers.LayerRuntimeCompiler import org.apache.daffodil.runtime1.layers.LayerRuntimeData import org.apache.daffodil.runtime1.processors.DataProcessor @@ -61,6 +63,16 @@ trait SchemaSetRuntime1Mixin { unp }.value + // Built for every unparse-capable compile, whatever infosetBuilderMode says, + // so the tunable can be changed on a compiled DataProcessor. + lazy val builder: InfosetBuilder = LV(Symbol("builder")) { + if (generateUnparser) { + root.document.builder + } else { + NadaInfosetBuilder + } + }.value + private lazy val layerRuntimeCompiler = new LayerRuntimeCompiler private lazy val allLayers: Seq[LayerRuntimeData] = LV(Symbol("allLayers")) { @@ -88,6 +100,7 @@ trait SchemaSetRuntime1Mixin { new SchemaSetRuntimeData( parser, unparser, + builder, root.elementRuntimeData, variableMap, allLayers, diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala index bc70863fe3..5b885956d8 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala @@ -118,9 +118,11 @@ object TestUtils { testSchema: scala.xml.Elem, infosetXML: Node, unparseTo: String, - areTracing: Boolean = false + areTracing: Boolean = false, + tunables: Map[String, String] = Map.empty ): java.util.List[api.Diagnostic] = { - val compiler = Compiler().withTunable("allowExternalPathExpressions", "true") + val compiler = + Compiler().withTunable("allowExternalPathExpressions", "true").withTunables(tunables) val pf = compiler.compileNode(testSchema) if (pf.isError) throwDiagnostics(pf.getDiagnostics) var u = saveAndReload(pf.onPath("/").asInstanceOf[DataProcessor]) diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/lib/TypedEquality.scala b/daffodil-core/src/main/scala/org/apache/daffodil/lib/TypedEquality.scala index bea3883631..343b35ea1a 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/lib/TypedEquality.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/lib/TypedEquality.scala @@ -43,9 +43,12 @@ package object equality { // Convertible types - strongly typed equality + // The operators below are inline so each call site compares with its + // operands' static types; compiled once, they would compare through the + // erased type and call BoxesRunTime.equals every time. implicit class ViewEqual[T](val left: T) extends AnyVal { - @inline def =#=(right: T) = left == right - @inline def !=#=(right: T) = left != right + inline def =#=(right: T) = left == right + inline def !=#=(right: T) = left != right } // implicit class ViewEqual[L](val left: L) extends AnyVal { // def =#=[R](right: R)(implicit equality: ViewEquality[L, R]): Boolean = @@ -87,15 +90,17 @@ package object equality { // Type wise - allows bi-directional subtypes, not just subtype on right. + // The implicit TypeEquality is only the compile-time proof that L and R are + // in a subtype relationship; the comparison itself is expanded in place. implicit class TypeEqual[L <: AnyRef](val left: L) extends AnyVal { - @inline def =:=[R <: AnyRef](right: R)(implicit equality: TypeEquality[L, R]): Boolean = - equality.areEqual(left, right) - @inline def !=:=[R <: AnyRef](right: R)(implicit equality: TypeEquality[L, R]): Boolean = - !equality.areEqual(left, right) - @inline def _eq_[R <: AnyRef](right: R)(implicit equality: TypeEquality[L, R]): Boolean = - equality.areEq(left, right) - @inline def _ne_[R <: AnyRef](right: R)(implicit equality: TypeEquality[L, R]): Boolean = - !equality.areEq(left, right) + inline def =:=[R <: AnyRef](right: R)(implicit equality: TypeEquality[L, R]): Boolean = + left == right + inline def !=:=[R <: AnyRef](right: R)(implicit equality: TypeEquality[L, R]): Boolean = + left != right + inline def _eq_[R <: AnyRef](right: R)(implicit equality: TypeEquality[L, R]): Boolean = + left eq right + inline def _ne_[R <: AnyRef](right: R)(implicit equality: TypeEquality[L, R]): Boolean = + left ne right } @implicitNotFound("Typed equality requires ${L} and ${R} to be in a subtype relationship!") diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/debugger/DaffodilDebugger.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/debugger/DaffodilDebugger.scala index c5d53cdfc7..d1bc022582 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/debugger/DaffodilDebugger.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/debugger/DaffodilDebugger.scala @@ -49,6 +49,7 @@ import org.apache.daffodil.runtime1.processors.* import org.apache.daffodil.runtime1.processors.parsers.* import org.apache.daffodil.runtime1.processors.unparsers.UState import org.apache.daffodil.runtime1.processors.unparsers.UStateForSuspension +import org.apache.daffodil.runtime1.processors.unparsers.UStateMainForBuildAhead import org.apache.daffodil.runtime1.processors.unparsers.Unparser case class DebuggerExitException() extends UnsuppressableException("Debugger exit") @@ -468,22 +469,29 @@ class DaffodilDebugger( } } - private def infosetToString(ie: InfosetElement): String = { + private def infosetToString(ie: InfosetElement, state: ParseOrUnparseState): String = { val bos = new java.io.ByteArrayOutputStream() val xml = new XMLTextInfosetOutputter(bos, pretty = true, minimal = true) + // When unparsing builds the infoset ahead, show only the part the unparse + // has reached, as an event-driven unparse would have built it by now. + val reached = state match { + case buildAhead: UStateMainForBuildAhead => buildAhead.reachedChildCounts() + case _ => null + } val iw = StreamingInfosetWalker( ie.asInstanceOf[DIElement], xml, walkHidden = !DebuggerConfig.removeHidden, ignoreBlocks = true, - releaseUnneededInfoset = false + releaseUnneededInfoset = false, + visibleChildCounts = reached ) iw.walk(lastWalk = true) bos.toString("UTF-8") } - private def debugPrettyPrintXML(ie: InfosetElement): Unit = { - val infosetString = infosetToString(ie) + private def debugPrettyPrintXML(ie: InfosetElement, state: ParseOrUnparseState): Unit = { + val infosetString = infosetToString(ie, state) debugPrintln(infosetString) } @@ -1123,11 +1131,11 @@ class DaffodilDebugger( debugPrintln(_) } res match { - case ie: InfosetElement => debugPrettyPrintXML(ie) + case ie: InfosetElement => debugPrettyPrintXML(ie, state) case nodeSeq: Seq[Any] => nodeSeq.foreach { a => a match { - case ie: InfosetElement => debugPrettyPrintXML(ie) + case ie: InfosetElement => debugPrettyPrintXML(ie, state) case _ => debugPrintln(a) } } @@ -1164,7 +1172,7 @@ class DaffodilDebugger( // // Displays the empty element since it has no value. // - debugPrettyPrintXML(nd.diElement) + debugPrettyPrintXML(nd.diElement, state) state.suppressDiagnosticAndSucceed(r) } case _ => throw r @@ -1810,7 +1818,7 @@ class DaffodilDebugger( debugPrintln("No Infoset", " ") } case _ => { - val infosetString = infosetToString(node) + val infosetString = infosetToString(node, state) val lines = infosetString.split("\r?\n") val dropCount = diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/DFDLFunctions2.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/DFDLFunctions2.scala index fa3c0153ff..1a42dbe201 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/DFDLFunctions2.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/DFDLFunctions2.scala @@ -35,7 +35,7 @@ sealed abstract class DFDLLengthFunctionBase(kind: String, recipes: List[Compile protected def getLength(elt: DIElement, units: LengthUnits, dstate: DState): ULong = { val len: ULong = - DState.withRetryIfBlocking(dstate) { + dstate.withRetryIfBlocking { units match { case LengthUnits.Bits => lengthState(elt).lengthInBits case LengthUnits.Bytes => lengthState(elt).lengthInBytes diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/DState.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/DState.scala index 183eeb21dc..626c6f6fe9 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/DState.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/DState.scala @@ -31,7 +31,6 @@ import org.apache.daffodil.runtime1.infoset.DISimple import org.apache.daffodil.runtime1.infoset.DataValue import org.apache.daffodil.runtime1.infoset.FakeDINode import org.apache.daffodil.runtime1.infoset.InfosetNoNextSiblingException -import org.apache.daffodil.runtime1.infoset.RetryableException import org.apache.daffodil.runtime1.processors.ParseOrUnparseState import org.apache.daffodil.runtime1.processors.SchemaSetRuntimeData; object EqualityNoWarn2 { EqualitySuppressUnusedImportWarning() } @@ -364,43 +363,12 @@ case class DState( // do nothing } - // @inline // TODO: Performance maybe this won't allocate a closure if this is inline? If not replace with macro - final def withRetryIfBlocking[T](body: => T): T = - DState.withRetryIfBlocking(this)(body) -} - -object DState { - - // private object ToBeIgnored - - // @inline - final def withRetryIfBlocking[T](ds: DState)(body: => T): T = { // TODO: Performance maybe this won't allocate a closure if this is inline? If not replace with macro - ds.mode match { - case _: ParserMode => body - case UnparserNonBlocking => body - case UnparserBlocking => { - var isDone = false - var res: T = null.asInstanceOf[T] - while (!isDone) { - try { - res = body - isDone = true - } catch { - case e: RetryableException => { - // we're to block here, and retry subsequently. - isDone = false - // if (ds.thisExpressionCoroutine.isDefined && ds.coroutineToResumeIfBlocked.isDefined) { - // ds.thisExpressionCoroutine.get.resume(ds.coroutineToResumeIfBlocked.get, ToBeIgnored) - // } else { - throw e - // } - } - } - } - res - } - } - } + /** + * Evaluates body in place. A blocked evaluation's RetryableException + * propagates to the suspension machinery, which retries it later. It is + * inline so the by-name body costs no closure per call. + */ + inline final def withRetryIfBlocking[T](inline body: => T): T = body } class DStateForConstantFolding( diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/ChoiceBranchEvent.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/ChoiceBranchEvent.scala deleted file mode 100644 index c19e9c1cad..0000000000 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/ChoiceBranchEvent.scala +++ /dev/null @@ -1,74 +0,0 @@ -/* - * Licensed to the Apache Software Foundation (ASF) under one or more - * contributor license agreements. See the NOTICE file distributed with - * this work for additional information regarding copyright ownership. - * The ASF licenses this file to You under the Apache License, Version 2.0 - * (the "License"); you may not use this file except in compliance with - * the License. You may obtain a copy of the License at - * - * http://www.apache.org/licenses/LICENSE-2.0 - * - * Unless required by applicable law or agreed to in writing, software - * distributed under the License is distributed on an "AS IS" BASIS, - * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. - * See the License for the specific language governing permissions and - * limitations under the License. - */ - -package org.apache.daffodil.runtime1.infoset - -import org.apache.daffodil.lib.exceptions.Assert -import org.apache.daffodil.lib.util.Misc -import org.apache.daffodil.lib.util.UniquenessCache -import org.apache.daffodil.lib.xml.NamedQName - -object ChoiceBranchEvent - extends UniquenessCache[NamedQName, (ChoiceBranchStartEvent, ChoiceBranchEndEvent)] { - - override def apply(nqn: NamedQName) = { - Assert.usage(nqn != null) - super.apply(nqn) - } - - protected def valueFromKey(nqn: NamedQName) = { - Assert.usage(nqn != null) - (new ChoiceBranchStartEvent(nqn), new ChoiceBranchEndEvent(nqn)) - } - - protected def keyFromValue(pair: (ChoiceBranchStartEvent, ChoiceBranchEndEvent)) = { - Assert.usage(pair != null) - Some(pair._1.qname) - } -} - -sealed trait ChoiceBranchEvent extends Serializable { - val qname: NamedQName - - override def toString = Misc.getNameFromClass(this) + "(" + qname + ")" - - override def hashCode = qname.hashCode - -} - -class ChoiceBranchStartEvent(val qname: NamedQName) extends ChoiceBranchEvent { - override def equals(x: Any) = { - x match { - case x: ChoiceBranchStartEvent => x.qname == qname - case _ => false - } - } -} -object ChoiceBranchStartEvent { - def apply(nqn: NamedQName) = ChoiceBranchEvent(nqn)._1 -} -class ChoiceBranchEndEvent(val qname: NamedQName) extends ChoiceBranchEvent { - override def equals(x: Any) = { - x match { - case x: ChoiceBranchEndEvent => x.qname == qname - case _ => false - } - } -} -object ChoiceBranchEndEvent { - def apply(nqn: NamedQName) = ChoiceBranchEvent(nqn)._2 -} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetBuilder.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetBuilder.scala new file mode 100644 index 0000000000..c2f89afb6e --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetBuilder.scala @@ -0,0 +1,473 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.infoset + +import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.MStackOf +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One +import org.apache.daffodil.lib.xml.NamedQName +import org.apache.daffodil.runtime1.processors.ElementRuntimeData +import org.apache.daffodil.runtime1.processors.ModelGroupRuntimeData +import org.apache.daffodil.runtime1.processors.TermRuntimeData +import org.apache.daffodil.runtime1.processors.unparsers.InfosetBuildState +import org.apache.daffodil.runtime1.processors.unparsers.InfosetTreeState +import org.apache.daffodil.runtime1.processors.unparsers.UnparseError +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase +import org.apache.daffodil.unparsers.runtime1.RepeatingChildUnparser +import org.apache.daffodil.unparsers.runtime1.SequenceChildUnparser + +/** + * A node in a much smaller, dedicated tree of InfosetBuilders that parallels the + * full Unparser tree: only Grams that actually create or select infoset + * content (elements, sequences, choices, hidden groups) contribute one, so + * building the infoset never has to dispatch through the many wrapper + * unparsers that build no infoset events (delimiters, escape schemes, + * layers, padding, specified-length) that sit between them in the + * Unparser tree. + * + * An InfosetBuilder is immutable compiled-schema state shared by every unparse. The + * per-unparse position lives in the InfosetBuildFrame it creates, on an InfosetBuildCursor's + * explicit stack, so building can stop after any step and continue later + * without holding a thread or a JVM call stack. + */ +trait InfosetBuilder extends Serializable { + def newFrame(): InfosetBuildFrame + + // True only for NadaInfosetBuilder, which contributes nothing to the tree. + def isEmpty: Boolean = false + + // This builder if it builds anything, else the other one. + final def orElse(other: InfosetBuilder): InfosetBuilder = if (!isEmpty) this else other +} + +/** + * One InfosetBuilder's in-progress state for one unparse. `step` performs one + * transition and must either push exactly one child frame onto the cursor + * (this frame is stepped again once that child pops) or pop itself from the + * cursor to signal it is complete. + */ +abstract class InfosetBuildFrame { + def step(cursor: InfosetBuildCursor): Unit +} + +/** + * The explicit stack of InfosetBuildFrames that stands in for the call stack of a + * recursive build. `advance` runs it until the lead window is full, so a + * caller that needs more infoset tree can pull it forward directly. Driven + * against an `InfosetBuildState`. + */ +final class InfosetBuildCursor( + root: InfosetBuilder, + val buildState: InfosetBuildState +) { + def state: InfosetTreeState = buildState + + private val stack = new MStackOf[InfosetBuildFrame](64) + + push(root.newFrame()) + + def push(frame: InfosetBuildFrame): Unit = stack.push(frame) + + def pop(): Unit = stack.pop + + def isFinished: Boolean = stack.isEmpty + + /** + * Steps until the lead window is full or building completes, so it takes a + * single step when the window is already full. With lastAdvance it ignores + * the window and runs until building completes. A failure propagates and + * ends the unparse, so the cursor is not used again. + */ + def advance(lastAdvance: Boolean = false): Unit = { + while (!stack.isEmpty) { + stack.top.step(this) + if (!lastAdvance && buildState.leadExceedsBuildAheadLimit) { + return + } + } + } +} + +/** + * Builder for a Gram that creates or selects no infoset content. A builder + * with nothing to build collapses into this one, and a sequence drops its + * children that have it, so a parent only holds one where it needs a value + * in that slot, such as a choice branch with no content, which the branch map + * still needs an entry for. A parent that holds one recognizes it by isEmpty + * and never builds it. + */ +object NadaInfosetBuilder extends InfosetBuilder { + override def isEmpty = true + + override def toString = "Nada" + + override def newFrame(): InfosetBuildFrame = + Assert.abort("NadaInfosetBuilders are all supposed to optimize out!") +} + +/** + * Builds each of several sibling Grams' content in order. Used only where a + * `~` composition has more than one child that actually builds infoset + * content; the common case of at most one such child never needs this. + */ +final private class SeqCompInfosetBuilder(children: Array[InfosetBuilder]) + extends InfosetBuilder { + override def newFrame(): InfosetBuildFrame = new InfosetBuildFrame { + private var i = 0 + override def step(cursor: InfosetBuildCursor): Unit = { + if (i < children.length) { + val child = children(i) + i += 1 + cursor.push(child.newFrame()) + } else { + cursor.pop() + } + } + } +} + +object SeqCompInfosetBuilder { + def apply(children: Array[InfosetBuilder]): InfosetBuilder = { + if (children.isEmpty) { + NadaInfosetBuilder + } else if (children.length == 1) { + children.head + } else { + new SeqCompInfosetBuilder(children) + } + } +} + +/** + * Builds one element's infoset node and, for complex types, builds + * descendant nodes via contentBuilder. The element unparser's + * unparseBegin/unparseEnd are the same element-kind-specific + * (plain/nillable/OVC/etc.) node-creation logic unparse() itself uses, + * including the bounded-lookahead lead-counter hookup and the deferred + * simple-value finalization; only the "what does this element contain" + * step is redirected to the builder tree instead of back into the + * unparser tree. + */ +final class ElementInfosetBuilder( + erd: ElementRuntimeData, + elementUnparser: ElementUnparserBase, + contentBuilder: InfosetBuilder +) extends InfosetBuilder { + + override def newFrame(): InfosetBuildFrame = new InfosetBuildFrame { + private var contentPushed = false + + override def step(cursor: InfosetBuildCursor): Unit = { + val state = cursor.state + if (!contentPushed) { + elementUnparser.unparseBegin(state) + if (erd.isComplexType) { + state.pushTRD(erd.optComplexTypeModelGroupRuntimeData.get) + if (!contentBuilder.isEmpty) { + contentPushed = true + cursor.push(contentBuilder.newFrame()) + return + } + } + } + if (erd.isComplexType) { + state.popTRD(erd.optComplexTypeModelGroupRuntimeData.get) + } + elementUnparser.unparseEnd(state) + cursor.pop() + } + } +} + +/** + * Pairs a sequence child's existing occurs-count/array bookkeeping (reused + * as-is from the Unparser tree, since it is cheap, pure state bookkeeping + * unrelated to the tree-walking overhead this InfosetBuilder tree exists to avoid) + * with that same child's own InfosetBuilder, which SequenceInfosetBuilder builds + * instead of the child's full Unparser. + */ +final case class SequenceChildInfosetBuildInfo( + childUnparser: SequenceChildUnparser, + childBuilder: InfosetBuilder +) + +/** + * Where a sequence's frame is in its walk over the children. Start is before + * the first step. NextChild is between children. AfterScalar and + * AfterOccurrence are right after a child's frame popped. InArray is an array + * or optional loop between occurrences. + */ +private enum SequenceBuildPhase { + case Start, NextChild, AfterScalar, InArray, AfterOccurrence +} + +/** + * Builds an entire sequence's children, scalar and array/optional alike, + * each through its own InfosetBuilder. + */ +final private class SequenceInfosetBuilder(children: Array[SequenceChildInfosetBuildInfo]) + extends InfosetBuilder { + + override def newFrame(): InfosetBuildFrame = new SequenceInfosetBuildFrame + + private final class SequenceInfosetBuildFrame extends InfosetBuildFrame { + import SequenceBuildPhase.* + + private var phase: SequenceBuildPhase = Start + private var index = 0 + // children(index), read once per child and used by every later phase. + private var current: SequenceChildInfosetBuildInfo = null + private var rep: RepeatingChildUnparser = null + private var numOccurrences = 0 + private var maxReps = 0L + + override def step(cursor: InfosetBuildCursor): Unit = { + val state = cursor.state + phase match { + case Start => { + state.groupIndexStack.push(1L) + phase = NextChild + nextChild(cursor, state) + } + case NextChild => nextChild(cursor, state) + case AfterScalar => { + current.childUnparser.trd match { + case erd: ElementRuntimeData if !erd.isRepresented => // ok, skip group advance + case _ => state.moveOverOneGroupIndexOnly() + } + finishChild(state) + } + case AfterOccurrence => { + numOccurrences += 1 + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + state.moveOverOneGroupIndexOnly() + phase = InArray + } + case InArray => { + if (rep.shouldDoUnparser(rep, state)) { + phase = AfterOccurrence + cursor.push(current.childBuilder.newFrame()) + } else { + rep.checkFinalOccursCountBetweenMinAndMaxOccurs( + state, + rep, + numOccurrences, + maxReps, + state.arrayIterationPos - 1 + ) + rep.consumeEndArrayEvent(rep.erd, state) + finishRepeating(state) + } + } + } + } + + private def nextChild(cursor: InfosetBuildCursor, state: InfosetTreeState): Unit = { + if (index == children.length) { + state.groupIndexStack.pop() + cursor.pop() + } else { + current = children(index) + val cu = current.childUnparser + state.pushTRD(cu.trd) + cu match { + case r: RepeatingChildUnparser => { + rep = r + state.arrayIterationIndexStack.push(1L) + state.occursIndexStack.push(1L) + numOccurrences = 0 + maxReps = r.maxRepeatsConst + + Assert.invariant(state.inspect, "No event for building.") + val ev = state.inspectAccessor + if (ev.erd eq r.erd) { + r.startArrayOrOptional(state) + phase = InArray + } else { + r.checkFinalOccursCountBetweenMinAndMaxOccurs( + state, + r, + numOccurrences, + maxReps, + 0 + ) + finishRepeating(state) + } + } + case _ => { + phase = AfterScalar + cursor.push(current.childBuilder.newFrame()) + } + } + } + } + + private def finishRepeating(state: InfosetTreeState): Unit = { + state.arrayIterationIndexStack.pop() + state.occursIndexStack.pop() + rep = null + finishChild(state) + } + + private def finishChild(state: InfosetTreeState): Unit = { + state.popTRD(children(index).childUnparser.trd) + index += 1 + phase = NextChild + } + } +} + +object SequenceInfosetBuilder { + def apply(children: Array[SequenceChildInfosetBuildInfo]): InfosetBuilder = { + val nonEmptyChildren = children.filterNot(_.childBuilder.isEmpty) + if (nonEmptyChildren.isEmpty) { + NadaInfosetBuilder + } else { + new SequenceInfosetBuilder(nonEmptyChildren) + } + } +} + +/** + * Builds just the one structurally-present branch of a choice, resolved + * from the next infoset event, through that branch's own InfosetBuilder. + */ +final class ChoiceInfosetBuilder( + mgrd: ModelGroupRuntimeData, + branchMap: Map[NamedQName, (TermRuntimeData, InfosetBuilder)], + defaultBranch: Maybe[(TermRuntimeData, InfosetBuilder)] +) extends InfosetBuilder { + + private def resolveBranch(state: InfosetTreeState): (TermRuntimeData, InfosetBuilder) = { + if (state.withinHiddenNest) { + defaultBranch.get + } else { + state.pushTRD(mgrd) + val event = state.inspectOrError + // An end event never starts a branch, so it always takes the default. + val fromTable = if (event.isStart) { + branchMap.get(event.erd.namedQName) + } else { + None + } + val resolved = if (fromTable.isDefined) { + fromTable + } else { + defaultBranch.toOption + } + if (resolved.isEmpty) { + UnparseError( + One(mgrd.schemaFileLocation), + Nope, + "Found next element %s, but expected one of %s.", + event.erd.namedQName.toExtendedSyntax, + branchMap.keys.map { _.toExtendedSyntax }.mkString(", ") + ) + } + state.popTRD(mgrd) + resolved.get + } + } + + override def newFrame(): InfosetBuildFrame = new InfosetBuildFrame { + private var branchTRD: TermRuntimeData = null + + override def step(cursor: InfosetBuildCursor): Unit = { + val state = cursor.state + if (branchTRD == null) { + val (trd, builder) = resolveBranch(state) + branchTRD = trd + state.pushTRD(trd) + if (!builder.isEmpty) { + cursor.push(builder.newFrame()) + } + } else { + state.popTRD(branchTRD) + cursor.pop() + } + } + } +} + +/** + * Builds the body of a hidden group. withinHiddenNest must stay maintained + * during build too: it is what tells a hidden element's unparseBegin/ + * unparseEnd to manufacture a node instead of consuming an event that will + * never exist. + */ +final private class HiddenGroupInfosetBuilder(bodyBuilder: InfosetBuilder) + extends InfosetBuilder { + override def newFrame(): InfosetBuildFrame = new InfosetBuildFrame { + private var bodyPushed = false + + override def step(cursor: InfosetBuildCursor): Unit = { + if (!bodyPushed) { + bodyPushed = true + cursor.state.incrementHiddenDef() + cursor.push(bodyBuilder.newFrame()) + } else { + cursor.state.decrementHiddenDef() + cursor.pop() + } + } + } +} + +object HiddenGroupInfosetBuilder { + def apply(bodyBuilder: InfosetBuilder): InfosetBuilder = { + if (bodyBuilder.isEmpty) { + NadaInfosetBuilder + } else { + new HiddenGroupInfosetBuilder(bodyBuilder) + } + } +} + +/** + * A nilled complex element has no children to build; nilled-ness is only + * known once the node exists, so this checks it at build time rather than + * resolving statically to either branch. + */ +final private class NilOrContentInfosetBuilder(contentBuilder: InfosetBuilder) + extends InfosetBuilder { + override def newFrame(): InfosetBuildFrame = new InfosetBuildFrame { + private var contentPushed = false + + override def step(cursor: InfosetBuildCursor): Unit = { + if (!contentPushed && !cursor.state.currentInfosetNode.asComplex.isNilled) { + contentPushed = true + cursor.push(contentBuilder.newFrame()) + } else { + cursor.pop() + } + } + } +} + +object NilOrContentInfosetBuilder { + def apply(contentBuilder: InfosetBuilder): InfosetBuilder = { + if (contentBuilder.isEmpty) { + NadaInfosetBuilder + } else { + new NilOrContentInfosetBuilder(contentBuilder) + } + } +} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetWalker.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetWalker.scala index b2d36dc92c..2ab29cc6a3 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetWalker.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetWalker.scala @@ -150,6 +150,13 @@ object StreamingInfosetWalker { * (e.g. due to an unresolved point of uncertainty) and increases the * number of walk() calls to skip before trying again. This defines the * maximum number of skipped calls, even as that number increases. + * + * @param visibleChildCounts + * + * When not null, the number of children of a container that the walk may + * visit, for a container that has an entry; any other container is walked + * in full. This should only be used while debugging, to leave out nodes + * that exist in the infoset but that unparsing has not reached. */ def apply( root: DIElement, @@ -158,7 +165,8 @@ object StreamingInfosetWalker { ignoreBlocks: Boolean, releaseUnneededInfoset: Boolean, walkSkipMin: Int = 32, - walkSkipMax: Int = 2048 + walkSkipMax: Int = 2048, + visibleChildCounts: java.util.Map[DINode, Integer] = null ): StreamingInfosetWalker = { // Determine the container of the root node and the index in which it @@ -187,7 +195,8 @@ object StreamingInfosetWalker { ignoreBlocks, releaseUnneededInfoset, walkSkipMin, - walkSkipMax + walkSkipMax, + visibleChildCounts ) } @@ -252,6 +261,12 @@ object StreamingInfosetWalker { * being blocked for removal (e.g. due to an unresolved point of uncertainty) * and increases the number of walk() calls to skip before trying again. This * defines the maximum number of skiped calls, even as this number increases. + * + * @param visibleChildCounts + * + * When not null, the number of children of a container that the walk may + * visit, for a container that has an entry; any other container is walked + * in full. This should only be used while debugging. */ class StreamingInfosetWalker private ( startingContainerNode: DINode, @@ -261,7 +276,8 @@ class StreamingInfosetWalker private ( ignoreBlocks: Boolean, releaseUnneededInfoset: Boolean, walkSkipMin: Int, - walkSkipMax: Int + walkSkipMax: Int, + visibleChildCounts: java.util.Map[DINode, Integer] ) extends InfosetWalker { /** @@ -563,12 +579,25 @@ class StreamingInfosetWalker private ( finished = true } + private def visibleChildCount(containerNode: DINode): Int = { + if (visibleChildCounts eq null) { + containerNode.numChildren + } else { + val count = visibleChildCounts.get(containerNode) + if (count eq null) { + containerNode.numChildren + } else { + count.intValue + } + } + } + /** * Output start/end events for DIComplex/DIArray/DISimple, and mutate state * so we are looking at the next node in the infoset. */ private def infosetWalkerStepMove(containerNode: DINode, containerIndex: Int): Unit = { - if (containerIndex < containerNode.numChildren) { + if (containerIndex < visibleChildCount(containerNode)) { // This block means we need to create a start event for the element // at containerIndex. Once we create that event we // need to mutate the state of the InfosetWalker so that the next time we diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala index 3b16d3680e..50ab43a300 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala @@ -39,6 +39,7 @@ import org.apache.daffodil.api.validation.ValidatorInitializationException import org.apache.daffodil.api.validation.Validators import org.apache.daffodil.lib.equality.* import org.apache.daffodil.lib.iapi.DaffodilTunables +import org.apache.daffodil.lib.iapi.InfosetBuilderMode import org.apache.daffodil.lib.iapi.WithDiagnostics import org.apache.daffodil.runtime1.dsom.* import org.apache.daffodil.runtime1.iapi.DFDL @@ -61,6 +62,7 @@ import org.apache.daffodil.lib.util.ThreadSafePool import org.apache.daffodil.runtime1.events.MultipleEventHandler import org.apache.daffodil.runtime1.externalvars.ExternalVariablesLoader import org.apache.daffodil.runtime1.infoset.DIElement +import org.apache.daffodil.runtime1.infoset.InfosetBuildCursor import org.apache.daffodil.runtime1.infoset.InfosetException import org.apache.daffodil.runtime1.infoset.InfosetInputter import org.apache.daffodil.runtime1.infoset.TeeInfosetOutputter @@ -68,6 +70,10 @@ import org.apache.daffodil.runtime1.infoset.XMLTextInfosetOutputter import org.apache.daffodil.runtime1.processors.parsers.PState import org.apache.daffodil.runtime1.processors.parsers.ParseError import org.apache.daffodil.runtime1.processors.parsers.Parser +import org.apache.daffodil.runtime1.processors.unparsers.ChildNotBuiltException +import org.apache.daffodil.runtime1.processors.unparsers.InfosetBuildState +import org.apache.daffodil.runtime1.processors.unparsers.NotUnparsableUnparser +import org.apache.daffodil.runtime1.processors.unparsers.TreeEventState import org.apache.daffodil.runtime1.processors.unparsers.UState import org.apache.daffodil.runtime1.processors.unparsers.UnparseError @@ -458,6 +464,189 @@ class DataProcessor( } def unparse(actualInputter: api.infoset.InfosetInputter, outStream: java.io.OutputStream) = { + // A NotUnparsableUnparser (dfdl:parseUnparsePolicy="parseOnly") has no + // builders to run ahead, and unparseEventDriven already gives the correct + // diagnostic for it, so it never builds ahead. + val canBuildAhead = !ssrd.unparser.isInstanceOf[NotUnparsableUnparser] + if ((tunables.infosetBuilderMode eq InfosetBuilderMode.BuildAhead) && canBuildAhead) { + unparseBuildAhead(actualInputter, outStream) + } else { + unparseEventDriven(actualInputter, outStream) + } + } + + /** + * Shared by unparseBuildAhead and unparseEventDriven's top-level + * catch blocks: maps an exception caught during unparsing to a failed + * `state` plus its `unparseResult`, or rethrows if it's not one of the + * known unparse-error categories. + */ + private def unparseErrorResult(state: UState, t: Throwable): UnparseResult = t match { + case ue: UnparseError => { + state.addUnparseError(ue) + state.unparseResult + } + case procErr: ProcessingError => { + state.setFailed(procErr.toUnparseError) + state.unparseResult + } + case sde: SchemaDefinitionError => { + // A SDE was detected at runtime (perhaps due to a runtime-valued property like byteOrder or encoding) + // These are fatal, and there's no notion of backtracking them, so they propagate to top level + // here. + state.setFailed(sde) + state.unparseResult + } + case sdefw: SchemaDefinitionErrorFromWarning => { + state.setFailed(sdefw) + state.unparseResult + } + case e: ErrorAlreadyHandled => { + state.setFailed(e.th) + state.unparseResult + } + case e: TunableLimitExceededError => { + state.setFailed(e) + state.unparseResult + } + case se: org.xml.sax.SAXException => { + state.setFailed(new UnparseError(None, None, se)) + state.unparseResult + } + case e: scala.xml.parsing.FatalError => { + state.setFailed(new UnparseError(None, None, e)) + state.unparseResult + } + case ie: InfosetException => { + state.setFailed(new UnparseError(None, None, ie)) + state.unparseResult + } + case th: Throwable => throw th + } + + /** + * Unparses with a build pass running ahead of the unparse (gated on + * `infosetBuilderMode`). An `InfosetBuildCursor` over `InfosetBuildState` + * builds the infoset tree from the inputter's events, and the unparse + * reads that tree back as events, advancing the cursor whenever it needs a + * node that does not exist yet. + */ + private def unparseBuildAhead( + actualInputter: api.infoset.InfosetInputter, + outStream: java.io.OutputStream + ): UnparseResult = { + val rootUnparser = ssrd.unparser + + val inputter = new InfosetInputter(actualInputter) + + // Build side. The root element always has a builder: it is exactly the + // case that gets ElementInfosetBuilder wrapped around it, regardless of + // schema content. + val infosetBuildState = new InfosetBuildState(inputter, tunables) + val cursor = new InfosetBuildCursor(ssrd.builder, infosetBuildState) + + // Unparse side. The tree events are the events of the tree build is making, + // so the unparsers read it as they read an inputter's. The state holds the + // output stream, which must be cleaned up. + val treeEvents = new TreeEventState( + cursor, + !areDebugging && tunables.releaseUnneededInfoset + ) + val unparseState = + UState.createInitialUStateForBuildAhead( + outStream, + this, + inputter, + areDebugging, + treeEvents + ) + + try { + inputter.initialize(ssrd.elementRuntimeData, tunables) + + if (areDebugging) { + Assert.invariant(optDebugger.isDefined) + addEventHandler(debugger) + } + if (areDebugging) { + unparseState.notifyDebugging(true) + } + // The root TRD is on the stack of the tree events, as it is + // on an inputter's when it is initialized. + unparseState.pushTRD(ssrd.elementRuntimeData) + init(unparseState, rootUnparser) + // Forces evaluation of non-constant defineVariable defaults; the unparse + // side reads and writes variables, so it needs this on its own copy. + unparseState.initializeVariables() + unparseState.getDataOutputStream.setPriorBitOrder( + ssrd.elementRuntimeData.defaultBitOrder + ) + + try { + rootUnparser.unparse1(unparseState) + unparseState.popTRD(rootUnparser.context.asInstanceOf[TermRuntimeData]) + } catch { + // A genuine deadlock (if any) surfaces via the final + // evalSuspensions(isFinal = true) below. + case _: ChildNotBuiltException => + } + + // The unparse only ever advances build as far as it needs, so build may + // still have its trailing end events left to consume. + cursor.advance(lastAdvance = true) + + // Build's stacks must end up balanced and the inputter must have nothing + // left unconsumed. + infosetBuildState.popTRD(rootUnparser.context.asInstanceOf[TermRuntimeData]) + Assert.invariant(infosetBuildState.arrayIterationIndexStack.length == 1) + Assert.invariant(infosetBuildState.occursIndexStack.length == 1) + Assert.invariant(infosetBuildState.groupIndexStack.length == 1) + Assert.invariant(infosetBuildState.currentInfosetNodeMaybe.isEmpty) + Assert.invariant(infosetBuildState.maybeTopTRD().isEmpty) + val remainingEvent = infosetBuildState.advanceMaybe + if (remainingEvent.isDefined) { + UnparseError( + Nope, + Nope, + "Expected no remaining events, but received %s.", + remainingEvent.get + ) + } + + // The final suspension sweep runs BEFORE the stack-depth invariants: one + // tripping first could mask the real SuspensionDeadlockException diagnostic + // this ordering exists to surface. + unparseState.setProcessor(rootUnparser) + unparseState.evalSuspensions(isFinal = true) + Assert.invariant(unparseState.arrayIterationIndexStack.length == 1) + Assert.invariant(unparseState.occursIndexStack.length == 1) + Assert.invariant(unparseState.groupIndexStack.length == 1) + Assert.invariant(unparseState.childIndexStack.length == 1) + Assert.invariant(unparseState.currentInfosetNodeMaybe.isEmpty) + Assert.invariant(unparseState.escapeSchemeEVCache.isEmpty) + Assert.invariant(unparseState.maybeTopTRD().isEmpty) + Assert.invariant(!unparseState.withinHiddenNest) + Assert.invariant(!unparseState.getDataOutputStream.isFinished) + try { + unparseState.getDataOutputStream.setFinished(unparseState) + } catch { + case boc: BitOrderChangeException => + unparseState.SDE(boc) + case fio: FileIOException => + unparseState.SDE(fio) + } + unparseState.unparseResult + } catch { + case t: Throwable => unparseErrorResult(unparseState, t) + } finally { + unparseState.getDataOutputStream.cleanUp() + } + } + + private def unparseEventDriven( + actualInputter: api.infoset.InfosetInputter, + outStream: java.io.OutputStream + ) = { val inputter = new InfosetInputter(actualInputter) val unparserState = UState.createInitialUState(outStream, this, inputter, areDebugging) @@ -477,47 +666,7 @@ class DataProcessor( unparserState.evalSuspensions(isFinal = true) unparserState.unparseResult } catch { - case ue: UnparseError => { - unparserState.addUnparseError(ue) - unparserState.unparseResult - } - case procErr: ProcessingError => { - val x = procErr - unparserState.setFailed(x.toUnparseError) - unparserState.unparseResult - } - case sde: SchemaDefinitionError => { - // A SDE was detected at runtime (perhaps due to a runtime-valued property like byteOrder or encoding) - // These are fatal, and there's no notion of backtracking them, so they propagate to top level - // here. - unparserState.setFailed(sde) - unparserState.unparseResult - } - case sdefw: SchemaDefinitionErrorFromWarning => { - unparserState.setFailed(sdefw) - unparserState.unparseResult - } - case e: ErrorAlreadyHandled => { - unparserState.setFailed(e.th) - unparserState.unparseResult - } - case e: TunableLimitExceededError => { - unparserState.setFailed(e) - unparserState.unparseResult - } - case se: org.xml.sax.SAXException => { - unparserState.setFailed(new UnparseError(None, None, se)) - unparserState.unparseResult - } - case e: scala.xml.parsing.FatalError => { - unparserState.setFailed(new UnparseError(None, None, e)) - unparserState.unparseResult - } - case ie: InfosetException => { - unparserState.setFailed(new UnparseError(None, None, ie)) - unparserState.unparseResult - } - case th: Throwable => throw th + case t: Throwable => unparseErrorResult(unparserState, t) } finally { unparserState.getDataOutputStream.cleanUp() } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala index e830aead7b..ae93ebde92 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala @@ -18,6 +18,7 @@ package org.apache.daffodil.runtime1.processors import org.apache.daffodil.lib.exceptions.ThrowsSDE +import org.apache.daffodil.runtime1.infoset.InfosetBuilder import org.apache.daffodil.runtime1.layers.LayerRuntimeCompiler import org.apache.daffodil.runtime1.layers.LayerRuntimeData import org.apache.daffodil.runtime1.layers.LayerVarsRuntime @@ -27,6 +28,8 @@ import org.apache.daffodil.runtime1.processors.unparsers.Unparser final class SchemaSetRuntimeData( val parser: Parser, val unparser: Unparser, + /** Nada only when the schema was compiled without an unparser. */ + val builder: InfosetBuilder, val elementRuntimeData: ElementRuntimeData, /* * The original variables determined by the schema compiler. diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/parsers/SequenceChildBases.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/parsers/SequenceChildBases.scala index fc079317c0..865db6879c 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/parsers/SequenceChildBases.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/parsers/SequenceChildBases.scala @@ -494,7 +494,7 @@ trait MinMaxRepeatsMixin { final val ock = erd.maybeOccursCountKind.get - private val minRepeats_ = { + final val minRepeatsConst: Long = { val mr = if (ock eq OccursCountKind.Parsed) 0 else erd.minOccurs @@ -506,7 +506,7 @@ trait MinMaxRepeatsMixin { * For example, when occursCountKind is parsed, then minRepeats is 0, regardless * of the value of minOccurs. */ - def minRepeats(state: ParseOrUnparseState): Long = minRepeats_ + def minRepeats(state: ParseOrUnparseState): Long = minRepeatsConst /** * True if the loop has a finite upper bound on number of iterations. @@ -514,7 +514,7 @@ trait MinMaxRepeatsMixin { * for speculative parsing cases, it's not OCK parsed, or OCK implicit with * maxOccurs unbounded. */ - private val maxRepeats_ = { + final val maxRepeatsConst: Long = { if (ock eq OccursCountKind.Parsed) Long.MaxValue else if (erd.maxOccurs == -1) Long.MaxValue else erd.maxOccurs @@ -525,9 +525,9 @@ trait MinMaxRepeatsMixin { * For example, when occursCountKind is parsed, then maxRepeats is -1 (meaning unbounded) * regardless of the value of maxOccurs. */ - def maxRepeats(state: ParseOrUnparseState): Long = maxRepeats_ + def maxRepeats(state: ParseOrUnparseState): Long = maxRepeatsConst - private val isBoundedMax_ = maxRepeats_ < Long.MaxValue + private val isBoundedMax_ = maxRepeatsConst < Long.MaxValue def isBoundedMax: Boolean = isBoundedMax_ diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/InfosetBuildState.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/InfosetBuildState.scala new file mode 100644 index 0000000000..8885dab038 --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/InfosetBuildState.scala @@ -0,0 +1,135 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.iapi.DaffodilTunables +import org.apache.daffodil.lib.util.MStackOfMaybe +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.runtime1.infoset.DIDocument +import org.apache.daffodil.runtime1.infoset.DIElement +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.infoset.InfosetAccessor +import org.apache.daffodil.runtime1.infoset.InfosetInputter +import org.apache.daffodil.runtime1.processors.ElementRuntimeData +import org.apache.daffodil.runtime1.processors.TermRuntimeData + +/** + * The "build" side of the build and unparse split: it walks the infoset + * events from an actual `InfosetInputter` and builds the infoset tree ahead + * of the unparse. It needs only the tree state, so it is not a `UState`: + * it has no output stream, variables or debugger state, and build never + * writes content. + * + * Used only when the `infosetBuilderMode` tunable is buildAhead when unparsing; + * otherwise only `UStateMain` is constructed. + */ +final class InfosetBuildState( + private val inputter: InfosetInputter, + override val tunable: DaffodilTunables +) extends InfosetTreeState + with TraversalIndexStacks + with InfosetFromEvents { + + // Build never frees a node, so finishing one only marks it final. A simple + // node stays open for the unparse, which gives it its value. + override def finishElement(cur: DINode, erd: ElementRuntimeData): Unit = { + if (cur.isComplex) { + val lastChild = cur.maybeLastChild + if (lastChild.isDefined && lastChild.get.isArray) { + lastChild.get.setFinal() + } + if (!withinHiddenNest || erd.isRepresented) { + cur.setFinal() + } + } + markDocumentFinalIfRootEnded() + } + + override def finishOvcElement(cur: DINode): Unit = markDocumentFinalIfRootEnded() + + private val eventState: InfosetEventState = new InputterEventState(inputter, "building") + + override def advance: Boolean = eventState.advance + override def advanceAccessor: InfosetAccessor = eventState.advanceAccessor + override def inspect: Boolean = eventState.inspect + override def inspectAccessor: InfosetAccessor = eventState.inspectAccessor + override def fini(): Unit = Assert.usageError("Not to be used on InfosetBuildState") + override def inspectOrError: InfosetAccessor = eventState.inspectOrError + override def advanceOrError: InfosetAccessor = eventState.advanceOrError + override def isInspectArrayEnd: Boolean = eventState.isInspectArrayEnd + + override def pushTRD(trd: TermRuntimeData): Unit = eventState.pushTRD(trd) + override def maybeTopTRD(): Maybe[TermRuntimeData] = eventState.maybeTopTRD() + override def popTRD(trd: TermRuntimeData): TermRuntimeData = eventState.popTRD(trd) + + override def documentElement: DIDocument = inputter.documentElement + + override val currentInfosetNodeStack = new MStackOfMaybe[DINode] + + override def currentInfosetNode: DINode = { + if (currentInfosetNodeMaybe.isEmpty) { + null + } else { + currentInfosetNodeMaybe.get + } + } + + override def currentInfosetNodeMaybe: Maybe[DINode] = { + if (currentInfosetNodeStack.isEmpty) { + Nope + } else { + currentInfosetNodeStack.top + } + } + + // Build tracks child position in its own frames, never in a stack. + override def moveOverOneElementChildOnly(): Unit = () + + private var hiddenDepth = 0 + override def incrementHiddenDef(): Unit = hiddenDepth += 1 + override def decrementHiddenDef(): Unit = hiddenDepth -= 1 + override def withinHiddenNest: Boolean = hiddenDepth > 0 + + // Build runs ahead of the unparse, which frees each node once it is done + // with it, so build leaves freeing to the unparse. + override def freeChildIfNoLongerNeeded(parent: DINode, index: Int): Unit = () + + // The lead is how far build is ahead of the unparse: the nodes build has + // constructed that the unparse has not yet finished. Build counts a node when + // it joins the tree, which keeps an ancestor's count ahead of its + // descendants', and the unparse uncounts it when it finishes the node. + private var lead: Long = 0 + + override def attachElement(newElem: DIElement): Unit = { + super.attachElement(newElem) + lead += 1 + } + + def decrementLead(): Unit = { + lead -= 1 + Assert.invariant(lead >= 0) + } + + def currentLead: Long = lead + + // Build stops advancing once the lead exceeds the build ahead limit, which + // bounds how far ahead of the unparse it may run. + def leadExceedsBuildAheadLimit: Boolean = lead > tunable.unparseBuildAheadWindowNodes +} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/InfosetFromEvents.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/InfosetFromEvents.scala new file mode 100644 index 0000000000..604bb1424c --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/InfosetFromEvents.scala @@ -0,0 +1,146 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One +import org.apache.daffodil.runtime1.infoset.DIComplex +import org.apache.daffodil.runtime1.infoset.DIElement +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.infoset.DISimple +import org.apache.daffodil.runtime1.infoset.InfosetAccessor +import org.apache.daffodil.runtime1.processors.ElementRuntimeData + +/** + * How unparsing the events of an inputter makes the infoset: it creates the + * node of an element that has no event, attaches each new node to its parent, + * and finishes a node when its end is reached. A state that unparses a tree + * that was already built overrides these. + */ +trait InfosetFromEvents { self: InfosetTreeState => + + override def getHiddenElement(erd: ElementRuntimeData): DIElement = { + // Since we never get events for elements in hidden contexts, their infoset elements + // will have never been created. This means we need to manually create them + val hiddenElem = if (erd.isComplexType) { + new DIComplex(erd) + } else { + new DISimple(erd) + } + hiddenElem.setHidden() + hiddenElem + } + + override def getOvcElement( + startEvent: InfosetAccessor, + erd: ElementRuntimeData + ): DIElement = { + val e = new DISimple(erd) + // Remove any state that was set by what created this event. Later + // code asserts that OVC elements do not have a value + e.resetValue() + e + } + + override def attachElement(newElem: DIElement): Unit = { + val parentNodeMaybe = currentInfosetNodeMaybe + if (parentNodeMaybe.isDefined) { + val parentComplex = parentNodeMaybe.get.asComplex + Assert.invariant(!parentComplex.isFinal) + if (parentComplex.isNilled) { + // cannot add content to a nilled complex element + UnparseError( + One(newElem.erd.schemaFileLocation), + Nope, + "Nilled complex element %s has content from %s", + parentComplex.erd.namedQName.toExtendedSyntax, + newElem.erd.namedQName.toExtendedSyntax + ) + } + + // We are about to add a child to this complex element. Before we do + // that, if the last child added to this complex is a DIArray, and this + // new child isn't part of that array, that implies that the DIArray + // will have no more children added and should be marked as final, and + // we can attempt to free that array. + val lastChildMaybe = parentComplex.maybeLastChild + if (lastChildMaybe.isDefined) { + val lastChild = lastChildMaybe.get + if (lastChild.isArray && (lastChild.erd ne newElem.erd)) { + lastChild.setFinal() + freeChildIfNoLongerNeeded(parentComplex, parentComplex.numChildren - 1) + } + } + + parentComplex.addChild(newElem, tunable) + } else { + // We do not yet have an infoset element (this new element is the + // root), so add the infoset node to the DIDocument + documentElement.addChild(newElem, tunable) + } + } + + override def finishElement(cur: DINode, erd: ElementRuntimeData): Unit = { + if (cur.isComplex) { + // We are ending a complex element. If the last child of this complex + // is a DIArray, that implies that the array will have no more children + // and should be marked as isFinal. Normally this happens when we add a + // new sibling after an array in attachElement, but in this case there + // is no sibling following the array, so it must be set here. + val lastChild = cur.maybeLastChild + if (lastChild.isDefined && lastChild.get.isArray) { + lastChild.get.setFinal() + freeChildIfNoLongerNeeded(cur, cur.numChildren - 1) + } + } + + // cur is finished: mark it final and free via its container (not + // parent, so an array-member frees from the array), except hidden + // IVC elements (never get a value). + if (!withinHiddenNest || erd.isRepresented) { + cur.setFinal() + } + val curContainer = if (cur.erd.isArray) { + cur.diParent.maybeLastChild.get + } else { + cur.diParent + } + freeChildIfNoLongerNeeded(curContainer, curContainer.numChildren - 1) + markDocumentFinalIfRootEnded() + } + + override def finishOvcElement(cur: DINode): Unit = { + // OVC elements are not allowed in arrays, so we can directly get the + // diParent to get the container DINode + val ovcContainer = cur.diParent + freeChildIfNoLongerNeeded(ovcContainer, ovcContainer.numChildren - 1) + markDocumentFinalIfRootEnded() + } + + protected def markDocumentFinalIfRootEnded(): Unit = { + if (currentInfosetNodeStack.isEmpty) { + // If there is nothing else on the infoset stack after popping off the + // current infoset node, that means we have finished the root element, + // so mark the DIDocument as final + val doc = documentElement + Assert.invariant(!doc.isFinal) + doc.setFinal() + } + } +} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/InfosetFromTree.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/InfosetFromTree.scala new file mode 100644 index 0000000000..00b161cd3d --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/InfosetFromTree.scala @@ -0,0 +1,67 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.runtime1.infoset.DIElement +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.infoset.InfosetAccessor +import org.apache.daffodil.runtime1.processors.ElementRuntimeData + +/** + * How unparsing a tree that build made ahead of it finds the infoset nodes, in place of InfosetFromEvents: build already made, attached + * and finished each node, so this takes each node from the tree, and + * finishing one only leaves what build has counted for it. + */ +trait InfosetFromTree { self: InfosetTreeState => + + protected def treeEvents: TreeEventState + + // Hidden elements have no events, but build already created this one. + override def getHiddenElement(erd: ElementRuntimeData): DIElement = + treeEvents.takeExistingHiddenChild(currentInfosetNode, erd) + + // Build already made this element and removed what created its event. + override def getOvcElement( + startEvent: InfosetAccessor, + erd: ElementRuntimeData + ): DIElement = + startEvent.info.element + + // Build already attached it. + override def attachElement(newElem: DIElement): Unit = () + + override def finishElement(cur: DINode, erd: ElementRuntimeData): Unit = + finishExistingNode(cur) + + override def finishOvcElement(cur: DINode): Unit = finishExistingNode(cur) + + /** + * Build attached the node and marked it final, apart from a simple node + * still waiting for its value, and the event state frees it once its end + * event is out. Unparsing the node is done, so it leaves build's lead. + */ + private def finishExistingNode(cur: DINode): Unit = { + if (cur.isSimple && !cur.isFinal && cur.asSimple.hasValue) { + cur.setFinal() + } + treeEvents.decrementLead() + if (withinHiddenNest) { + treeEvents.freeExistingHiddenChild() + } + } +} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/TreeEventState.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/TreeEventState.scala new file mode 100644 index 0000000000..8d31186dd7 --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/TreeEventState.scala @@ -0,0 +1,392 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import scala.annotation.tailrec + +import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.util.MStackOf +import org.apache.daffodil.lib.util.MStackOfInt +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.lib.util.Maybe.One +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.runtime1.infoset.DIDocument +import org.apache.daffodil.runtime1.infoset.DIElement +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.infoset.DISimple +import org.apache.daffodil.runtime1.infoset.Info +import org.apache.daffodil.runtime1.infoset.InfosetAccessor +import org.apache.daffodil.runtime1.infoset.InfosetBuildCursor +import org.apache.daffodil.runtime1.infoset.InfosetEventKind +import org.apache.daffodil.runtime1.processors.ElementRuntimeData +import org.apache.daffodil.runtime1.processors.TermRuntimeData + +/** + * The infoset events of the tree that build is constructing, so the + * unparsers consume them exactly as they consume the events of an + * InfosetInputter. A node build has not produced yet is waited for by + * advancing the build cursor. Hidden nodes get no events, as + * in the event stream the tree was built from. + * + * The walk is an explicit stack of the containers being visited, so it can + * stop after any event and resume later. + */ +final class TreeEventState( + cursor: InfosetBuildCursor, + releaseUnneededInfoset: Boolean +) extends InfosetEventState { + + // The walk's stack of containers being visited, innermost last. It is held + // in parallel arrays, so visiting a node allocates nothing. + private var containerStack = new Array[DINode](16) + + // For the container at the same position of the stack, the index of the next + // of its children to visit. + private var nextChildIndexes = new Array[Int](16) + + // For the container at the same position of the stack, what kind of node it + // is, found once when the container was first reached. + private var containerKinds = new Array[NodeKind](16) + + // How many containers are on the stack. + private var stackDepth = 0 + + // The simple node whose start event was emitted and whose end event is next. + // A simple node has no children to visit, so it never goes on the stack. + private var pendingEnd: DINode = null + + // The document may not exist yet when this is constructed, so the walk + // starts at the first event asked for. + private var started = false + + // What a node is, found once when it is first reached, so the walk does not + // test its type again for each of its events. + private def kindOf(node: DINode): NodeKind = { + node match { + case _: DISimple => NodeKind.Simple + case _: DIArray => NodeKind.Array + case _: DIDocument => NodeKind.Document + case _ => NodeKind.Complex + } + } + + private def pushContainer(node: DINode, kind: NodeKind): Unit = { + if (stackDepth == containerStack.length) { + containerStack = java.util.Arrays.copyOf(containerStack, stackDepth * 2) + nextChildIndexes = java.util.Arrays.copyOf(nextChildIndexes, stackDepth * 2) + containerKinds = java.util.Arrays.copyOf(containerKinds, stackDepth * 2) + } + containerStack(stackDepth) = node + nextChildIndexes(stackDepth) = 0 + containerKinds(stackDepth) = kind + stackDepth += 1 + } + + private def popContainer(): Unit = { + stackDepth -= 1 + containerStack(stackDepth) = null + } + + // Two accessors trade roles: one holds the event that was computed and not + // yet consumed, the other holds the event consumed last. Consuming an event + // swaps them instead of copying the event. + private var inspected = InfosetAccessor() + private var advanced = InfosetAccessor() + private var hasInspectedEvent = false + + private val trdStack = new MStackOf[TermRuntimeData]() + + // Makes the next event the inspected one, if there is one. + private def fill(): Boolean = { + if (!hasInspectedEvent) { + if (!started) { + started = true + pushContainer(cursor.buildState.documentElement, NodeKind.Document) + } + hasInspectedEvent = computeNext() + } + hasInspectedEvent + } + + // Sets the inspected accessor to the next event. False if there are no more. + @tailrec + private def computeNext(): Boolean = { + if (pendingEnd ne null) { + val ended = pendingEnd + pendingEnd = null + setEvent(ended, NodeKind.Simple, isStart = false) + freeEndedChild() + true + } else if (stackDepth == 0) { + false + } else { + val innermost = stackDepth - 1 + val currentNode = containerStack(innermost) + val currentKind = containerKinds(innermost) + if (childExistsOrFinal(currentNode, nextChildIndexes(innermost))) { + // A node ahead of this walk that is already freed was a hidden one, + // which finished before the walk got here. + val nextChild = currentNode.child(nextChildIndexes(innermost)) + nextChildIndexes(innermost) += 1 + if (nextChild eq null) { + computeNext() + } else { + val nextKind = kindOf(nextChild) + if (isHidden(nextChild, nextKind)) { + computeNext() + } else { + setEvent(nextChild, nextKind, isStart = true) + if (nextKind eq NodeKind.Simple) { + pendingEnd = nextChild + } else { + pushContainer(nextChild, nextKind) + } + true + } + } + } else if (currentKind eq NodeKind.Document) { + popContainer() + computeNext() + } else { + emitEnd(currentNode, currentKind) + true + } + } + } + + // Advances build until the child exists, or the parent is final with no + // child there. The unparse only needs the child to exist, so it knows which + // branch or occurrence it is on, not for it to have a value. Build having + // finished with the child still missing is a stall. + private def childExistsOrFinal(parent: DINode, index: Int): Boolean = { + while (index >= parent.numChildren) { + if (parent.isFinal) { + return false + } + if (cursor.isFinished) { + throw new ChildNotBuiltException + } + cursor.advance() + } + true + } + + /** + * For each container the unparse is inside, how many of its children the + * unparse has reached, so a debugger can show the infoset as an event-driven + * unparse would have built it by now. A container the unparse is not inside + * is fully reached or not reached at all, so it has no entry. A node whose + * start event is computed but not yet consumed is left out. + */ + def reachedChildCounts(): java.util.Map[DINode, Integer] = { + val counts = new java.util.IdentityHashMap[DINode, Integer] + var i = 0 + while (i < stackDepth) { + counts.put(containerStack(i), nextChildIndexes(i)) + i += 1 + } + if (hasInspectedEvent && inspected.isStart) { + if (pendingEnd ne null) { + // The started node is simple, so it is not on the stack. + counts.put(containerStack(stackDepth - 1), nextChildIndexes(stackDepth - 1) - 1) + } else if (stackDepth >= 2) { + counts.remove(containerStack(stackDepth - 1)) + counts.put(containerStack(stackDepth - 2), nextChildIndexes(stackDepth - 2) - 1) + } + } + counts + } + + // The unparse is done with a node, so build is no longer ahead of it by one. + def decrementLead(): Unit = cursor.buildState.decrementLead() + + // Once a node's end event is out, the node is no longer needed in its + // parent, whose frame is now on top and whose next index is just past it. + private def freeEndedChild(): Unit = { + val parent = stackDepth - 1 + containerStack(parent).freeChildIfNoLongerNeeded( + nextChildIndexes(parent) - 1, + releaseUnneededInfoset + ) + } + + private def isHidden(node: DINode, kind: NodeKind): Boolean = { + if (kind eq NodeKind.Array) { + node.numChildren > 0 && node.isHidden + } else { + node.isHidden + } + } + + private def emitEnd(endedNode: DINode, kind: NodeKind): Unit = { + setEvent(endedNode, kind, isStart = false) + popContainer() + freeEndedChild() + } + + private def setEvent(node: DINode, kind: NodeKind, isStart: Boolean): Unit = { + if (kind eq NodeKind.Array) { + inspected.kind = if (isStart) { + InfosetEventKind.StartArray + } else { + InfosetEventKind.EndArray + } + inspected.info = Info(node.erd) + } else { + inspected.kind = if (isStart) { + InfosetEventKind.StartElement + } else { + InfosetEventKind.EndElement + } + inspected.info = Info(node.asInstanceOf[DIElement]) + } + } + + // The child index in each container where the search for its next hidden + // child resumes. Hidden nodes are asked for in the order they appear in the tree. + private lazy val hiddenCursors = new java.util.IdentityHashMap[DINode, Integer](8) + + // The container and index of each hidden node taken for unparsing and not yet freed. + private lazy val handedOutContainers = new MStackOf[DINode](8) + private lazy val handedOutIndexes = MStackOfInt(8) + + def takeExistingHiddenChild(parent: DINode, erd: ElementRuntimeData): DIElement = { + val container = if (erd.isArray) { + parent.child(findNextHiddenIndex(parent, erd, isArray = true)) + } else { + parent + } + val index = findNextHiddenIndex(container, erd, isArray = false) + handedOutContainers.push(container) + handedOutIndexes.push(index) + container.child(index).asInstanceOf[DIElement] + } + + def freeExistingHiddenChild(): Unit = { + val container = handedOutContainers.pop + container.freeChildIfNoLongerNeeded(handedOutIndexes.pop(), releaseUnneededInfoset) + } + + // The cursor stays on a hidden array while its occurrences are handed out one + // at a time, and moves past the array when a different hidden node is asked for. + private def findNextHiddenIndex( + container: DINode, + erd: ElementRuntimeData, + isArray: Boolean + ): Int = { + val cursor = hiddenCursors.get(container) + var index = if (cursor eq null) { + 0 + } else { + cursor.intValue + } + var found = false + while (!found) { + Assert.invariant(childExistsOrFinal(container, index)) + // A hidden node is always ready; a node already freed is not hidden. + val child = container.child(index) + if ((child ne null) && isHidden(child, kindOf(child)) && (child.erd eq erd)) { + found = true + } else { + index += 1 + } + } + val nextCursor = if (isArray) { + index + } else { + index + 1 + } + hiddenCursors.put(container, nextCursor) + index + } + + override def advance: Boolean = { + if (fill()) { + val consumed = inspected + inspected = advanced + advanced = consumed + hasInspectedEvent = false + true + } else { + false + } + } + + override def advanceAccessor: InfosetAccessor = advanced + + override def inspect: Boolean = fill() + + override def inspectAccessor: InfosetAccessor = inspected + + override def inspectOrError: InfosetAccessor = { + if (inspect) { + inspectAccessor + } else { + Assert.invariantFailed( + "An InfosetEvent was required for unparsing, but no InfosetEvent was available." + ) + } + } + + override def advanceOrError: InfosetAccessor = { + if (advance) { + advanceAccessor + } else { + Assert.invariantFailed( + "An InfosetEvent was required for unparsing, but no InfosetEvent was available." + ) + } + } + + override def isInspectArrayEnd: Boolean = inspect && inspected.isEnd && inspected.isArray + + override def pushTRD(trd: TermRuntimeData): Unit = trdStack.push(trd) + + override def maybeTopTRD(): Maybe[TermRuntimeData] = { + if (trdStack.isEmpty) { + Nope + } else { + One(trdStack.top) + } + } + + override def popTRD(trd: TermRuntimeData): TermRuntimeData = { + val popped = trdStack.pop + if (popped ne trd) { + Assert.invariantFailed("TRDs do not match. Expected: " + trd + " got " + popped) + } + popped + } +} + +/** + * What a node the walk visits is, which decides the events it gets. A complex + * node is a complex element other than the document. + */ +private[unparsers] enum NodeKind { + case Simple, Complex, Array, Document +} + +/** + * Thrown when the unparse needs a child that does not exist after build has + * finished. Caught only by the top-level driver, which still runs its normal + * finalization (the genuine, diagnostic-producing final suspension drain) + * rather than treating this as the final outcome itself. + */ +final class ChildNotBuiltException extends Exception diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala index d60290ce79..67d142e14b 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala @@ -54,6 +54,7 @@ import org.apache.daffodil.runtime1.infoset.InfosetInputter import org.apache.daffodil.runtime1.processors.DataLoc import org.apache.daffodil.runtime1.processors.DataProcessor import org.apache.daffodil.runtime1.processors.DelimiterStackUnparseNode +import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.EscapeSchemeUnparserHelper import org.apache.daffodil.runtime1.processors.Failure import org.apache.daffodil.runtime1.processors.NonTermRuntimeData @@ -75,9 +76,11 @@ abstract class UState( diagnosticsArg: Seq[api.Diagnostic], dataProcArg: Maybe[DataProcessor], tunable: DaffodilTunables, - areDebugging: Boolean + val areDebugging: Boolean, + eventState: InfosetEventState, + delimiterEscapePosition: DelimiterEscapePositionState ) extends ParseOrUnparseState(vbox, diagnosticsArg, dataProcArg, tunable) - with Cursor[InfosetAccessor] + with InfosetTreeState with ThrowsSDE with SavesErrorsAndWarnings { @@ -110,18 +113,18 @@ abstract class UState( /** * Push onto the dynamic TRD context stack */ - def pushTRD(trd: TermRuntimeData): Unit + final def pushTRD(trd: TermRuntimeData): Unit = eventState.pushTRD(trd) /** * Returns the top of the stack if it exists. No state change to stack contents. */ - def maybeTopTRD(): Maybe[TermRuntimeData] + final def maybeTopTRD(): Maybe[TermRuntimeData] = eventState.maybeTopTRD() /** * Pop the dynamic TRD context stack. The popped TRD should be the same as the argument rd. * The popped TRD is returned. */ - def popTRD(trd: TermRuntimeData): TermRuntimeData + final def popTRD(trd: TermRuntimeData): TermRuntimeData = eventState.popTRD(trd) override def toString = { val elt = @@ -140,30 +143,63 @@ abstract class UState( def currentInfosetNode: DINode def currentInfosetNodeMaybe: Maybe[DINode] - def escapeSchemeEVCache: MStackOfMaybe[EscapeSchemeUnparserHelper] - - def withUnparserDataInputStream: LocalStack[StringDataInputStreamForUnparse] - def withByteArrayOutputStream - : LocalStack[(ByteArrayOutputStream, DirectOrBufferedDataOutputStream)] - - def allTerminatingMarkup: List[DFADelimiter] - def localDelimiters: DelimiterStackUnparseNode - def pushDelimiters(node: DelimiterStackUnparseNode): Unit - def popDelimiters(): Unit + final def escapeSchemeEVCache: MStackOfMaybe[EscapeSchemeUnparserHelper] = + delimiterEscapePosition.escapeSchemeEVCache + + final def withUnparserDataInputStream: LocalStack[StringDataInputStreamForUnparse] = + delimiterEscapePosition.withUnparserDataInputStream + final def withByteArrayOutputStream + : LocalStack[(ByteArrayOutputStream, DirectOrBufferedDataOutputStream)] = + delimiterEscapePosition.withByteArrayOutputStream + + final def allTerminatingMarkup: List[DFADelimiter] = + delimiterEscapePosition.allTerminatingMarkup + final def localDelimiters: DelimiterStackUnparseNode = delimiterEscapePosition.localDelimiters + final def pushDelimiters(node: DelimiterStackUnparseNode): Unit = + delimiterEscapePosition.pushDelimiters(node) + final def popDelimiters(): Unit = delimiterEscapePosition.popDelimiters() + + final def childIndexStack: MStackOfLong = delimiterEscapePosition.childIndexStack + final def moveOverOneElementChildOnly(): Unit = + delimiterEscapePosition.moveOverOneElementChildOnly() + final override def childPos: Long = delimiterEscapePosition.childPos def currentInfosetNodeStack: MStackOfMaybe[DINode] def arrayIterationIndexStack: MStackOfLong def occursIndexStack: MStackOfLong - def childIndexStack: MStackOfLong def groupIndexStack: MStackOfLong def moveOverOneArrayIterationIndexOnly(): Unit def moveOverOneOccursIndexOnly(): Unit def moveOverOneGroupIndexOnly(): Unit - def moveOverOneElementChildOnly(): Unit - def inspectOrError: InfosetAccessor - def advanceOrError: InfosetAccessor - def isInspectArrayEnd: Boolean + // A dfdl:occursIndex() expression in an occurrence's own content reads + // arrayIterationIndexStack/occursIndexStack's top, which must track the + // occurrence currently being unparsed; the build pass pushes the same pair + // directly around each array/optional occurrence group. + final def pushOccurrenceIndices(): Unit = { + arrayIterationIndexStack.push(1L) + occursIndexStack.push(1L) + } + final def popOccurrenceIndices(): Unit = { + arrayIterationIndexStack.pop() + occursIndexStack.pop() + } + + final override def advance: Boolean = eventState.advance + final override def advanceAccessor: InfosetAccessor = eventState.advanceAccessor + final override def inspect: Boolean = eventState.inspect + final override def inspectAccessor: InfosetAccessor = eventState.inspectAccessor + // $COVERAGE-OFF$ // unused, but necessary to meet requirements of Cursor[T] + override def fini(): Unit = Assert.usageError("Not to be used on UState") + // $COVERAGE-ON$ + + /** + * Use these so if there isn't an event we get a clean diagnostic message saying + * that is what has gone wrong. + */ + final def inspectOrError: InfosetAccessor = eventState.inspectOrError + final def advanceOrError: InfosetAccessor = eventState.advanceOrError + final def isInspectArrayEnd: Boolean = eventState.isInspectArrayEnd override def dataStream = Maybe(getDataOutputStream) @@ -389,7 +425,7 @@ abstract class UState( case m: UStateMain => m.cloneForSuspension(dos) case _ => Assert.invariantFailed( - "State must be a UStateMain when splitting for bit order change" + "State must be UStateMain when splitting for bit order change" ) } @@ -404,9 +440,301 @@ abstract class UState( def documentElement: DIDocument - final val releaseUnneededInfoset: Boolean = !areDebugging && tunable.releaseUnneededInfoset + // Whether a node is freed once it is done with. + private[unparsers] def releaseUnneededInfoset: Boolean + + final def freeChildIfNoLongerNeeded(parent: DINode, index: Int): Unit = + parent.freeChildIfNoLongerNeeded(index, releaseUnneededInfoset) def delimitedParseResult = Nope + + // Retries the suspensions that can be resolved now. + def runSuspensions(): Unit + +} + +/** + * The state of the infoset tree as unparsing builds it: the event cursor, the + * TRD and node stacks, and the position within the current group, array and + * occurrence. Both an event-driven unparse (UState) and the build ahead build + * (InfosetBuildState) walk the infoset through it; the build has no output + * stream, variables or debugger state, so it is not a UState. + */ +trait InfosetTreeState extends Cursor[InfosetAccessor] { + def tunable: DaffodilTunables + + // How the infoset's nodes are made and kept as unparsing consumes events. + // InfosetFromEvents does it for the events of an inputter; a state that + // unparses a tree that was already built overrides these. + + // The node of an element in a hidden group, which has no events. + def getHiddenElement(erd: ElementRuntimeData): DIElement + + // The node of an outputValueCalc element whose start event was just consumed. + def getOvcElement(startEvent: InfosetAccessor, erd: ElementRuntimeData): DIElement + + // Adds a node whose start was just reached to the infoset. + def attachElement(newElem: DIElement): Unit + + // Finishes a node whose end was just reached. + def finishElement(cur: DINode, erd: ElementRuntimeData): Unit + def finishOvcElement(cur: DINode): Unit + + def inspectOrError: InfosetAccessor + def advanceOrError: InfosetAccessor + def isInspectArrayEnd: Boolean + + def pushTRD(trd: TermRuntimeData): Unit + def maybeTopTRD(): Maybe[TermRuntimeData] + def popTRD(trd: TermRuntimeData): TermRuntimeData + + def currentInfosetNode: DINode + def currentInfosetNodeMaybe: Maybe[DINode] + def currentInfosetNodeStack: MStackOfMaybe[DINode] + def documentElement: DIDocument + + def arrayIterationIndexStack: MStackOfLong + def occursIndexStack: MStackOfLong + def groupIndexStack: MStackOfLong + def moveOverOneArrayIterationIndexOnly(): Unit + def moveOverOneOccursIndexOnly(): Unit + def moveOverOneGroupIndexOnly(): Unit + def arrayIterationPos: Long + def occursPos: Long + def groupPos: Long + + def withinHiddenNest: Boolean + def incrementHiddenDef(): Unit + def decrementHiddenDef(): Unit + + def moveOverOneElementChildOnly(): Unit + + // unparseBegin and unparseEnd free children, and build runs them too. Build + // runs ahead of the unparse, which may already have freed the node, so build + // must not free: it would find a null slot or free a node the unparse has + // not read yet. + def freeChildIfNoLongerNeeded(parent: DINode, index: Int): Unit +} + +/** + * The part of a UState that consumes infoset events from an InfosetInputter: + * the event cursor and the TRD stack. Only the UStates that read the + * infoset events hold a real one. + */ +trait InfosetEventState { + def advance: Boolean + def advanceAccessor: InfosetAccessor + def inspect: Boolean + def inspectAccessor: InfosetAccessor + def inspectOrError: InfosetAccessor + def advanceOrError: InfosetAccessor + def isInspectArrayEnd: Boolean + def pushTRD(trd: TermRuntimeData): Unit + def maybeTopTRD(): Maybe[TermRuntimeData] + def popTRD(trd: TermRuntimeData): TermRuntimeData +} + +/** + * Events read from an InfosetInputter. The purpose names what the caller is + * doing, for the diagnostic when an event is required but none is available. + */ +final class InputterEventState(inputter: InfosetInputter, purpose: String) + extends InfosetEventState { + + override def advance: Boolean = inputter.advance + override def advanceAccessor: InfosetAccessor = inputter.advanceAccessor + override def inspect: Boolean = inputter.inspect + override def inspectAccessor: InfosetAccessor = inputter.inspectAccessor + + override def inspectOrError: InfosetAccessor = { + if (inspect) { + inspectAccessor + } else { + Assert.invariantFailed( + "An InfosetEvent was required for " + purpose + ", but no InfosetEvent was available." + ) + } + } + + override def advanceOrError: InfosetAccessor = { + if (advance) { + advanceAccessor + } else { + Assert.invariantFailed( + "An InfosetEvent was required for " + purpose + ", but no InfosetEvent was available." + ) + } + } + + override def isInspectArrayEnd: Boolean = { + if (!inspect) { + false + } else { + inspectAccessor match { + case e if e.isEnd && e.isArray => true + case _ => false + } + } + } + + override def pushTRD(trd: TermRuntimeData): Unit = inputter.pushTRD(trd) + override def maybeTopTRD(): Maybe[TermRuntimeData] = inputter.maybeTopTRD() + override def popTRD(trd: TermRuntimeData): TermRuntimeData = { + val poppedTRD = inputter.popTRD() + if (poppedTRD ne trd) { + Assert.invariantFailed("TRDs do not match. Expected: " + trd + " got " + poppedTRD) + } + poppedTRD + } +} + +/** + * For a UState that never reads infoset events: a clone made to resume a + * suspension. + */ +object NoInfosetEventState extends InfosetEventState { + private def die = + Assert.invariantFailed("Function should never be needed in UStateForSuspension") + + override def advance: Boolean = die + override def advanceAccessor: InfosetAccessor = die + override def inspect: Boolean = die + override def inspectAccessor: InfosetAccessor = die + override def inspectOrError: InfosetAccessor = die + override def advanceOrError: InfosetAccessor = die + override def isInspectArrayEnd: Boolean = die + override def pushTRD(trd: TermRuntimeData): Unit = die + override def maybeTopTRD(): Maybe[TermRuntimeData] = die + override def popTRD(trd: TermRuntimeData): TermRuntimeData = die +} + +/** + * The part of a UState that only emitting delimited, escaped text uses: the + * delimiter stack, the escape scheme cache, the scratch buffers for measuring + * and escaping text, and the position within the current sequence or choice. + */ +trait DelimiterEscapePositionState { + def escapeSchemeEVCache: MStackOfMaybe[EscapeSchemeUnparserHelper] + def withUnparserDataInputStream: LocalStack[StringDataInputStreamForUnparse] + def withByteArrayOutputStream + : LocalStack[(ByteArrayOutputStream, DirectOrBufferedDataOutputStream)] + def allTerminatingMarkup: List[DFADelimiter] + def localDelimiters: DelimiterStackUnparseNode + def pushDelimiters(node: DelimiterStackUnparseNode): Unit + def popDelimiters(): Unit + def childIndexStack: MStackOfLong + def moveOverOneElementChildOnly(): Unit + def childPos: Long +} + +final class MainDelimiterEscapePositionState(tunable: DaffodilTunables) + extends DelimiterEscapePositionState { + + override lazy val escapeSchemeEVCache = new MStackOfMaybe[EscapeSchemeUnparserHelper](8) + + override lazy val withUnparserDataInputStream = + new LocalStack[StringDataInputStreamForUnparse](new StringDataInputStreamForUnparse) + + override lazy val withByteArrayOutputStream = + new LocalStack[(ByteArrayOutputStream, DirectOrBufferedDataOutputStream)]( + { + val baos = + new ByteArrayOutputStream() // TODO: PERFORMANCE: Allocates new object. Can reuse one from an onStack/pool via reset() + val dos = DirectOrBufferedDataOutputStream( + baos, + null, + false, + tunable.outputStreamChunkSizeInBytes, + tunable.maxByteArrayOutputStreamBufferSizeInBytes, + tunable.tempFilePath + ) + (baos, dos) + }, + pair => + pair match { + case (baos, dos) => + baos.reset() + dos.resetAllBitPos() + } + ) + + private val delimiterStack = new MStackOf[DelimiterStackUnparseNode]() + override def pushDelimiters(node: DelimiterStackUnparseNode): Unit = delimiterStack.push(node) + override def popDelimiters(): Unit = delimiterStack.pop + override def localDelimiters: DelimiterStackUnparseNode = delimiterStack.top + override def allTerminatingMarkup: List[DFADelimiter] = { + delimiterStack.iterator.flatMap { dnode => + dnode.separator ++ dnode.terminator + }.toList + } + + // Sequence and choice unparsers read it to find their current child; build + // tracks position in its own frames instead. + override val childIndexStack = MStackOfLong(16) + childIndexStack.push(1L) + override def moveOverOneElementChildOnly(): Unit = + childIndexStack.setTop(childIndexStack.top + 1) + override def childPos: Long = childIndexStack.top + + /** + * The surface for a clone that resumes a suspension: it needs only the + * current escape scheme and delimiters, and shares the scratch buffers. + */ + def cloneForSuspension(): DelimiterEscapePositionState = { + val es = + if (!escapeSchemeEVCache.isEmpty) { + // If there are any escape schemes, clone the whole MStack, since the + // escape scheme cache logic requires one (only the top is really + // needed, but changing the cache access isn't trivial). Sized to the + // source's depth: nothing pushes onto the clone afterward. + val esClone = new MStackOfMaybe[EscapeSchemeUnparserHelper](escapeSchemeEVCache.length) + esClone.copyFrom(escapeSchemeEVCache) + Maybe(esClone) + } else { + Nope + } + val ds = + if (!delimiterStack.isEmpty) { + // If there are any delimiters, clone them all since they may be + // needed for escaping. Sized to the source's depth: push and pop + // both die on this clone, so it never grows past that depth. + val dsClone = new MStackOf[DelimiterStackUnparseNode](delimiterStack.length) + dsClone.copyFrom(delimiterStack) + Maybe(dsClone) + } else { + Nope + } + new SuspendedDelimiterEscapePositionState(this, es, ds) + } +} + +final private class SuspendedDelimiterEscapePositionState( + main: MainDelimiterEscapePositionState, + escapeSchemeEVCacheMaybe: Maybe[MStackOfMaybe[EscapeSchemeUnparserHelper]], + delimiterStackMaybe: Maybe[MStackOf[DelimiterStackUnparseNode]] +) extends DelimiterEscapePositionState { + + private def die = + Assert.invariantFailed("Function should never be needed in UStateForSuspension") + + override def escapeSchemeEVCache: MStackOfMaybe[EscapeSchemeUnparserHelper] = + escapeSchemeEVCacheMaybe.get + override def withUnparserDataInputStream = main.withUnparserDataInputStream + override def withByteArrayOutputStream = main.withByteArrayOutputStream + + override def localDelimiters: DelimiterStackUnparseNode = delimiterStackMaybe.get.top + override def allTerminatingMarkup: List[DFADelimiter] = { + delimiterStackMaybe.get.iterator.flatMap { dnode => + dnode.separator ++ dnode.terminator + }.toList + } + override def pushDelimiters(node: DelimiterStackUnparseNode): Unit = die + override def popDelimiters(): Unit = die + + override def childIndexStack: MStackOfLong = die + override def moveOverOneElementChildOnly(): Unit = die + // Called when copying state during debugging. + override def childPos: Long = 0L } /** @@ -424,11 +752,22 @@ final class UStateForSuspension( override val currentInfosetNode: DINode, arrayIterationIndex: Long, occursIndex: Long, - escapeSchemeEVCacheMaybe: Maybe[MStackOfMaybe[EscapeSchemeUnparserHelper]], - delimiterStackMaybe: Maybe[MStackOf[DelimiterStackUnparseNode]], + delimiterEscapePosition: DelimiterEscapePositionState, tunable: DaffodilTunables, areDebugging: Boolean -) extends UState(vbox, mainUState.diagnostics, mainUState.dataProc, tunable, areDebugging) { +) extends UState( + vbox, + mainUState.diagnostics, + mainUState.dataProc, + tunable, + areDebugging, + NoInfosetEventState, + delimiterEscapePosition + ) { + + // Follows the main state this suspension state was cloned from. + override private[unparsers] def releaseUnneededInfoset: Boolean = + mainUState.releaseUnneededInfoset _dataOutputStream = dataOutputStream dState.setMode(UnparserBlocking) @@ -444,51 +783,26 @@ final class UStateForSuspension( override def suspensions = mainUState.suspensions - // override def charBufferDataOutputStream = mainUState.charBufferDataOutputStream - override def withUnparserDataInputStream = mainUState.withUnparserDataInputStream - override def withByteArrayOutputStream = mainUState.withByteArrayOutputStream - // $COVERAGE-OFF$ - override def advance: Boolean = die - override def advanceAccessor: InfosetAccessor = die - override def inspect: Boolean = die - override def inspectAccessor: InfosetAccessor = die - override def fini(): Unit = die - override def inspectOrError = die - override def advanceOrError = die - override def isInspectArrayEnd = die override def currentInfosetNodeStack = die + override def getHiddenElement(erd: ElementRuntimeData) = die + override def getOvcElement(startEvent: InfosetAccessor, erd: ElementRuntimeData) = die + override def attachElement(newElem: DIElement) = die + override def finishElement(cur: DINode, erd: ElementRuntimeData) = die + override def finishOvcElement(cur: DINode) = die + override def runSuspensions() = die override def arrayIterationIndexStack = die override def moveOverOneArrayIterationIndexOnly() = die override def occursIndexStack = die override def moveOverOneOccursIndexOnly() = die override def groupIndexStack = die override def moveOverOneGroupIndexOnly() = die - override def childIndexStack = die - override def moveOverOneElementChildOnly() = die - override def pushDelimiters(node: DelimiterStackUnparseNode) = die - override def popDelimiters() = die // $COVERAGE-ON$ override def groupPos = 0 // was die, but this is called when copying state during debugging override def currentInfosetNodeMaybe = Maybe(currentInfosetNode) override def arrayIterationPos = arrayIterationIndex override def occursPos = occursIndex - override def childPos = 0 // was die, but this is called when copying state during debugging. - - override def localDelimiters = delimiterStackMaybe.get.top - override def allTerminatingMarkup = { - delimiterStackMaybe.get.iterator.flatMap { dnode => - dnode.separator ++ dnode.terminator - }.toList - } - - override def escapeSchemeEVCache: MStackOfMaybe[EscapeSchemeUnparserHelper] = - escapeSchemeEVCacheMaybe.get - - override def pushTRD(trd: TermRuntimeData): Unit = die - override def maybeTopTRD() = die - override def popTRD(trd: TermRuntimeData): TermRuntimeData = die override def documentElement = mainUState.documentElement @@ -503,15 +817,60 @@ final class UStateForSuspension( } } -final class UStateMain private ( +/** + * Stack-backed array-iteration/occurs/group/child index tracking, shared + * by UStateMain and InfosetBuildState: each stack starts seeded with 1L, and + * moveOverOne*Only bumps its top by one as navigation advances. + * UStateForSuspension needs none of this (it stubs the stacks to die and + * tracks arrayIterationPos/occursPos as plain frozen Longs instead), so + * this lives in a mixin rather than directly on UState. + */ +trait TraversalIndexStacks { self: InfosetTreeState => + override val arrayIterationIndexStack = MStackOfLong(16) + arrayIterationIndexStack.push(1L) + override def moveOverOneArrayIterationIndexOnly(): Unit = + arrayIterationIndexStack.setTop(arrayIterationIndexStack.top + 1) + override def arrayIterationPos = arrayIterationIndexStack.top + + override val occursIndexStack = MStackOfLong(16) + occursIndexStack.push(1L) + override def moveOverOneOccursIndexOnly(): Unit = + occursIndexStack.setTop(occursIndexStack.top + 1) + override def occursPos = occursIndexStack.top + + override val groupIndexStack = MStackOfLong() + groupIndexStack.push(1L) + override def moveOverOneGroupIndexOnly(): Unit = + groupIndexStack.setTop(groupIndexStack.top + 1) + override def groupPos = groupIndexStack.top +} + +class UStateMain private[unparsers] ( private val inputter: InfosetInputter, outStream: java.io.OutputStream, vbox: VariableBox, diagnosticsArg: Seq[api.Diagnostic], dataProcArg: DataProcessor, tunable: DaffodilTunables, - areDebugging: Boolean -) extends UState(vbox, diagnosticsArg, One(dataProcArg), tunable, areDebugging) { + areDebugging: Boolean, + mainDelimiterEscapePosition: MainDelimiterEscapePositionState, + eventState: InfosetEventState +) extends UState( + vbox, + diagnosticsArg, + One(dataProcArg), + tunable, + areDebugging, + eventState, + mainDelimiterEscapePosition + ) + with TraversalIndexStacks + with InfosetFromEvents { + + override def runSuspensions(): Unit = evalSuspensions(isFinal = false) + + private[unparsers] final val releaseUnneededInfoset: Boolean = + !areDebugging && tunable.releaseUnneededInfoset dState.setMode(UnparserBlocking) @@ -522,7 +881,8 @@ final class UStateMain private ( diagnosticsArg: Seq[api.Diagnostic], dataProcArg: DataProcessor, tunable: DaffodilTunables, - areDebugging: Boolean + areDebugging: Boolean, + eventState: InfosetEventState ) = this( inputter, @@ -531,7 +891,9 @@ final class UStateMain private ( diagnosticsArg, dataProcArg, tunable, - areDebugging + areDebugging, + new MainDelimiterEscapePositionState(tunable), + eventState ) setDataOutputStream({ @@ -547,30 +909,6 @@ final class UStateMain private ( }) def cloneForSuspension(suspendedDOS: DirectOrBufferedDataOutputStream): UState = { - val es = - if (!escapeSchemeEVCache.isEmpty) { - // If there are any escape schemes, then we need to clone the whole - // MStack, since the escape scheme cache logic requires an MStack. We - // reallyjust need the top for cloning for suspensions, but that - // requires changes to how the escape schema cache is accessed, which - // isn't a trivial change. - val esClone = new MStackOfMaybe[EscapeSchemeUnparserHelper](escapeSchemeEVCache.length) - esClone.copyFrom(escapeSchemeEVCache) - Maybe(esClone) - } else { - Nope - } - val ds = - if (!delimiterStack.isEmpty) { - // If there are any delimiters, then we need to clone them all since - // they may be needed for escaping - val dsClone = new MStackOf[DelimiterStackUnparseNode](delimiterStack.length) - dsClone.copyFrom(delimiterStack) - Maybe(dsClone) - } else { - Nope - } - val clone = new UStateForSuspension( this, suspendedDOS, @@ -578,8 +916,7 @@ final class UStateMain private ( currentInfosetNodeStack.top.get, // only need the to of the stack, not the whole thing arrayIterationIndexStack.top, // only need the top of the stack, not the whole thing occursIndexStack.top, - es, - ds, + mainDelimiterEscapePosition.cloneForSuspension(), tunable, areDebugging ) @@ -589,72 +926,6 @@ final class UStateMain private ( clone } - override lazy val withUnparserDataInputStream = - new LocalStack[StringDataInputStreamForUnparse](new StringDataInputStreamForUnparse) - override lazy val withByteArrayOutputStream = - new LocalStack[(ByteArrayOutputStream, DirectOrBufferedDataOutputStream)]( - { - val baos = - new ByteArrayOutputStream() // TODO: PERFORMANCE: Allocates new object. Can reuse one from an onStack/pool via reset() - val dos = DirectOrBufferedDataOutputStream( - baos, - null, - false, - tunable.outputStreamChunkSizeInBytes, - tunable.maxByteArrayOutputStreamBufferSizeInBytes, - tunable.tempFilePath - ) - (baos, dos) - }, - pair => - pair match { - case (baos, dos) => - baos.reset() - dos.resetAllBitPos() - } - ) - - override def advance: Boolean = inputter.advance - override def advanceAccessor: InfosetAccessor = inputter.advanceAccessor - override def inspect: Boolean = inputter.inspect - override def inspectAccessor: InfosetAccessor = inputter.inspectAccessor - // $COVERAGE-OFF$ // unused, but necessary to meet requirements of Cursor[T] - override def fini() = Assert.usageError("Not to be used on UState") - // $COVERAGE-ON$ - /** - * Use this so if there isn't an event we get a clean diagnostic message saying - * that is what has gone wrong. - */ - override def inspectOrError = { - if (inspect) - inspectAccessor - else - Assert.invariantFailed( - "An InfosetEvent was required for unparsing, but no InfosetEvent was available." - ) - } - - override def advanceOrError = { - if (advance) - advanceAccessor - else - Assert.invariantFailed( - "An InfosetEvent was required for unparsing, but no InfosetEvent was available." - ) - } - - override def isInspectArrayEnd = { - if (!inspect) false - else { - val p = inspectAccessor - val res = p match { - case e if e.isEnd && e.isArray => true - case _ => false - } - res - } - } - def currentInfosetNode: DINode = if (currentInfosetNodeMaybe.isEmpty) null else currentInfosetNodeMaybe.get @@ -665,41 +936,6 @@ final class UStateMain private ( override val currentInfosetNodeStack = new MStackOfMaybe[DINode](16) - override val arrayIterationIndexStack = MStackOfLong(16) - arrayIterationIndexStack.push(1L) - override def moveOverOneArrayIterationIndexOnly() = - arrayIterationIndexStack.setTop(arrayIterationIndexStack.top + 1) - override def arrayIterationPos = arrayIterationIndexStack.top - - override val occursIndexStack = MStackOfLong(16) - occursIndexStack.push(1L) - override def moveOverOneOccursIndexOnly() = occursIndexStack.setTop(occursIndexStack.top + 1) - override def occursPos = occursIndexStack.top - - override val groupIndexStack = MStackOfLong() - groupIndexStack.push(1L) - override def moveOverOneGroupIndexOnly() = groupIndexStack.setTop(groupIndexStack.top + 1) - override def groupPos = groupIndexStack.top - - // TODO: it doesn't look anything is actually reading the value of childindex - // stack. Can we get rid of it? - override val childIndexStack = MStackOfLong(16) - childIndexStack.push(1L) - override def moveOverOneElementChildOnly() = childIndexStack.setTop(childIndexStack.top + 1) - override def childPos = childIndexStack.top - - override lazy val escapeSchemeEVCache = new MStackOfMaybe[EscapeSchemeUnparserHelper](8) - - val delimiterStack = new MStackOf[DelimiterStackUnparseNode]() - override def pushDelimiters(node: DelimiterStackUnparseNode) = delimiterStack.push(node) - override def popDelimiters() = delimiterStack.pop - override def localDelimiters = delimiterStack.top - override def allTerminatingMarkup = { - delimiterStack.iterator.flatMap { dnode => - dnode.separator ++ dnode.terminator - }.toList - } - /** * For outputValueCalc we accumulate the suspendables here. * @@ -721,19 +957,6 @@ final class UStateMain private ( def suspensions = suspensionTracker.suspensions - final override def pushTRD(trd: TermRuntimeData) = - inputter.pushTRD(trd) - - final override def maybeTopTRD(): Maybe[TermRuntimeData] = - inputter.maybeTopTRD() - - final override def popTRD(trd: TermRuntimeData) = { - val poppedTRD = inputter.popTRD() - if (poppedTRD ne trd) - Assert.invariantFailed("TRDs do not match. Expected: " + trd + " got " + poppedTRD) - poppedTRD - } - final override def documentElement = inputter.documentElement override def toString = { @@ -768,8 +991,33 @@ object UState { diagnostics, dataProc.asInstanceOf[DataProcessor], dataProc.tunables, - areDebugging + areDebugging, + new InputterEventState(inputter, "unparsing") ) newState } + + /** + * For the unparse of a tree that is being built, which reads it as events. + * The inputter still owns the infoset document the events refer to. + */ + def createInitialUStateForBuildAhead( + outStream: java.io.OutputStream, + dataProc: DFDL.DataProcessor, + inputter: InfosetInputter, + areDebugging: Boolean, + treeEvents: TreeEventState + ): UStateMainForBuildAhead = { + val variables = dataProc.variableMap.copy() + new UStateMainForBuildAhead( + inputter, + outStream, + variables, + Nil, + dataProc.asInstanceOf[DataProcessor], + dataProc.tunables, + areDebugging, + treeEvents + ) + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UStateMainForBuildAhead.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UStateMainForBuildAhead.scala new file mode 100644 index 0000000000..ed97fbdadb --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UStateMainForBuildAhead.scala @@ -0,0 +1,54 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.lib.iapi.DaffodilTunables +import org.apache.daffodil.lib.iapi as api +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.infoset.InfosetInputter +import org.apache.daffodil.runtime1.processors.DataProcessor +import org.apache.daffodil.runtime1.processors.VariableMap + +/** + * The UState for unparsing a tree that build made ahead of it, by reading + * the tree as events and taking each node from the tree. + */ +final class UStateMainForBuildAhead private[unparsers] ( + inputter: InfosetInputter, + outStream: java.io.OutputStream, + vmap: VariableMap, + diagnosticsArg: Seq[api.Diagnostic], + dataProcArg: DataProcessor, + tunable: DaffodilTunables, + areDebugging: Boolean, + override protected val treeEvents: TreeEventState +) extends UStateMain( + inputter, + outStream, + vmap, + diagnosticsArg, + dataProcArg, + tunable, + areDebugging, + treeEvents + ) + with InfosetFromTree { + + // How far the unparse of the built tree has reached, for the debugger. + def reachedChildCounts(): java.util.Map[DINode, Integer] = treeEvents.reachedChildCounts() +} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala index 592544ff0d..7c7f750350 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala @@ -79,6 +79,11 @@ sealed trait Unparser extends Processor { UnparseError(One(context.schemaFileLocation), One(ustate.currentLocation), s, args*) } + // Code shared with build has no data location to report. + def UE(state: InfosetTreeState, s: String, args: Any*) = { + UnparseError(One(context.schemaFileLocation), Nope, s, args*) + } + } /** diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala index 9cfff9e970..040bea2eb2 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala @@ -23,17 +23,26 @@ import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.lib.util.MaybeInt import org.apache.daffodil.lib.util.ProperlySerializableMap.* +import org.apache.daffodil.lib.xml.NamedQName import org.apache.daffodil.runtime1.infoset.* import org.apache.daffodil.runtime1.processors.* import org.apache.daffodil.runtime1.processors.unparsers.* +/** + * Maps the name of the element that starts a branch to the unparser for that + * branch. An end event never starts a branch, so it always takes the default. + */ case class ChoiceBranchMap( - lookupTable: ProperlySerializableMap[ChoiceBranchEvent, Unparser], + lookupTable: ProperlySerializableMap[NamedQName, Unparser], unmappedDefault: Option[Unparser] ) extends Serializable { - def get(cbe: ChoiceBranchEvent): Maybe[Unparser] = { - val fromTable = lookupTable.get(cbe) + def get(event: InfosetAccessor): Maybe[Unparser] = { + val fromTable = if (event.isStart) { + lookupTable.get(event.erd.namedQName) + } else { + null + } val res = if (fromTable != null) One(fromTable) else { @@ -92,28 +101,16 @@ class ChoiceCombinatorUnparser( } else { state.pushTRD(mgrd) val event: InfosetAccessor = state.inspectOrError - val key: ChoiceBranchEvent = event match { - // - // The ChoiceBranchStartEvent(...) is not a case class constructor. It is a - // hash-table lookup for a cached value. This avoids constructing these - // objects over and over again. - // - case e if e.isStart && e.isElement => ChoiceBranchStartEvent(e.erd.namedQName) - case e if e.isEnd && e.isElement => ChoiceBranchEndEvent(e.erd.namedQName) - case e if e.isStart && e.isArray => ChoiceBranchStartEvent(e.erd.namedQName) - case e if e.isEnd && e.isArray => ChoiceBranchEndEvent(e.erd.namedQName) - } - - val maybeChildUnparser = choiceBranchMap.get(key) + val maybeChildUnparser = choiceBranchMap.get(event) if (maybeChildUnparser.isEmpty) { UnparseError( One(mgrd.schemaFileLocation), One(state.currentLocation), "Found next element %s, but expected one of %s.", - key.qname.toExtendedSyntax, + event.erd.namedQName.toExtendedSyntax, choiceBranchMap.keys .map { - _.qname.toExtendedSyntax + _.toExtendedSyntax } .mkString(", ") ) diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala index c1343654ba..101ad34887 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala @@ -23,7 +23,6 @@ import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.lib.util.MaybeULong import org.apache.daffodil.runtime1.dpath.SuspendableExpression import org.apache.daffodil.runtime1.dsom.CompiledExpression -import org.apache.daffodil.runtime1.infoset.DIComplex import org.apache.daffodil.runtime1.infoset.DISimple import org.apache.daffodil.runtime1.infoset.DataValue.DataValuePrimitive import org.apache.daffodil.runtime1.infoset.RetryableException @@ -43,14 +42,14 @@ class ElementUnspecifiedLengthUnparser( eBeforeUnparser: Maybe[Unparser], eUnparser: Maybe[Unparser], eAfterUnparser: Maybe[Unparser], - eReptypeUnparser: Maybe[Unparser] + eRepTypeUnparser: Maybe[Unparser] ) extends ElementUnparserBase( erd, setVarUnparsers, eBeforeUnparser, eUnparser, eAfterUnparser, - eReptypeUnparser + eRepTypeUnparser ) with RegularElementUnparserStartEndStrategy with RepMoveMixin { @@ -60,8 +59,8 @@ class ElementUnspecifiedLengthUnparser( } sealed trait RepMoveMixin { - def move(start: UState): Unit = { - start.childIndexStack.setTop(start.childIndexStack.top + 1) + def move(start: InfosetTreeState): Unit = { + start.moveOverOneElementChildOnly() } } @@ -84,8 +83,8 @@ class ElementUnparserInputValueCalc(erd: ElementRuntimeData, setVarUnparsers: Ar * Move over in the element children, but not in the group. * This avoids separators for this IVC element. */ - override def move(state: UState): Unit = { - state.childIndexStack.setTop(state.childIndexStack.top + 1) + override def move(state: InfosetTreeState): Unit = { + state.moveOverOneElementChildOnly() } } @@ -123,13 +122,13 @@ sealed abstract class ElementUnparserBase( val eBeforeUnparser: Maybe[Unparser], val eUnparser: Maybe[Unparser], val eAfterUnparser: Maybe[Unparser], - val eReptypeUnparser: Maybe[Unparser] + val eRepTypeUnparser: Maybe[Unparser] ) extends CombinatorUnparser(erd) with RepMoveMixin with ElementUnparserStartEndStrategy { final override def childProcessors = - (eBeforeUnparser.toList ++ eUnparser.toList ++ eAfterUnparser.toList ++ eReptypeUnparser.toList ++ setVarUnparsers.toList).toVector + (eBeforeUnparser.toList ++ eUnparser.toList ++ eAfterUnparser.toList ++ eRepTypeUnparser.toList ++ setVarUnparsers.toList).toVector private val name = erd.name @@ -139,7 +138,7 @@ sealed abstract class ElementUnparserBase( "" + (if (eBeforeUnparser.isDefined) eBeforeUnparser.value.toBriefXML(depthLimit - 1) else "") + - (if (eReptypeUnparser.isDefined) eReptypeUnparser.value.toBriefXML(depthLimit - 1) + (if (eRepTypeUnparser.isDefined) eRepTypeUnparser.value.toBriefXML(depthLimit - 1) else "") + (if (eUnparser.isDefined) eUnparser.value.toBriefXML(depthLimit - 1) else "") + (if (eAfterUnparser.isDefined) eAfterUnparser.value.toBriefXML(depthLimit - 1) @@ -172,8 +171,8 @@ sealed abstract class ElementUnparserBase( } protected def runContentUnparser(state: UState): Unit = { - if (eReptypeUnparser.isDefined) { - eReptypeUnparser.get.unparse1(state) + if (eRepTypeUnparser.isDefined) { + eRepTypeUnparser.get.unparse1(state) } else if (eUnparser.isDefined) eUnparser.get.unparse1(state) } @@ -207,6 +206,8 @@ sealed abstract class ElementUnparserBase( unparseEnd(state) + retrySuspensionsAfterEnd(state) + if (state.dataProc.isDefined) state.dataProc.value.endElement(state, this) } @@ -284,14 +285,14 @@ class ElementSpecifiedLengthUnparser( eBeforeUnparser: Maybe[Unparser], eUnparser: Maybe[Unparser], eAfterUnparser: Maybe[Unparser], - eReptypeUnparser: Maybe[Unparser] + eRepTypeUnparser: Maybe[Unparser] ) extends ElementUnparserBase( context, setVarUnparsers, eBeforeUnparser, eUnparser, eAfterUnparser, - eReptypeUnparser + eRepTypeUnparser ) with RegularElementUnparserStartEndStrategy with ElementSpecifiedLengthMixin { @@ -381,16 +382,20 @@ sealed trait ElementUnparserStartEndStrategy { * Consumes the required infoset events and changes context so that the * element's DIElement node is the context element. */ - protected def unparseBegin(state: UState): Unit + def unparseBegin(state: InfosetTreeState): Unit /** * Restores prior context. Consumes end-element event. */ - protected def unparseEnd(state: UState): Unit + def unparseEnd(state: InfosetTreeState): Unit protected def captureRuntimeValuedExpressionValues(ustate: UState): Unit - protected def move(start: UState): Unit + // Only an unparse creates suspensions, so retrying them is not part of + // unparseEnd, which build runs too. + protected def retrySuspensionsAfterEnd(ustate: UState): Unit + + protected def move(start: InfosetTreeState): Unit protected def erd: ElementRuntimeData @@ -403,7 +408,7 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart * Consumes the required infoset events and changes context so that the * element's DIElement node is the context element. */ - final override protected def unparseBegin(state: UState): Unit = { + final override def unparseBegin(state: InfosetTreeState): Unit = { if (erd.isQuasiElement) { // Quasi elements are used for RepType and PrefixedLength, and have no corresponding // events in the infoset inputter. The parent parser will push a DIElement for us to @@ -423,7 +428,7 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart // this indicates that the incoming infoset (as events) doesn't match the schema UnparseError( Nope, - One(state.currentLocation), + Nope, "Expected element start event for %s, but received %s.", erd.namedQName.toExtendedSyntax, event @@ -432,53 +437,11 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart event.info.element } else { Assert.invariant(state.withinHiddenNest) - // Since we never get events for elements in hidden contexts, their infoset elements - // will have never been created. This means we need to manually create them - val hiddenElem = if (erd.isComplexType) new DIComplex(erd) else new DISimple(erd) - hiddenElem.setHidden() - hiddenElem + state.getHiddenElement(erd) } // now add this new elem to the infoset - val parentNodeMaybe = state.currentInfosetNodeMaybe - if (parentNodeMaybe.isDefined) { - val parentComplex = parentNodeMaybe.get.asComplex - Assert.invariant(!parentComplex.isFinal) - if (parentComplex.isNilled) { - // cannot add content to a nilled complex element - UnparseError( - One(erd.schemaFileLocation), - Nope, - "Nilled complex element %s has content from %s", - parentComplex.erd.namedQName.toExtendedSyntax, - newElem.erd.namedQName.toExtendedSyntax - ) - } - - // We are about to add a child to this complex element. Before we do - // that, if the last child added to this complex is a DIArray, and this - // new child isn't part of that array, that implies that the DIArray - // will have no more children added and should be marked as final, and - // we can attempt to free that array. - val lastChildMaybe = parentComplex.maybeLastChild - if (lastChildMaybe.isDefined) { - val lastChild = lastChildMaybe.get - if (lastChild.isArray && (lastChild.erd ne newElem.erd)) { - lastChild.setFinal() - parentComplex.freeChildIfNoLongerNeeded( - parentComplex.numChildren - 1, - state.releaseUnneededInfoset - ) - } - } - - parentComplex.addChild(newElem, state.tunable) - } else { - // We do not yet have an infoset element (this new element is the - // root), so add the infoset node to the DIDocument - val doc = state.documentElement - doc.addChild(newElem, state.tunable) - } + state.attachElement(newElem) // When the infoset events are being advanced, the currentInfosetNodeStack // is pushing and popping to match the events. This provides the proper @@ -490,7 +453,7 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart /** * Restores prior context. Consumes end-element event. */ - final override protected def unparseEnd(state: UState): Unit = { + final override def unparseEnd(state: InfosetTreeState): Unit = { if (erd.isQuasiElement) { // Quasi elements are used for TypeValueCalc, and have no corresponding events in the infoset inputter // The parent parser will handle pushing and poping the Infoset, so we do not need to do anything here. @@ -507,7 +470,7 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart // this indicates that the incoming infoset (as events) doesn't match the schema UnparseError( Nope, - One(state.currentLocation), + Nope, "Expected element end event for %s, but received %s.", erd.namedQName.toExtendedSyntax, event @@ -516,49 +479,15 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart } val cur = state.currentInfosetNodeStack.pop.get - - if (cur.isComplex) { - // We are ending a complex element. If the last child of this complex - // is a DIArray, that implies that the array will have no more children - // and should be marked as isFinal. Normally this happens when we add a - // new sibling after an array in unparseBegin, but in this case there - // is no sibling following the array, so it must be set here. - val lastChild = cur.maybeLastChild - if (lastChild.isDefined && lastChild.get.isArray) { - lastChild.get.setFinal() - cur.freeChildIfNoLongerNeeded(cur.numChildren - 1, state.releaseUnneededInfoset) - } - } - - // cur is finished, mark it as final and free if possible. Note that we - // need the container and not the parent of the current element to free - // it. This way if this element is in an array, we free this element - // from the array. We also do not set hidden IVC elements as - // final--although we allow hidden IVC elements when unparsing, they - // never get a value so we can't set them as final without breaking - // assertions. Nothing can access hidden IVC elements, so this should - // not break anything - if (!state.withinHiddenNest || erd.isRepresented) cur.setFinal() - val curContainer = - if (cur.erd.isArray) cur.diParent.maybeLastChild.get - else cur.diParent - curContainer.freeChildIfNoLongerNeeded( - curContainer.numChildren - 1, - state.releaseUnneededInfoset - ) - - if (state.currentInfosetNodeStack.isEmpty) { - // If there is nothing else on the infoset stack after popping off the - // current infoset node, that means we have finished the root element, - // so mark the DIDocument as final - val doc = state.documentElement - Assert.invariant(!doc.isFinal) - doc.setFinal() - } + state.finishElement(cur, erd) move(state) + } + } - state.asInstanceOf[UStateMain].evalSuspensions(isFinal = false) + final override protected def retrySuspensionsAfterEnd(ustate: UState): Unit = { + if (!erd.isQuasiElement) { + ustate.runSuspensions() } } @@ -573,7 +502,7 @@ trait OVCStartEndStrategy extends ElementUnparserStartEndStrategy { /** * For OVC, the behavior w.r.t. consuming infoset events is different. */ - protected final override def unparseBegin(state: UState): Unit = { + final override def unparseBegin(state: InfosetTreeState): Unit = { val ovcElem = if (!state.withinHiddenNest) { // outputValueCalc elements are optional in the infoset. If the next event @@ -589,11 +518,7 @@ trait OVCStartEndStrategy extends ElementUnparserStartEndStrategy { val endEv = state.advanceOrError // Consume the end event Assert.invariant(endEv.isEnd && endEv.erd == erd) - val e = new DISimple(erd) - // Remove any state that was set by what created this event. Later - // code asserts that OVC elements do not have a value - e.resetValue() - e + state.getOvcElement(startEv, erd) } else { // Event was optional and didn't exist, create a new InfosetElement and add it val e = new DISimple(erd) @@ -601,58 +526,24 @@ trait OVCStartEndStrategy extends ElementUnparserStartEndStrategy { } } else { // Event was hidden and will never exist, create a new InfosetElement and add it - val e = new DISimple(erd) - e.setHidden() - e - } - - // We are about to add a new OVC child to this complex element. Before we - // do that, if the last child added to this complex is a DIArray, that - // implies that the DIArray will have no more children added and should be - // marked as final, and we can attempt to free that array. - val parentNode = state.currentInfosetNode - val parentComplex = parentNode.asComplex - val lastChildMaybe = parentComplex.maybeLastChild - if (lastChildMaybe.isDefined) { - val lastChild = lastChildMaybe.get - if (lastChild.isArray) { - lastChild.setFinal() - parentComplex.freeChildIfNoLongerNeeded( - parentComplex.numChildren - 1, - state.releaseUnneededInfoset - ) + state.getHiddenElement(erd) } - } - parentComplex.addChild(ovcElem, state.tunable) + state.attachElement(ovcElem) state.currentInfosetNodeStack.push(One(ovcElem)) } - protected final override def unparseEnd(state: UState): Unit = { + final override def unparseEnd(state: InfosetTreeState): Unit = { // if an OVC element existed, the start AND end events were consumed in // unparseBegin. No need to advance the cursor here. - - // ovcElem is finished, free it if possible. OVC elements are not allowed in - // arrays, so we can directly get the diParent to get the container DINode val ovcElem = state.currentInfosetNodeStack.pop - val ovcContainer = ovcElem.get.diParent - ovcContainer.freeChildIfNoLongerNeeded( - ovcContainer.numChildren - 1, - state.releaseUnneededInfoset - ) - - if (state.currentInfosetNodeStack.isEmpty) { - // If there is nothing else on the infoset stack after popping off the - // current infoset node, that means we have finished the root element, - // so mark the DIDocument as final - val doc = state.documentElement - Assert.invariant(!doc.isFinal) - doc.setFinal() - } + state.finishOvcElement(ovcElem.get) move(state) } + final override protected def retrySuspensionsAfterEnd(ustate: UState): Unit = {} + // For OVC, or for a target length expression, // // If we delayed evaluating the expressions, some variables might not be read diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala index 3856c0b272..2bd9f19895 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala @@ -95,6 +95,9 @@ class LayeredSequenceUnparser( // layer stack is potentially still needed, so // nothing can be cleaned up at this point. } catch { + // A signal from waiting on build, not a failure of the layer; rewrapping + // it would hide the stall diagnostic. + case e: ChildNotBuiltException => throw e case t: Throwable if (layerDriver ne null) => layerDriver.handleThrowable(t) case t: Throwable => LayerDriver.handleThrowableWithoutLayer(t) // otherwise we have no layer driver, so we were unable to load the layer. diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala index 7c451f1ddb..eb5941fcc7 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala @@ -108,8 +108,8 @@ class RepOrderedSeparatedSequenceChildUnparser( ) extends RepeatingChildUnparser(childUnparser, srd, erd) with Separated { - override def checkArrayPosAgainstMaxOccurs(state: UState) = - state.arrayIterationPos <= maxRepeats(state) + override def checkArrayPosAgainstMaxOccurs(state: InfosetTreeState) = + state.arrayIterationPos <= maxRepeatsConst } class OrderedSeparatedSequenceUnparser( @@ -319,7 +319,7 @@ class OrderedSeparatedSequenceUnparser( state.occursIndexStack.push(1L) val erd = unparser.erd var numOccurrences = 0 - val maxReps = unparser.maxRepeats(state) + val maxReps = unparser.maxRepeatsConst // // The number of occurrances we unparse is always exactly driven // by the number of infoset events for the repeating/optional element. @@ -565,7 +565,7 @@ class OrderedSeparatedSequenceUnparser( unparser.isPositional && unparser.isBoundedMax && (!unparser.isDeclaredLast || !unparser.isPotentiallyTrailing) ) { - val maxReps = unparser.maxRepeats(state) + val maxReps = unparser.maxRepeatsConst while (numOccurrences < maxReps) { unparseOneWithSuppression( unparser, @@ -613,7 +613,7 @@ class OrderedSeparatedSequenceUnparser( Assert.invariant(erd.isRepresented) // arrays/optionals cannot have inputValueCalc var numOccurrences = 0 - val maxReps = unparser.maxRepeats(state) + val maxReps = unparser.maxRepeatsConst Assert.invariant(state.inspect) val ev = state.inspectAccessor diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SequenceChildUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SequenceChildUnparsers.scala index b1e7ac96fb..009c499c53 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SequenceChildUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SequenceChildUnparsers.scala @@ -82,7 +82,7 @@ abstract class RepeatingChildUnparser( * Sets up for the start of an array/optional. For true array, pulls an event, which must be a start-array * event. */ - final def startArrayOrOptional(state: UState): Unit = { + final def startArrayOrOptional(state: InfosetTreeState): Unit = { val ev = state.inspectAccessor if (ev.erd.isArray) { // only pull start array event for a true array, not an optional. @@ -98,6 +98,21 @@ abstract class RepeatingChildUnparser( * Validates array dimensions if validation has been requested. */ final def endArrayOrOptional(currentArrayERD: ElementRuntimeData, state: UState): Unit = { + consumeEndArrayEvent(currentArrayERD, state) + + // State could be Success or Failure here. + endArray(state, state.occursPos - 1) + } + + /** + * The part of ending an array/optional that only consumes the end-array + * event, which build shares with unparse. Array validation needs the full + * UState, so it stays in endArrayOrOptional. + */ + final def consumeEndArrayEvent( + currentArrayERD: ElementRuntimeData, + state: InfosetTreeState + ): Unit = { if (currentArrayERD.isArray) { // only pull end array event for a true array, not an optional val event = state.advanceOrError @@ -110,9 +125,6 @@ abstract class RepeatingChildUnparser( ) } } - - // State could be Success or Failure here. - endArray(state, state.occursPos - 1) } /** @@ -125,7 +137,10 @@ abstract class RepeatingChildUnparser( * If the event is not a start, it must be an endArray for the enclosing complex element, and * the answer is false. */ - final def shouldDoUnparser(unparser: RepeatingChildUnparser, state: UState): Boolean = { + final def shouldDoUnparser( + unparser: RepeatingChildUnparser, + state: InfosetTreeState + ): Boolean = { val childRD = unparser.trd val res = childRD match { @@ -176,7 +191,7 @@ abstract class RepeatingChildUnparser( * bound occurrences with maxOccurs. * */ - def checkArrayPosAgainstMaxOccurs(state: UState): Boolean + def checkArrayPosAgainstMaxOccurs(state: InfosetTreeState): Boolean /** * For OccursCountKind 'implicit', we need to check for arrayPos in range @@ -189,7 +204,7 @@ abstract class RepeatingChildUnparser( * 'unbounded', then maxReps will be Long.MaxValue. */ def checkFinalOccursCountBetweenMinAndMaxOccurs( - state: UState, + state: InfosetTreeState, unparser: RepeatingChildUnparser, numOccurrences: Int, maxReps: Long, @@ -197,7 +212,7 @@ abstract class RepeatingChildUnparser( ): Unit = { import OccursCountKind.* - val minReps = unparser.minRepeats(state) + val minReps = unparser.minRepeatsConst val ev = state.inspectAccessor val erd = unparser.erd diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala index 8b4488a884..2662868611 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala @@ -45,9 +45,9 @@ class RepOrderedUnseparatedSequenceChildUnparser( ) extends RepeatingChildUnparser(childUnparser, srd, erd) with Unseparated { - override def checkArrayPosAgainstMaxOccurs(state: UState): Boolean = { + override def checkArrayPosAgainstMaxOccurs(state: InfosetTreeState): Boolean = { if (ock eq OccursCountKind.Implicit) - state.arrayIterationPos <= maxRepeats(state) + state.arrayIterationPos <= maxRepeatsConst else true } @@ -105,7 +105,7 @@ class OrderedUnseparatedSequenceUnparser( state.occursIndexStack.push(1L) val erd = unparser.erd var numOccurrences = 0 - val maxReps = unparser.maxRepeats(state) + val maxReps = unparser.maxRepeatsConst // // The number of occurrances we unparse is always exactly driven diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/TestBuildAheadDataProcessor.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/TestBuildAheadDataProcessor.scala new file mode 100644 index 0000000000..4a9ba36b2c --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/TestBuildAheadDataProcessor.scala @@ -0,0 +1,253 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors + +import java.io.ByteArrayOutputStream +import scala.xml.Node + +import org.apache.daffodil.api +import org.apache.daffodil.core.compiler.Compiler +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter + +import org.junit.Assert.* +import org.junit.Test + +/** + * Checks that a compiled schema always carries an infoset builder, that the + * infosetBuilderMode tunable can be changed on a compiled DataProcessor, and how + * DataProcessor.unparse behaves when it cannot start. Whether unparse output is the same in both + * modes is covered by buildAhead.tdml, run with DAFFODIL_TDML_TUNABLES set to + * each value of infosetBuilderMode. + */ +class TestBuildAheadDataProcessor { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + private def unparseToBytes(dp: DataProcessor, infosetXML: Node): Array[Byte] = { + val out = new ByteArrayOutputStream() + val res = dp.unparse(new ScalaXMLInfosetInputter(infosetXML), out) + assertFalse(res.getDiagnostics.toString, res.isError) + out.toByteArray + } + + // A setup failure (inputter never produces StartDocument) must yield + // a failed UnparseResult, not an NPE from a null error-path state. + @Test def testMalformedInfosetInputterGetsCleanErrorNotNPE(): Unit = { + // Needs build ahead in use (the tunable on), or DataProcessor.unparse takes the + // event-driven path and unparseBuildAhead (the method under test) would never run. + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("infosetBuilderMode", "buildAhead") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + + // hasNext() = false immediately means initialize()'s + // "!delegate.hasNext" check fires straight away, before any actual + // infoset event is produced; exactly the "never starts with + // StartDocument" failure this guards. + val neverStartsInputter = new api.infoset.InfosetInputter { + override def getEventType() = null + override def getLocalName() = null + override def getNamespaceURI() = null + override def getSimpleText( + primType: org.apache.daffodil.runtime1.dpath.NodeInfo.Kind, + runtimeProperties: java.util.Map[String, String] + ) = null + override def isNilled(): java.lang.Boolean = null + override def hasNext() = false + override def next(): Unit = () + override def fini(): Unit = () + } + + val out = new ByteArrayOutputStream() + val res = dp.unparse(neverStartsInputter, out) + + assertTrue("expected a failed UnparseResult, not a successful one", res.isError) + assertTrue( + res.getDiagnostics.get(0).getMessage.contains("does not start with StartDocument") + ) + } + + // A schema with no OVC at all still gets a builder. + @Test def testSchemaWithNoOVCAtAllStillGetsBuilder(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("infosetBuilderMode", "buildAhead") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertFalse(dp.ssrd.builder.isEmpty) + } + + // A schema where every OVC is resolvable-without-writing gets a builder: + // the common case. + @Test def testSchemaWithOnlyResolvableOVCGetsBuilder(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("infosetBuilderMode", "buildAhead") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertFalse(dp.ssrd.builder.isEmpty) + } + + // The builder exists even when the tunable is off at compile time, and the + // same compiled processor unparses identically with the tunable switched + // either way afterward. + @Test def testTunableCanChangeAfterCompile(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("infosetBuilderMode", "eventDriven") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertFalse(dp.ssrd.builder.isEmpty) + + val infoset = 7 + val eventDriven = unparseToBytes(dp, infoset) + val buildAhead = unparseToBytes( + dp.copy(tunables = dp.tunables.withTunable("infosetBuilderMode", "buildAhead")), + infoset + ) + assertArrayEquals(eventDriven, buildAhead) + assertArrayEquals( + eventDriven, + unparseToBytes( + dp.copy(tunables = dp.tunables.withTunable("infosetBuilderMode", "eventDriven")), + infoset + ) + ) + } + + // Confirms the builder survives a save/reload round trip. + @Test def testBuilderSurvivesSaveReload(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("infosetBuilderMode", "buildAhead") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertFalse(dp.ssrd.builder.isEmpty) + + val os = new ByteArrayOutputStream() + dp.save(java.nio.channels.Channels.newChannel(os)) + val reloadedDp = Compiler() + .reload(new java.io.ByteArrayInputStream(os.toByteArray)) + .asInstanceOf[DataProcessor] + assertFalse(reloadedDp.ssrd.builder.isEmpty) + + val infoset = xyz + assertArrayEquals( + unparseToBytes(dp, infoset), + unparseToBytes(reloadedDp, infoset) + ) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/InfosetBuildTestFixture.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/InfosetBuildTestFixture.scala new file mode 100644 index 0000000000..9f9df8db12 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/InfosetBuildTestFixture.scala @@ -0,0 +1,105 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream +import scala.jdk.CollectionConverters.* +import scala.xml.Node + +import org.apache.daffodil.api +import org.apache.daffodil.core.compiler.Compiler +import org.apache.daffodil.runtime1.infoset.InfosetBuildCursor +import org.apache.daffodil.runtime1.infoset.InfosetInputter +import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter +import org.apache.daffodil.runtime1.processors.DataProcessor +import org.apache.daffodil.runtime1.processors.TermRuntimeData + +/** + * Shared helpers for tests of the infoset build cursor and build state. + */ +object InfosetBuildTestFixture { + private def throwDiagnostics(ds: java.util.List[api.Diagnostic]): Nothing = + throw new Exception(ds.asScala.map(_.getMessage()).mkString("\n")) + + /** + * Compiles testSchema with the given tunables and returns the resulting + * DataProcessor without a saveAndReload round-trip, since the tests build + * state directly off the live object. + */ + def compileForUnparse( + testSchema: Node, + tunables: Map[String, String] = Map.empty + ): DataProcessor = { + val pf = Compiler().withTunables(tunables).compileNode(testSchema) + if (pf.isError) throwDiagnostics(pf.getDiagnostics) + val dp = pf.onPath("/").asInstanceOf[DataProcessor] + if (dp.isError) throwDiagnostics(dp.getDiagnostics) + dp + } + + /** + * Builds a fresh InfosetInputter walking infosetXML against dp, already + * initialized with the root TRD pushed. + */ + def newInitializedInputter(infosetXML: Node, dp: DataProcessor): InfosetInputter = { + val inputter = new InfosetInputter(new ScalaXMLInfosetInputter(infosetXML)) + inputter.initialize(dp.ssrd.elementRuntimeData, dp.tunables) + inputter + } + + /** + * Unparses infosetXML event driven, then again by building the infoset ahead + * and unparsing the built tree. Returns (eventDrivenBytes, buildAheadBytes) for + * the caller to assert equality on. + */ + def getEventDrivenAndBuildAheadBytes( + dp: DataProcessor, + infosetXML: Node + ): (Array[Byte], Array[Byte]) = { + val eventDrivenOut = new ByteArrayOutputStream() + val eventDrivenResult = dp.unparse(new ScalaXMLInfosetInputter(infosetXML), eventDrivenOut) + if (eventDrivenResult.isError) throwDiagnostics(eventDrivenResult.getDiagnostics) + + val buildAheadOut = new ByteArrayOutputStream() + val buildAheadInputter = newInitializedInputter(infosetXML, dp) + val cursor = new InfosetBuildCursor( + dp.ssrd.builder, + new InfosetBuildState(buildAheadInputter, dp.tunables) + ) + val buildAheadState = UState.createInitialUStateForBuildAhead( + buildAheadOut, + dp, + buildAheadInputter, + false, + new TreeEventState(cursor, false) + ) + buildAheadState.getDataOutputStream.setPriorBitOrder( + dp.ssrd.elementRuntimeData.defaultBitOrder + ) + + val rootUnparser = dp.ssrd.unparser + buildAheadState.pushTRD(dp.ssrd.elementRuntimeData) + rootUnparser.unparse1(buildAheadState) + buildAheadState.popTRD(rootUnparser.context.asInstanceOf[TermRuntimeData]) + cursor.advance(lastAdvance = true) + buildAheadState.evalSuspensions(isFinal = true) + buildAheadState.getDataOutputStream.setFinished(buildAheadState) + + (eventDrivenOut.toByteArray, buildAheadOut.toByteArray) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestInfosetBuildCursor.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestInfosetBuildCursor.scala new file mode 100644 index 0000000000..5a3fddeaf0 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/TestInfosetBuildCursor.scala @@ -0,0 +1,419 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream +import java.nio.charset.StandardCharsets +import scala.xml.Node + +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.runtime1.infoset.InfosetBuildCursor +import org.apache.daffodil.runtime1.infoset.StreamingInfosetWalker +import org.apache.daffodil.runtime1.infoset.XMLTextInfosetOutputter +import org.apache.daffodil.runtime1.processors.TermRuntimeData + +import org.junit.Assert.* +import org.junit.Test + +/** + * Tests the infoset build cursor and build state against the unparse that + * reads the built tree as events: the event sequence build surfaces, the lead + * counter, the build ahead window, and the build ahead unparse matching an + * event-driven unparse for scalar, array and choice content. + */ +class TestInfosetBuildCursor { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + // A separated sequence of three scalars. + private val separatedRowSchema = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + private val separatedRowInfoset = + + Alice + 30 + Boston + + + // A scalar, an array and a choice in one separated sequence. + private val arrayChoiceSchema = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + // Three "item" occurrences exercise the array loop; typeB (not typeA) + // exercises actual choice resolution. lengthKind=delimited (not explicit) + // avoids double-firing CaptureStartOfContentLengthUnparser's non-idempotent + // marker, since a tree reused from a completed unparse runs it again. + private val arrayChoiceInfoset = + +
H
+ a + b + c + X +
+ + /** + * A build side, an InfosetBuildCursor over an InfosetBuildState, and an + * unparse side that reads the tree the build side makes. + */ + private final class BuildAheadRun(sch: Node, infoset: Node, buildAheadLimit: Long = 100) { + val dp = InfosetBuildTestFixture.compileForUnparse( + sch, + Map( + "releaseUnneededInfoset" -> "false", + "infosetBuilderMode" -> "buildAhead", + "unparseBuildAheadWindowNodes" -> buildAheadLimit.toString + ) + ) + val buildInputter = InfosetBuildTestFixture.newInitializedInputter(infoset, dp) + val buildState = new InfosetBuildState(buildInputter, dp.tunables) + val cursor = new InfosetBuildCursor(dp.ssrd.builder, buildState) + + def buildAll(): Unit = cursor.advance(lastAdvance = true) + + def rootNode = buildInputter.documentElement.child(0).asComplex + + // Unparses the tree build produced, pulling build forward as the unparse + // needs it, and returns the output. + def unparseBuiltTree(): String = { + val out = new ByteArrayOutputStream() + val state = UState.createInitialUStateForBuildAhead( + out, + dp, + buildInputter, + false, + new TreeEventState(cursor, false) + ) + state.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + val rootUnparser = dp.ssrd.unparser + state.pushTRD(dp.ssrd.elementRuntimeData) + rootUnparser.unparse1(state) + state.popTRD(rootUnparser.context.asInstanceOf[TermRuntimeData]) + // Build may still have trailing end events to consume, and a speculative + // separator is written via a suspension that must drain before the DOS + // is finalized. + buildAll() + state.evalSuspensions(isFinal = true) + state.getDataOutputStream.setFinished(state) + new String(out.toByteArray, StandardCharsets.US_ASCII) + } + } + + // A tree built outside an event-driven unparse, compared against that unparse. + private def assertBuildAheadMatchesEventDriven(sch: Node, infoset: Node): Array[Byte] = { + // Event-driven on purpose: the tunable would otherwise replace it. + val dp = InfosetBuildTestFixture.compileForUnparse( + sch, + Map("releaseUnneededInfoset" -> "false", "infosetBuilderMode" -> "eventDriven") + ) + val (eventDrivenBytes, buildAheadBytes) = + InfosetBuildTestFixture.getEventDrivenAndBuildAheadBytes(dp, infoset) + assertArrayEquals(eventDrivenBytes, buildAheadBytes) + eventDrivenBytes + } + + @Test def testBuildStateSurfacesCorrectEventSequence(): Unit = { + val run = new BuildAheadRun(separatedRowSchema, separatedRowInfoset) + + // Drives through the actual InfosetBuilder frames rather than hand-driven + // advance() calls, since next-element resolution depends on the TRD + // push/pop the element frame performs. This schema's separator never + // reaches InfosetBuildState: the InfosetBuilder tree skips the + // delimiter-stack wrapper. + run.buildAll() + + assertEquals(4L, run.buildState.currentLead) // row, name, age, city + + assertEquals(3, run.rootNode.numChildren) + assertEquals("name", run.rootNode.child(0).erd.name) + assertEquals("age", run.rootNode.child(1).erd.name) + assertEquals("city", run.rootNode.child(2).erd.name) + } + + @Test def testLeadCounterIncrementsOnBuildAndDecrementsOnUnparse(): Unit = { + // Fixed dfdl:length is safe here because this tree comes from + // InfosetBuildState, which never runs content-unparsing (including + // CaptureStartOfContentLengthUnparser). + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + val run = new BuildAheadRun(sch, separatedRowInfoset) + + assertEquals(0L, run.buildState.currentLead) + run.buildAll() + // row itself, name, age, city = 4 elements total, each incrementing once + // via unparseBegin's actual hookup. + assertEquals(4L, run.buildState.currentLead) + + // The unparse reads the same already-built tree, decrementing the lead + // counter as it goes. + assertEquals("Alice30Boston", run.unparseBuiltTree()) + + // The unparse decremented once per element too, so the counter is back to + // 0: build and the unparse agree on how many nodes exist. + assertEquals(0L, run.buildState.currentLead) + } + + // With a small buildAheadLimit, one advance() leaves the lead counter just + // past it, and the unparse is still correct across the refills the unparse + // triggers while it runs. + @Test def testBuildStopsAtBuildAheadLimit(): Unit = { + val numItems = 40 + val buildAheadLimit = 3L + + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + , + elementFormDefault = "unqualified" + ) + + val items = (0 until numItems).map(i => {s"i$i"}) + val infoset = + + {items} + + val expectedBytes = (0 until numItems).map(i => s"i$i").mkString(",") + + val run = new BuildAheadRun(sch, infoset, buildAheadLimit) + + run.cursor.advance() + + // numItems + 1 (row + all items) is what currentLead would equal here if + // advance() ignored the build ahead limit. It counts one node per element, so + // it stops at the first node past the limit. + assertFalse("expected build to stop with more left to build", run.cursor.isFinished) + assertEquals( + s"expected build to stop as soon as the lead passed buildAheadLimit=$buildAheadLimit " + + s"(numItems=$numItems)", + buildAheadLimit + 1, + run.buildState.currentLead + ) + + // The unparse pulls the rest of the tree forward as it needs it; the + // output must be byte-for-byte correct despite having been built across + // many separate advance() calls rather than a single one-shot build pass. + assertEquals(expectedBytes, run.unparseBuiltTree()) + } + + @Test def testBuildAheadMatchesEventDriven(): Unit = { + assertBuildAheadMatchesEventDriven(separatedRowSchema, separatedRowInfoset) + } + + @Test def testArrayAndChoiceBuildAheadMatchesEventDriven(): Unit = { + val eventDrivenBytes = + assertBuildAheadMatchesEventDriven(arrayChoiceSchema, arrayChoiceInfoset) + assertEquals("H,a,b,c,X", new String(eventDrivenBytes, StandardCharsets.US_ASCII)) + } + + // Drives InfosetBuildState directly, then feeds its tree to the unparse + // (end-to-end build-then-unparse). + @Test def testStandaloneBuildStateNavigatesArrayChoiceSeparator(): Unit = { + val run = new BuildAheadRun(arrayChoiceSchema, arrayChoiceInfoset) + + // The cursor builds the whole tree from the inputter, including the array + // and choice content. + run.buildAll() + + // row, header, item x3, typeB = 6 elements total. + assertEquals(6L, run.buildState.currentLead) + + assertEquals(3, run.rootNode.numChildren) + assertEquals("header", run.rootNode.child(0).erd.name) + assertEquals("item", run.rootNode.child(1).erd.name) + assertEquals(3, run.rootNode.child(1).asInstanceOf[DIArray].numChildren) + assertEquals("typeB", run.rootNode.child(2).erd.name) + + // Unparsing the tree InfosetBuildState just constructed confirms it's a + // usable, fully-built tree, not just a navigation exercise. + assertEquals("H,a,b,c,X", run.unparseBuiltTree()) + } + + // Each event as "start element name", "end array name" and so on. + private def eventDescriptions(events: InfosetEventState): List[String] = { + val descriptions = List.newBuilder[String] + while (events.advance) { + val event = events.advanceAccessor + val position = if (event.isStart) { + "start" + } else { + "end" + } + val kind = if (event.isElement) { + "element" + } else { + "array" + } + descriptions += position + " " + kind + " " + event.erd.name + } + descriptions.result() + } + + @Test def testTreeEventsForScalars(): Unit = { + val run = new BuildAheadRun(separatedRowSchema, separatedRowInfoset) + val treeEvents = new TreeEventState(run.cursor, true) + assertEquals( + List( + "start element row", + "start element name", + "end element name", + "start element age", + "end element age", + "start element city", + "end element city", + "end element row" + ), + eventDescriptions(treeEvents) + ) + } + + private val arrayChoiceEvents = List( + "start element row", + "start element header", + "end element header", + "start array item", + "start element item", + "end element item", + "start element item", + "end element item", + "start element item", + "end element item", + "end array item", + "start element typeB", + "end element typeB", + "end element row" + ) + + @Test def testTreeEventsForArrayAndChoice(): Unit = { + val run = new BuildAheadRun(arrayChoiceSchema, arrayChoiceInfoset) + val treeEvents = new TreeEventState(run.cursor, true) + assertEquals(arrayChoiceEvents, eventDescriptions(treeEvents)) + } + + @Test def testTreeEventsPullBuildOnlyAsFarAsNeeded(): Unit = { + val run = new BuildAheadRun(arrayChoiceSchema, arrayChoiceInfoset, buildAheadLimit = 1) + val treeEvents = new TreeEventState(run.cursor, true) + assertTrue(treeEvents.advance) + assertEquals("row", treeEvents.advanceAccessor.erd.name) + // Only the root has been needed so far, so build has not run to the end. + assertFalse(run.cursor.isFinished) + assertEquals(arrayChoiceEvents.tail, eventDescriptions(treeEvents)) + assertTrue(run.cursor.isFinished) + } + + // The infoset as a debugger shows it: the whole built tree, limited to what + // the unparse has reached. + private def reachedInfoset(run: BuildAheadRun, treeEvents: TreeEventState): String = { + val out = new ByteArrayOutputStream() + val xml = new XMLTextInfosetOutputter(out, pretty = false, minimal = true) + StreamingInfosetWalker( + run.buildInputter.documentElement, + xml, + walkHidden = false, + ignoreBlocks = true, + releaseUnneededInfoset = false, + visibleChildCounts = treeEvents.reachedChildCounts() + ).walk(lastWalk = true) + out.toString("UTF-8") + } + + @Test def testReachedChildCountsLimitTheDebuggerInfoset(): Unit = { + val run = new BuildAheadRun(separatedRowSchema, separatedRowInfoset) + run.buildAll() + val treeEvents = new TreeEventState(run.cursor, false) + + // start row, start name, end name + assertTrue(treeEvents.advance) + assertTrue(treeEvents.advance) + assertTrue(treeEvents.advance) + val afterName = reachedInfoset(run, treeEvents) + assertTrue(afterName, afterName.contains("Alice")) + assertFalse(afterName, afterName.contains("30")) + + // The start of age is computed but not consumed, so age is not reached. + assertTrue(treeEvents.inspect) + assertFalse(reachedInfoset(run, treeEvents).contains("30")) + + assertTrue(treeEvents.advance) + val afterAgeStart = reachedInfoset(run, treeEvents) + assertTrue(afterAgeStart, afterAgeStart.contains("30")) + assertFalse(afterAgeStart, afterAgeStart.contains("Boston")) + } +} diff --git a/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd b/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd index db5568846a..b91f2922f3 100644 --- a/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd +++ b/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd @@ -534,6 +534,29 @@ + + + + How unparsing gets the infoset. Values are: + - buildAhead: a build pass runs ahead of an unparse pass over the same + infoset tree, resolving forward references against the tree directly + instead of via Suspensions where possible. How far build may run + ahead of unparse is bounded by unparseBuildAheadWindowNodes. + - eventDriven: the events of the infoset inputter are unparsed as they + arrive, with no tree built ahead. + + + + + + + Only consulted when infosetBuilderMode is buildAhead. The maximum + number of infoset nodes build may construct ahead of unparse before + build pauses to let unparse catch up. Bounds peak memory use to + roughly this many infoset nodes rather than the whole infoset. + + + @@ -710,6 +733,13 @@ + + + + + + + diff --git a/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala b/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala index d6632308b8..84c3e505ac 100644 --- a/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala +++ b/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala @@ -79,7 +79,8 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm | if (configOpt.isDefined) { | val loader = new DaffodilXMLLoader() | val node = loader.load(URISchemaSource(Paths.get(configPath).toFile, configOpt.get), Some(XMLUtils.dafextURI)) - | tunablesMap(node) + | val optTunablesNode = (node \ "tunables").headOption + | optTunablesNode.map(tunablesMap(_)).getOrElse(Map.empty) | } else { | Map.empty | } @@ -98,15 +99,17 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm | tunables.foldLeft(this) { case (dafTuns, (tunable, value)) => dafTuns.withTunable(tunable, value) } | } | + | // One large match over every tunable name would exceed the JVM's + | // 64KB-per-method bytecode limit as tunables accumulate over time, so + | // this is split into a chain of smaller private methods instead, each + | // falling through to the next on no match. | def withTunable(tunable: String, value: String): DaffodilTunables = { - | tunable match { + | withTunablePart0(tunable, value) + | } + | """.trim.stripMargin val bottom = """ - | case _ => throw new IllegalArgumentException("Unknown tunable: " + tunable) - | } - | } - | | private def throwInvalidTunableValue(tunable: String, value: String) = { | throw new IllegalArgumentException("Invalid value for tunable " + tunable + ": " + value) | } @@ -114,6 +117,11 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm |} """.trim.stripMargin + // How many tunables' worth of case-match bytecode go into each + // withTunablePartN method; small enough to leave headroom against the + // JVM's 64KB-per-method limit as more tunables are added over time. + private val tunablesPerPart = 8 + val tunablesRoot = (schemaRootConfig \ "element").find(_ \@ "name" == "tunables").get val tunableNodes = tunablesRoot \\ "all" \ "element" @@ -165,15 +173,38 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm .map(_.scalaDefinition) .mkString(" ", ",\n ", ")") - val conversionString = - tunables - .map { tunable => - tunable.scalaConversion - .split("\n") - .filter(_.trim.length > 0) - .mkString(" ", "\n ", "") + // Split across several withTunablePartN methods rather than one large + // match: each chunk's case bodies, then a fallthrough to the next + // chunk's method, or to the final "unknown tunable" error on the last. + val tunableChunks = tunables.grouped(tunablesPerPart).toIndexedSeq + val numParts = tunableChunks.length + val partsString = + tunableChunks.zipWithIndex + .map { case (chunk, idx) => + val casesString = + chunk + .map { tunable => + tunable.scalaConversion + .split("\n") + .filter(_.trim.length > 0) + .mkString(" ", "\n ", "") + } + .mkString("\n") + val fallthrough = + if (idx == numParts - 1) + """ case _ => throw new IllegalArgumentException("Unknown tunable: " + tunable)""" + else + s" case _ => withTunablePart${idx + 1}(tunable, value)" + s""" + | private def withTunablePart${idx}(tunable: String, value: String): DaffodilTunables = { + | tunable match { + |${casesString} + |${fallthrough} + | } + | } + """.trim.stripMargin } - .mkString("\n") + .mkString("\n\n") w.write(top) w.write("\n") @@ -181,7 +212,7 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm w.write("\n") w.write(middle) w.write("\n") - w.write(conversionString) + w.write(partsString) w.write("\n") w.write(bottom) w.write("\n") diff --git a/daffodil-tdml-lib/src/main/scala/org/apache/daffodil/tdml/TDMLRunner.scala b/daffodil-tdml-lib/src/main/scala/org/apache/daffodil/tdml/TDMLRunner.scala index 74443cd32b..11c044552f 100644 --- a/daffodil-tdml-lib/src/main/scala/org/apache/daffodil/tdml/TDMLRunner.scala +++ b/daffodil-tdml-lib/src/main/scala/org/apache/daffodil/tdml/TDMLRunner.scala @@ -387,6 +387,15 @@ class DFDLTestSuite private[tdml] ( var checkAllTopLevel: Boolean = compileAllTopLevel + /** + * Tunables from the DAFFODIL_TDML_TUNABLES environment variable, a + * comma-separated list of name=value pairs. They apply to every test and + * are overridden by a test's own defineConfig tunables, so a whole suite + * can run under any combination of tunables. + */ + private[tdml] lazy val envTunables: Map[String, String] = + TDMLEnvTunables.parse(sys.env.getOrElse(TDMLEnvTunables.VariableName, "")) + def setCheckAllTopLevel(flag: Boolean): Unit = { checkAllTopLevel = flag } @@ -902,7 +911,8 @@ abstract class TestCase(testCaseXML: NodeSeq, val parent: DFDLTestSuite) { } } - lazy val tunables = cfg.map { _.tunablesMap }.getOrElse(Map.empty) + lazy val tunables = + parent.envTunables ++ cfg.map { _.tunablesMap }.getOrElse(Map.empty) lazy val tunableObj = DaffodilTunables(tunables) @@ -3151,3 +3161,29 @@ object GlobalTDMLCompileResultCache { new TDMLCompileResultCache(Some(expireTimeSeconds)) } } + +private[tdml] object TDMLEnvTunables { + val VariableName = "DAFFODIL_TDML_TUNABLES" + + /** + * Parses a comma-separated list of name=value pairs, as in + * "key=value,key=value". Blank entries are ignored, and a value may itself + * contain '='. An entry without a name or without '=' is an error. + */ + def parse(value: String): Map[String, String] = + value + .split(',') + .map(_.trim) + .filter(_.nonEmpty) + .map { pair => + pair.split("=", 2) match { + case Array(name, tunableValue) if name.trim.nonEmpty => + (name.trim, tunableValue.trim) + case _ => + throw new IllegalArgumentException( + s"Invalid $VariableName entry '$pair'; expected name=value" + ) + } + } + .toMap +} diff --git a/daffodil-tdml-lib/src/test/scala/org/apache/daffodil/tdml/UnitTestTDMLRunner.scala b/daffodil-tdml-lib/src/test/scala/org/apache/daffodil/tdml/UnitTestTDMLRunner.scala index ba7d0f8b3f..0fab12730f 100644 --- a/daffodil-tdml-lib/src/test/scala/org/apache/daffodil/tdml/UnitTestTDMLRunner.scala +++ b/daffodil-tdml-lib/src/test/scala/org/apache/daffodil/tdml/UnitTestTDMLRunner.scala @@ -869,4 +869,46 @@ class UnitTestTDMLRunner { assertTrue(e.getMessage().contains("'type' must appear on element 'documentPart'")) } + @Test def testEnvTunablesEmpty(): Unit = { + assertEquals(Map.empty[String, String], TDMLEnvTunables.parse("")) + assertEquals(Map.empty[String, String], TDMLEnvTunables.parse(" , ,")) + } + + @Test def testEnvTunablesSingle(): Unit = { + assertEquals( + Map("infosetBuilderMode" -> "eventDriven"), + TDMLEnvTunables.parse("infosetBuilderMode=eventDriven") + ) + } + + @Test def testEnvTunablesMultipleWithWhitespace(): Unit = { + assertEquals( + Map("a" -> "1", "b" -> "two"), + TDMLEnvTunables.parse(" a = 1 , b=two ,") + ) + } + + @Test def testEnvTunablesValueContainsEquals(): Unit = { + assertEquals(Map("a" -> "b=c"), TDMLEnvTunables.parse("a=b=c")) + } + + @Test def testEnvTunablesLastDuplicateWins(): Unit = { + assertEquals(Map("a" -> "2"), TDMLEnvTunables.parse("a=1,a=2")) + } + + @Test def testEnvTunablesMissingEquals(): Unit = { + val e = intercept[IllegalArgumentException] { + TDMLEnvTunables.parse("a=1,justAName") + } + assertTrue(e.getMessage.contains("justAName")) + assertTrue(e.getMessage.contains("expected name=value")) + } + + @Test def testEnvTunablesMissingName(): Unit = { + val e = intercept[IllegalArgumentException] { + TDMLEnvTunables.parse("=1") + } + assertTrue(e.getMessage.contains("expected name=value")) + } + } diff --git a/daffodil-test-integration/src/test/scala/org/apache/daffodil/cliTest/TestCLIDebugger.scala b/daffodil-test-integration/src/test/scala/org/apache/daffodil/cliTest/TestCLIDebugger.scala index 626dc019a0..e1c5ea9045 100644 --- a/daffodil-test-integration/src/test/scala/org/apache/daffodil/cliTest/TestCLIDebugger.scala +++ b/daffodil-test-integration/src/test/scala/org/apache/daffodil/cliTest/TestCLIDebugger.scala @@ -24,7 +24,10 @@ import org.apache.daffodil.cli.Main.ExitCode import org.apache.daffodil.cli.cliTest.Util.* import org.apache.daffodil.core.util.TestUtils.intercept +import net.sf.expectit.matcher.Matchers.eof import net.sf.expectit.matcher.Matchers.regexp +import org.junit.Assert.assertTrue +import org.junit.Assert.fail import org.junit.Test /** @@ -1325,4 +1328,50 @@ class TestCLIDebugger { }(ExitCode.Success) } + /** + * Tracing an unparse must step through the same bit positions whether or not + * the build-ahead path is used. The + * trace also shows the infoset, data and diff at each step; those differ + * in small ways on the build ahead path (the child and group indexes, and + * nodes built one step ahead), so they are not compared. + */ + @Test def test_CLI_Tdml_Trace_buildAheadUnparseMatchesEventDriven(): Unit = { + val tdml = path( + "daffodil-test/src/test/resources/org/apache/daffodil/unparser/buildAhead.tdml" + ) + + def steps(buildAhead: Boolean): Seq[String] = { + val tunables = Map("DAFFODIL_TDML_TUNABLES" -> s"infosetBuilderMode=${if (buildAhead) "buildAhead" else "eventDriven"}") + var transcript = "" + runCLI( + args"test -t $tdml nviScopedVariableWithValueLengthOVC", + fork = true, + envs = envs ++ tunables + ) { cli => + transcript = cli.expect(eof()).getInput + }(ExitCode.Success) + transcript.linesIterator + .filter { line => + line.startsWith("bitPosition:") || line.startsWith("-----") + } + .map(_.replaceAll("@[0-9a-f]+", "")) + .toSeq + } + + val eventDriven = steps(buildAhead = false) + val buildAhead = steps(buildAhead = true) + assertTrue("expected a trace of steps", eventDriven.exists(_.startsWith("bitPosition:"))) + val firstDifference = eventDriven.zipAll(buildAhead, "", "").indexWhere { + case (a, b) => a != b + } + if (firstDifference >= 0) { + def around(lines: Seq[String]) = + lines.slice(firstDifference - 3, firstDifference + 3).mkString("\n ") + fail( + s"first difference at line $firstDifference of ${eventDriven.length} event-driven and " + + s"${buildAhead.length} build ahead lines\n event-driven:\n ${around(eventDriven)}" + + s"\n build ahead:\n ${around(buildAhead)}" + ) + } + } } diff --git a/daffodil-test/src/test/resources/org/apache/daffodil/unparser/buildAhead.tdml b/daffodil-test/src/test/resources/org/apache/daffodil/unparser/buildAhead.tdml new file mode 100644 index 0000000000..d19735a99c --- /dev/null +++ b/daffodil-test/src/test/resources/org/apache/daffodil/unparser/buildAhead.tdml @@ -0,0 +1,1436 @@ + + + + + + + + 3 + + + + + + 2 + 1 + 1 + + + + + + 2 + + + + + + -1 + + + 2 + + + + + + + + + + + + + + + + + + + + 005 + + + 006005 + + + + + + + + + + + + + + + + + + + + + + + + + +
H
+ a + b + c + X +
+
+
+ H,a,b,c,X +
+ + + + + + + + + + + + + + + + + + + + a + b + + + + a,b + + + + + + + + + + + + + + + + + + + + + + + + + + 3 + + + [hello,3 + + + + + + + + + + + + + + + + + + + + + + a + b + c + + + + + 61 0A 62 0A 63 0A + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + 1 + + + first_defaultable1 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + 0 + 1 + + + + first_defaultable01 + + + + + + + + + + + + + + + + + + + + + i0 + i1 + i2 + i3 + i4 + i5 + i6 + i7 + i8 + i9 + i10 + i11 + + + + i0,i1,i2,i3,i4,i5,i6,i7,i8,i9,i10,i11 + + + + + + + + + + + + + + + + + + + + + + + + + + + B + I1 + I2 + A + + + + B,I1I2,A + + + + + + + + + + + + + + + + + + + + + + XY + + + + 58 59 20 20 20 + + + + + + + + + + + + + + + + + + + + + + + + + + V + + + HV + + + + + + + + + + + + + + + + + + + + + + + A + B + + + + [AB] + + + + + + + + + + + + + + + + + + + [] + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + V + + + V + + + + + + + + + + + + + + + + + + + + + + + + + + one, two + three + + + + one#, two,three + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + ABCDEFGH + Q + + + + ABCDEFGHQ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + ABCDEFGHIJ + Q + + + + + exceeded fixed layer length of 8 + + + + + + + + + + + + + + + + + + + + + + 005 + + + 007006005 + + + + + + + + + + + + + + + + + + + + + + 1 + 2 + 3 + + + + + 00 00 00 01 00 00 00 02 00 00 00 03 + + + + + + + + + + + + + + + + + + + + + + + + 5 + hello + + + + 05hello + + + + + + + + + + + + + + + + + + + + + + + + hello + + + 05hello + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + hi + bye + + + + + 06hi,bye + + + + + + + + + + + + + + + + + + + + + 1 + 2 + + + + 122 + + + + + + + + + + + + + + + + + + + + + 0 + 1 + 2 + 3 + 4 + 5 + 6 + 7 + 8 + 9 + 0 + 1 + 2 + 3 + 4 + 5 + 6 + 7 + 8 + 9 + 0 + 1 + 2 + 3 + 4 + 5 + 6 + 7 + 8 + 9 + 0 + 1 + 2 + 3 + 4 + 5 + 6 + 7 + 8 + 9 + 0 + 1 + 2 + 3 + 4 + 5 + 6 + 7 + 8 + 9 + 0 + 1 + 2 + 3 + 4 + 5 + 6 + 7 + 8 + 9 + 0 + 1 + 2 + 3 + 4 + 5 + 6 + 7 + 8 + 9 + 0 + 1 + 2 + 3 + 4 + 5 + 6 + 7 + 8 + 9 + 0 + 1 + 2 + 3 + 4 + 5 + 6 + 7 + 8 + 9 + 0 + 1 + 2 + 3 + 4 + 5 + 6 + 7 + 8 + 9 + + + + 0123456789012345678901234567890123456789012345678901234567890123456789012345678901234567890123456789100 + + + + + + + + + + + + + + + + + + + + + + + + + y + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + 1 + 2 + 3 + 4 + + true + + + + 1,2,3,4 + + + + + + + + + + + + + + + + + + + + + + + + + + + value0 + value1 + value2 + value3 + value4 + value5 + value6 + value7 + + + + 6 |value06 |value16 |value26 |value36 |value46 |value56 |value66 |value7 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + 00value0 + 01value1 + 02value2 + 03value3 + 04value4 + 05value5 + 06value6 + 07value7 + + + + 0 |6 |value01 |7 |value12 |8 |value23 |9 |value34 |10 |value45 |11 |value56 |12 |value67 |13 |value7 + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + 42 + hello + + + + 42|47 |hello + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + 00 + 01 + 02 + 03 + 04 + 05 + 06 + 07 + + + + + 30 20 31 20 32 20 33 20 34 20 35 20 36 20 37 20 2D 31 20 20 + + + + + + + + + + + + + + + + + + + + + + + + 5 + hello + + + + 05|hello + + + + + + + + + + + + + + + + + + + + + + + + + + + + + 3 + xyz + + + + 03|xyz + + + + + + + + + + + + + + + + + + + + xyz + + + 03xyz + + + + + + + + + + + + + + + + + + + + + + + 005 + xyz + + + + 00600503xyz + + +
diff --git a/daffodil-test/src/test/scala/org/apache/daffodil/unparser/TestBuildAhead.scala b/daffodil-test/src/test/scala/org/apache/daffodil/unparser/TestBuildAhead.scala new file mode 100644 index 0000000000..15830e92de --- /dev/null +++ b/daffodil-test/src/test/scala/org/apache/daffodil/unparser/TestBuildAhead.scala @@ -0,0 +1,67 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.unparser + +import org.apache.daffodil.junit.tdml.TdmlSuite +import org.apache.daffodil.junit.tdml.TdmlTests + +import org.junit.Test + +object TestBuildAhead extends TdmlSuite { + val tdmlResource = "/org/apache/daffodil/unparser/buildAhead.tdml" +} + +class TestBuildAhead extends TdmlTests { + val tdmlSuite = TestBuildAhead + + @Test def ovcSuspension = test + @Test def arrayChoiceSeparator = test + @Test def absentTrailingOptionalSuppressesSeparator = test + @Test def hiddenChoice = test + @Test def trailingArrayPostfixSeparator = test + @Test def choiceBranchWithAbsentOptionalElement = test + @Test def choiceBranchWithPresentOptionalElement = test + @Test def manyOccurrenceArrayWithSmallBuildAheadLimit = test + @Test def nestedBareSequence = test + @Test def fixedLengthChoicePadding = test + @Test def hiddenGroup = test + @Test def initiatorTerminator = test + @Test def nillableComplexWithEmptyContentNotNilled = test + @Test def nillableComplexWithEmptyContentNilled = test + @Test def hiddenGroupWithEmptyBody = test + @Test def escapeScheme = test + @Test def layeredSequence = test + @Test def layeredSequenceLengthMismatchError = test + @Test def nestedOvcSuspensions = test + @Test def binaryIntArray = test + @Test def variableLengthExpression = test + @Test def prefixedLength = test + @Test def prefixedLengthComplexContent = test + @Test def ovcCountOfPrecedingArray = test + @Test def ovcCountOfPrecedingArrayAfterBuildFullyFinishes = test + @Test def dynamicTerminatorReferencingArrayCount = test + @Test def ivcExistsOverNestedArray = test + @Test def manyValueLengthForwardReferencesThrottled = test + @Test def nviScopedVariableWithValueLengthOVC = test + @Test def singleNviScopedVariableWithValueLengthOVC = test + @Test def nviScopedSetVariableWithNoForwardReference = test + @Test def delimitedVariableLengthExpression = test + @Test def delimitedComplexVariableLengthExpression = test + @Test def purelyContentLengthOVC = test + @Test def mixedResolvableAndContentLengthOVC = test +}