diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ElementBaseRuntime1Mixin.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ElementBaseRuntime1Mixin.scala index b17c2b4d15..8a606c6436 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ElementBaseRuntime1Mixin.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/ElementBaseRuntime1Mixin.scala @@ -27,6 +27,7 @@ import org.apache.daffodil.lib.schema.annotation.props.gen.LengthKind import org.apache.daffodil.lib.schema.annotation.props.gen.Representation.Text import org.apache.daffodil.lib.util.Delay import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.runtime1.dpath.SuspendableExpression import org.apache.daffodil.runtime1.dsom.DPathElementCompileInfo import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.RuntimeData @@ -134,6 +135,19 @@ trait ElementBaseRuntime1Mixin { self: ElementBase => isReferenced || mightHaveSuspensions } + /** + * True if the schema has at least one dfdl:outputValueCalc element whose + * expression can resolve without writing (schema-wide; consult only via + * schemaSet.root). Gates useBuildWritePrefetch: if false, every OVC needs an + * actual written byte position, so racing build ahead is never beneficial. + */ + final lazy val hasAnyPrefetchBeneficialOVC: Boolean = + schemaSet.allSchemaComponents.exists { + case e: ElementBase if e.isOutputValueCalc => + SuspendableExpression.canResolveWithoutWriting(e.ovcCompiledExpression) + case _ => false + } + final override lazy val dpathCompileInfo = dpathElementCompileInfo /** diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala index 3e9c5634fe..de843c501f 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/runtime1/SchemaSetRuntime1Mixin.scala @@ -91,7 +91,8 @@ trait SchemaSetRuntime1Mixin { root.elementRuntimeData, variableMap, allLayers, - layerRuntimeCompiler + layerRuntimeCompiler, + root.hasAnyPrefetchBeneficialOVC ) if (root.numComponents > root.numUniqueComponents) Logger.log.debug( diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala b/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala index bc70863fe3..360f720724 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/core/util/TestUtils.scala @@ -43,6 +43,7 @@ import org.apache.daffodil.lib.iapi.* import org.apache.daffodil.lib.util.* import org.apache.daffodil.lib.xml.* import org.apache.daffodil.runtime1.iapi.DFDL +import org.apache.daffodil.runtime1.infoset.InfosetInputter import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetOutputter import org.apache.daffodil.runtime1.processors.DataProcessor @@ -118,9 +119,11 @@ object TestUtils { testSchema: scala.xml.Elem, infosetXML: Node, unparseTo: String, - areTracing: Boolean = false + areTracing: Boolean = false, + tunables: Map[String, String] = Map.empty ): java.util.List[api.Diagnostic] = { - val compiler = Compiler().withTunable("allowExternalPathExpressions", "true") + val compiler = + Compiler().withTunable("allowExternalPathExpressions", "true").withTunables(tunables) val pf = compiler.compileNode(testSchema) if (pf.isError) throwDiagnostics(pf.getDiagnostics) var u = saveAndReload(pf.onPath("/").asInstanceOf[DataProcessor]) @@ -198,6 +201,32 @@ object TestUtils { p } + /** + * Compiles testSchema with the given tunables and returns the resulting + * DataProcessor, with no saveAndReload round-trip (unlike compileSchema) + * since some callers build test-only state directly off the live object. + */ + def compileForUnparse( + testSchema: Node, + tunables: Map[String, String] = Map.empty + ): DataProcessor = { + val pf = Compiler().withTunables(tunables).compileNode(testSchema) + if (pf.isError) throwDiagnostics(pf.getDiagnostics) + val dp = pf.onPath("/").asInstanceOf[DataProcessor] + if (dp.isError) throwDiagnostics(dp.getDiagnostics) + dp + } + + /** + * Builds a fresh InfosetInputter walking infosetXML against dp, already + * initialized (root TRD pushed) the same way a real unparse would. + */ + def newInitializedInputter(infosetXML: Node, dp: DataProcessor): InfosetInputter = { + val inputter = new InfosetInputter(new ScalaXMLInfosetInputter(infosetXML)) + inputter.initialize(dp.ssrd.elementRuntimeData, dp.tunables) + inputter + } + private def runSchemaOnRBC( testSchema: Node, data: ReadableByteChannel, diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/MStack.scala b/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/MStack.scala index 6c25a867bb..f2823f0561 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/MStack.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/lib/util/MStack.scala @@ -24,6 +24,12 @@ import Maybe.* object MStack { final case class Mark(val v: Int) extends AnyVal val nullMark = Mark(0) + + /** + * Off by default: growing past initialSize isn't itself wrong, so paying + * this bookkeeping cost on every push isn't worth it normally. + */ + final val trackMaxSizeReached: Boolean = false } /** @@ -33,35 +39,25 @@ object MStack { * catches improper initialization. These were not initializing properly, * so the idiom evolved to use the scala initializers. */ -final class MStackOfBoolean private () - extends MStack[Boolean]((n: Int) => new Array[Boolean](n), false) +final class MStackOfBoolean private (initialSize: Int) + extends MStack[Boolean]((n: Int) => new Array[Boolean](n), false, initialSize) object MStackOfBoolean { - def apply() = { - val stk = new MStackOfBoolean() - stk.init() - stk - } + def apply(initialSize: Int = 32) = new MStackOfBoolean(initialSize) } -final class MStackOfInt extends MStack[Int]((n: Int) => new Array[Int](n), 0) +final class MStackOfInt(initialSize: Int) + extends MStack[Int]((n: Int) => new Array[Int](n), 0, initialSize) object MStackOfInt { - def apply() = { - val stk = new MStackOfInt() - stk.init() - stk - } + def apply(initialSize: Int = 32) = new MStackOfInt(initialSize) } -final class MStackOfLong extends MStack[Long]((n: Int) => new Array[Long](n), 0L) +final class MStackOfLong(initialSize: Int) + extends MStack[Long]((n: Int) => new Array[Long](n), 0L, initialSize) object MStackOfLong { - def apply() = { - val stk = new MStackOfLong() - stk.init() - stk - } + def apply(initialSize: Int = 32) = new MStackOfLong(initialSize) } /** @@ -75,11 +71,11 @@ object MStackOfLong { * So we use an Array[AnyRef] as the representation here, and we * convert null to Nope, and an actual object reference to One(x) */ -final class MStackOfMaybe[T <: AnyRef] { +final class MStackOfMaybe[T <: AnyRef](initialSize: Int = 32) { override def toString = delegate.toString - private val delegate = new MStackOf[T] + private val delegate = new MStackOf[T](initialSize) private val nullT = null.asInstanceOf[T] def copyFrom(other: MStackOfMaybe[T]) = delegate.copyFrom(other.delegate) @@ -120,6 +116,7 @@ final class MStackOfMaybe[T <: AnyRef] { def toListMaybe = delegate.toList.map { (x: AnyRef) => Maybe(x) // Scala compiler bug without this cast } + def maxSizeReached = delegate.maxSizeReached } /** @@ -135,7 +132,7 @@ final class MStackOfMaybe[T <: AnyRef] { * an object reference or null, and call Maybe(thing) explicitly outside the * iteration. Maybe(null) is Nope, and Maybe(thing) is One(thing) if thing is not null. */ -final class MStackOf[T <: AnyRef] extends Serializable { +final class MStackOf[T <: AnyRef](initialSize: Int = 32) extends Serializable { override def toString = delegate.toString @@ -143,7 +140,7 @@ final class MStackOf[T <: AnyRef] extends Serializable { @inline final def length = delegate.length - private val delegate = MStackOfAnyRef() + private val delegate = MStackOfAnyRef(initialSize) @inline final def mark = delegate.mark @inline final def reset(m: MStack.Mark) = delegate.reset(m) @@ -156,6 +153,7 @@ final class MStackOf[T <: AnyRef] extends Serializable { @inline final def isEmpty = delegate.isEmpty def clear() = delegate.clear() def toList = delegate.toList + def maxSizeReached = delegate.maxSizeReached def iterator = delegate.iterator.asInstanceOf[ResettableIterator[T]] @@ -163,15 +161,15 @@ final class MStackOf[T <: AnyRef] extends Serializable { } -private[util] final class MStackOfAnyRef private () - extends MStack[AnyRef]((n: Int) => new Array[AnyRef](n), null.asInstanceOf[AnyRef]) +private[util] final class MStackOfAnyRef private (initialSize: Int) + extends MStack[AnyRef]( + (n: Int) => new Array[AnyRef](n), + null.asInstanceOf[AnyRef], + initialSize + ) object MStackOfAnyRef { - def apply() = { - val stk = new MStackOfAnyRef() - stk.init() - stk - } + def apply(initialSize: Int = 32) = new MStackOfAnyRef(initialSize) } /** @@ -184,16 +182,23 @@ object MStackOfAnyRef { */ protected abstract class MStack[@specialized T] private[util] ( arrayAllocator: (Int) => Array[T], - nullValue: T + nullValue: T, + initialSize: Int = 32 ) { private var index = 0 - private var table: Array[T] = null + private var table: Array[T] = arrayAllocator(initialSize) - def init(): Unit = { - index = 0 - table = arrayAllocator(32) - } + private var maxSizeReached_ = 0 + + /** + * The largest this stack's length has ever grown to, across its + * whole lifetime (not just its current length; pops don't reduce + * this). Diagnostic only: useful for profiling to check whether a + * particular use's initialSize is well-chosen, not for any runtime + * decision. Always 0 unless MStack.trackMaxSizeReached is enabled. + */ + final def maxSizeReached: Int = maxSizeReached_ def copyFrom(other: MStack[T]): Unit = { this.index = other.index @@ -212,6 +217,9 @@ protected abstract class MStack[@specialized T] private[util] ( } } + if (MStack.trackMaxSizeReached && other.maxSizeReached_ > this.maxSizeReached_) { + this.maxSizeReached_ = other.maxSizeReached_ + } } // private var currentIteratorIndex = -1 @@ -254,6 +262,9 @@ protected abstract class MStack[@specialized T] private[util] ( if (index == table.length) table = growArray(table) table(index) = x index += 1 + if (MStack.trackMaxSizeReached && index > maxSizeReached_) { + maxSizeReached_ = index + } } /** diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/SuspendableExpression.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/SuspendableExpression.scala index 0d9decbd98..0a7bca949f 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/SuspendableExpression.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/dpath/SuspendableExpression.scala @@ -32,12 +32,25 @@ import org.apache.daffodil.runtime1.processors.unparsers.UState * dfdl:setVariable expressions (which variables are in-turn used by * dfdl:outputValueCalc. */ +object SuspendableExpression { + + /** True unless expr calls valueLength/contentLength (which needs a + * written DOS position, not merely a known value); other reads + * resolve once the value is known. A static property of expr alone; + * it doesn't know whether the referenced position is already written. */ + def canResolveWithoutWriting(expr: CompiledExpression[AnyRef]): Boolean = + expr.valueReferencedElementInfos.isEmpty && expr.contentReferencedElementInfos.isEmpty +} + trait SuspendableExpression extends Suspension { override val isReadOnly = true protected def expr: CompiledExpression[AnyRef] + override def canResolveWithoutWriting: Boolean = + SuspendableExpression.canResolveWithoutWriting(expr) + override def toString = "SuspendableExpression(" + rd.diagnosticDebugName + ", expr=" + expr.prettyExpr + ")" diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetImpl.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetImpl.scala index 5edb5196e9..678fc9387f 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetImpl.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/infoset/InfosetImpl.scala @@ -176,13 +176,13 @@ sealed trait DINode { private var _isFinal: Boolean = false /** - * Use to mark a node as final, indicating that its value will not change or have - * any children added to it. Setting an element as final does not preclude it from - * being discarded by backtracking, i.e. it is only locally final, but might still - * be inside an enclosing PoU. - * - * This cannot be called if an element is already marked as final to help ensure - * correct use. + * Use to mark a node as final, indicating that its value will not change or have + * any children added to it. Setting an element as final does not preclude it from + * being discarded by backtracking, i.e. it is only locally final, but might still + * be inside an enclosing PoU. + * + * This cannot be called if an element is already marked as final to help ensure + * correct use. */ def setFinal(): Unit = { Assert.invariant(!_isFinal) @@ -1334,6 +1334,11 @@ final class DIArray( final def freeChildIfNoLongerNeeded(index: Int, doFree: Boolean): Unit = { val node = _contents(index) + // Under build/write-prefetch, both BuildState and write's writeContent + // dispatch can reach this slot; write may have already freed (nulled) it + // by the time build's independent call arrives here (build's doFree is + // always false, so this would only mark wouldHaveBeenFreed, moot here). + if (node == null) return if (!node.erd.dpathElementCompileInfo.isReferencedByExpressions) { if (doFree) { // set to null so that the garbage collector can free this node @@ -1825,6 +1830,9 @@ sealed class DIComplex(override val erd: ElementRuntimeData) def freeChildIfNoLongerNeeded(index: Int, doFree: Boolean): Unit = { val node = child(index) + // Under build/write-prefetch, write may have already freed (nulled) + // this slot by the time build's independent call arrives here. + if (node == null) return if (!node.erd.dpathElementCompileInfo.isReferencedByExpressions) { if (doFree) { // set to null so that the garbage collector can free this node diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala index 3b16d3680e..a47c353b5c 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/DataProcessor.scala @@ -68,8 +68,18 @@ import org.apache.daffodil.runtime1.infoset.XMLTextInfosetOutputter import org.apache.daffodil.runtime1.processors.parsers.PState import org.apache.daffodil.runtime1.processors.parsers.ParseError import org.apache.daffodil.runtime1.processors.parsers.Parser +import org.apache.daffodil.runtime1.processors.unparsers.AwaitChildStalledException +import org.apache.daffodil.runtime1.processors.unparsers.BuildCoroutine +import org.apache.daffodil.runtime1.processors.unparsers.BuildFinished +import org.apache.daffodil.runtime1.processors.unparsers.BuildState +import org.apache.daffodil.runtime1.processors.unparsers.NotUnparsableUnparser +import org.apache.daffodil.runtime1.processors.unparsers.SuspensionCapableUState import org.apache.daffodil.runtime1.processors.unparsers.UState import org.apache.daffodil.runtime1.processors.unparsers.UnparseError +import org.apache.daffodil.runtime1.processors.unparsers.UnparseSharedContext +import org.apache.daffodil.runtime1.processors.unparsers.WriteCoroutine +import org.apache.daffodil.runtime1.processors.unparsers.WriteDone +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase /** * Implementation mixin - provides simple helper methods @@ -458,6 +468,275 @@ class DataProcessor( } def unparse(actualInputter: api.infoset.InfosetInputter, outStream: java.io.OutputStream) = { + // hasAnyPrefetchBeneficialOVC is false only when every OVC is + // content-length-dependent, which prefetch could never resolve early + // regardless of how far build races ahead; fall back to single-pass + // automatically in that case, regardless of the tunable. + if (tunables.useBuildWritePrefetch && ssrd.hasAnyPrefetchBeneficialOVC) { + unparseViaBuildThenWrite(actualInputter, outStream) + } else { + unparseSinglePass(actualInputter, outStream) + } + } + + /** + * Shared by unparseViaBuildThenWrite and unparseSinglePass's top-level + * catch blocks: maps an exception caught during unparsing to a failed + * `state` plus its `unparseResult`, or rethrows if it's not one of the + * known unparse-error categories. + */ + private def unparseErrorResult(state: UState, t: Throwable): UnparseResult = t match { + case ue: UnparseError => { + state.addUnparseError(ue) + state.unparseResult + } + case procErr: ProcessingError => { + state.setFailed(procErr.toUnparseError) + state.unparseResult + } + case sde: SchemaDefinitionError => { + // A SDE was detected at runtime (perhaps due to a runtime-valued property like byteOrder or encoding) + // These are fatal, and there's no notion of backtracking them, so they propagate to top level + // here. + state.setFailed(sde) + state.unparseResult + } + case sdefw: SchemaDefinitionErrorFromWarning => { + state.setFailed(sdefw) + state.unparseResult + } + case e: ErrorAlreadyHandled => { + state.setFailed(e.th) + state.unparseResult + } + case e: TunableLimitExceededError => { + state.setFailed(e) + state.unparseResult + } + case se: org.xml.sax.SAXException => { + state.setFailed(new UnparseError(None, None, se)) + state.unparseResult + } + case e: scala.xml.parsing.FatalError => { + state.setFailed(new UnparseError(None, None, e)) + state.unparseResult + } + case ie: InfosetException => { + state.setFailed(new UnparseError(None, None, ie)) + state.unparseResult + } + case th: Throwable => throw th + } + + /** + * Build/write-prefetch unparse path (gated on `useBuildWritePrefetch`). + * `BuildState` builds the tree via the ordinary Unparser recursion; + * write runs on its own thread, recursing via `writeContent` and + * blocking on `awaitChild`, handed off via `Coroutine[T]`. + */ + private def unparseViaBuildThenWrite( + actualInputter: api.infoset.InfosetInputter, + outStream: java.io.OutputStream + ): UnparseResult = { + // A NotUnparsableUnparser (dfdl:parseUnparsePolicy="parseOnly") can't be + // cast to ElementUnparserBase or driven through the build/write split; + // unparseSinglePass already runs it via unparse1 and gets the correct + // diagnostic, so reuse that instead of a ClassCastException here. + ssrd.unparser match { + case _: NotUnparsableUnparser => return unparseSinglePass(actualInputter, outStream) + case _ => // fall through to the actual build/write-prefetch path below + } + val inputter = new InfosetInputter(actualInputter) + var buildState: BuildState = null + // Cleaned up on write's OWN thread (inside writeCoroutine's runBody + // below), not this method's outer finally block: it's only ever + // touched from write's thread (io.ThreadCheckMixin's affinity + // invariant), so cleaning it up elsewhere would violate that. + var writeState: UState with SuspensionCapableUState = null + // A valid fallback from the start: setup below can throw before + // buildState/writeState exist, and the catch block needs an actual + // UState to report through. Safe before initialize(); it only copies + // variableMap, never touching inputter.documentElement. + var activeState: UState = + UState.createInitialUState(outStream, this, inputter, areDebugging) + // Hoisted so the outer catch below can reach it (to abort write's + // thread if it's still parked) even when the exception came from + // partway through setup, before the try block's own local val would + // otherwise have gone out of scope. + var sharedCtx: UnparseSharedContext = null + try { + inputter.initialize(ssrd.elementRuntimeData, tunables) + + sharedCtx = new UnparseSharedContext( + inputter.documentElement, + new VariableBox(variableMap.copy()), + // Shared by build and write, which together tick this tracker at + // roughly twice the per-node rate a single traversal would; + // doubling both thresholds restores the intended sweep density. + new SuspensionTracker( + tunables.unparseSuspensionWaitYoung * 2, + tunables.unparseSuspensionWaitOld * 2 + ), + this, + tunables, + prefetchLimit = tunables.unparsePrefetchWindowNodes + ) + + val rootUnparser = ssrd.unparser + + // Constructs write's UState and wires it into sharedCtx. Must run on + // write's own coroutine thread: writeState's DataOutputStream exists + // 1-to-1 with threads (io.ThreadCheckMixin), so constructing it + // eagerly on build's thread would bind it to the wrong one. + def constructWriteSide(): Unit = { + writeState = UState.createInitialUStateForSharedVariables( + outStream, + this, + inputter, + sharedCtx.variableBox, + areDebugging + ) + writeState.setSharedContext(sharedCtx) + if (areDebugging) writeState.notifyDebugging(true) + init(writeState, rootUnparser) + writeState.getDataOutputStream.setPriorBitOrder(ssrd.elementRuntimeData.defaultBitOrder) + } + + // evalSuspensions(isFinal = true) runs BEFORE the stack-depth + // invariants below: one tripping first could mask the real + // SuspensionDeadlockException diagnostic this ordering exists to + // surface. + def finishWriteSide(): Unit = { + writeState.setProcessor(rootUnparser) + + // Routed via sharedCtx.suspensionTracker, the SAME tracker + // BuildState registered into, so both build-side and write-side + // suspensions get resolved here. + writeState.evalSuspensions(isFinal = true) + + Assert.invariant(writeState.arrayIterationIndexStack.length == 1) + Assert.invariant(writeState.occursIndexStack.length == 1) + Assert.invariant(writeState.groupIndexStack.length == 1) + Assert.invariant(writeState.childIndexStack.length == 1) + Assert.invariant(writeState.currentInfosetNodeMaybe.isEmpty) + Assert.invariant(writeState.escapeSchemeEVCache.isEmpty) + Assert.invariant(writeState.maybeTopTRD().isEmpty) + Assert.invariant(!writeState.withinHiddenNest) + + Assert.invariant(!writeState.getDataOutputStream.isFinished) + try { + writeState.getDataOutputStream.setFinished(writeState) + } catch { + case boc: BitOrderChangeException => + writeState.SDE(boc) + case fio: FileIOException => + writeState.SDE(fio) + } + } + + val buildCoroutine = new BuildCoroutine() + val writeCoroutine = new WriteCoroutine({ (wc, firstSignal) => + // Computed before resumeFinal is called below (never in an outer + // finally): resumeFinal requires returning from run() immediately, + // so build's thread mustn't wake until write's cleanup has + // actually finished, or both threads would run at once. + var result: WriteDone = WriteDone(None) + try { + // firstSignal may already be BuildFinished (build never crossed + // the prefetch-lead threshold) or BuildAborted (build failed + // early); observeBuildSignal handles either before any real work. + sharedCtx.observeBuildSignal(firstSignal) + constructWriteSide() + val rootElemUnp = rootUnparser.asInstanceOf[ElementUnparserBase] + try { + val rootNode = sharedCtx.awaitChild(inputter.documentElement, 0) + rootElemUnp.writeContent(rootNode, writeState) + } catch { + // A genuine deadlock (if any) surfaces via finishWriteSide's + // own evalSuspensions(isFinal = true) call below, which runs + // regardless of how this try block exits. + case _: AwaitChildStalledException => + } + finishWriteSide() + } catch { + // Includes BuildAbortedException: by the time this fires, + // build's thread has already unwound via its own catch below, + // so this WriteDone is never actually read by anyone. + case t: Throwable => result = WriteDone(Some(t)) + } finally { + if (writeState != null) writeState.getDataOutputStream.cleanUp() + } + wc.resumeFinal(buildCoroutine, result) + }) + sharedCtx.setCoroutines(buildCoroutine, writeCoroutine) + + buildState = new BuildState(inputter, sharedCtx, Nil, areDebugging) + activeState = buildState + if (areDebugging) { + Assert.invariant(optDebugger.isDefined) + addEventHandler(debugger) + buildState.notifyDebugging(true) + } + init(buildState, rootUnparser) + buildState.initializeVariables() + + rootUnparser.unparse1(buildState) + buildState.popTRD(rootUnparser.context.asInstanceOf[TermRuntimeData]) + buildState.setProcessor(rootUnparser) + + Assert.invariant(buildState.arrayIterationIndexStack.length == 1) + Assert.invariant(buildState.occursIndexStack.length == 1) + Assert.invariant(buildState.groupIndexStack.length == 1) + Assert.invariant(buildState.childIndexStack.length == 1) + Assert.invariant(buildState.currentInfosetNodeMaybe.isEmpty) + Assert.invariant(buildState.maybeTopTRD().isEmpty) + + val remainingEvent = buildState.advanceMaybe + if (remainingEvent.isDefined) { + UnparseError( + Nope, + One(buildState.currentLocation), + "Expected no remaining events, but received %s.", + remainingEvent.get + ) + } + + // The tree is now entirely built, but some suspensions may still be + // pending. resumeWrite(BuildFinished) covers both the case where + // write's thread was never spawned (this is its first resume) and + // where it already finished mid-build. + val finalSignal = sharedCtx.resumeWrite(BuildFinished) + // Only reassign once writeState is known-good: a null here means + // constructWriteSide() itself failed (reported via finalSignal + // below), so keep reporting through buildState instead. + if (writeState != null) activeState = writeState + finalSignal match { + case WriteDone(Some(t)) => throw t + case WriteDone(None) => // writeState.unparseResult below + case other => + Assert.invariantFailed( + s"write coroutine yielded $other instead of signaling completion" + ) + } + + writeState.unparseResult + } catch { + case t: Throwable => + // If write's thread is still parked (this exception came from + // build's own recursion, before the normal resumeWrite(BuildFinished) + // handoff ran), wake it so it can clean up instead of leaking its + // thread and DOS temp files forever. + if (sharedCtx != null) sharedCtx.abortWrite() + unparseErrorResult(activeState, t) + } finally { + if (buildState != null) buildState.getDataOutputStream.cleanUp() + } + } + + private def unparseSinglePass( + actualInputter: api.infoset.InfosetInputter, + outStream: java.io.OutputStream + ) = { val inputter = new InfosetInputter(actualInputter) val unparserState = UState.createInitialUState(outStream, this, inputter, areDebugging) @@ -477,47 +756,7 @@ class DataProcessor( unparserState.evalSuspensions(isFinal = true) unparserState.unparseResult } catch { - case ue: UnparseError => { - unparserState.addUnparseError(ue) - unparserState.unparseResult - } - case procErr: ProcessingError => { - val x = procErr - unparserState.setFailed(x.toUnparseError) - unparserState.unparseResult - } - case sde: SchemaDefinitionError => { - // A SDE was detected at runtime (perhaps due to a runtime-valued property like byteOrder or encoding) - // These are fatal, and there's no notion of backtracking them, so they propagate to top level - // here. - unparserState.setFailed(sde) - unparserState.unparseResult - } - case sdefw: SchemaDefinitionErrorFromWarning => { - unparserState.setFailed(sdefw) - unparserState.unparseResult - } - case e: ErrorAlreadyHandled => { - unparserState.setFailed(e.th) - unparserState.unparseResult - } - case e: TunableLimitExceededError => { - unparserState.setFailed(e) - unparserState.unparseResult - } - case se: org.xml.sax.SAXException => { - unparserState.setFailed(new UnparseError(None, None, se)) - unparserState.unparseResult - } - case e: scala.xml.parsing.FatalError => { - unparserState.setFailed(new UnparseError(None, None, e)) - unparserState.unparseResult - } - case ie: InfosetException => { - unparserState.setFailed(new UnparseError(None, None, ie)) - unparserState.unparseResult - } - case th: Throwable => throw th + case t: Throwable => unparseErrorResult(unparserState, t) } finally { unparserState.getDataOutputStream.cleanUp() } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala index e830aead7b..9d798ead48 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SchemaSetRuntimeData.scala @@ -33,7 +33,12 @@ final class SchemaSetRuntimeData( */ variables: VariableMap, allLayers: Seq[LayerRuntimeData], - @transient layerRuntimeCompilerArg: LayerRuntimeCompiler + @transient layerRuntimeCompilerArg: LayerRuntimeCompiler, + /** True if this schema has an outputValueCalc element whose value could + * resolve without writing (see hasAnyPrefetchBeneficialOVC); baked in at + * compile time like the rest of this class. Gates useBuildWritePrefetch - + * zero benefit means automatic fallback to single-pass. */ + val hasAnyPrefetchBeneficialOVC: Boolean ) extends Serializable with ThrowsSDE { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/Suspension.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/Suspension.scala index d41b3accfa..d7d3d20007 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/Suspension.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/Suspension.scala @@ -25,8 +25,8 @@ import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.lib.util.MaybeInt import org.apache.daffodil.lib.util.MaybeULong +import org.apache.daffodil.runtime1.processors.unparsers.SuspensionCapableUState import org.apache.daffodil.runtime1.processors.unparsers.UState -import org.apache.daffodil.runtime1.processors.unparsers.UStateMain import org.apache.daffodil.runtime1.processors.unparsers.UnparseError /** @@ -51,6 +51,17 @@ trait Suspension extends Serializable { */ val isReadOnly = false + /** + * True if this suspension might resolve without any bytes written yet + * (e.g. value/variable read); false if it needs a real DOS bit position + * (e.g. valueLength/contentLength, padding/alignment); distinct from + * isReadOnly. A static, direction-blind heuristic (can't tell a + * backward reference to an already-resolved value from a forward + * one), used by build's retries to skip an attempt that's usually, + * but not always, doomed to block. + */ + def canResolveWithoutWriting: Boolean = false + def UE(ustate: UState, s: String, args: Any*) = { UnparseError(One(rd.schemaFileLocation), One(ustate.currentLocation), s, args*) } @@ -71,10 +82,9 @@ trait Suspension extends Serializable { protected def doTask(ustate: UState): Unit /** - * After calling this, call isDone and if that's false call isMakingProgress to - * understand whether it is done, blocked on the exactly same situation, or blocked elsewhere. - * - * This status is needed to implement circular deadlock detection + * After calling this, call isDone and if that's false call isMakingProgress + * to understand whether it is done, blocked on the exactly same situation, + * or blocked elsewhere. Needed for circular deadlock detection. */ final def runSuspension(): Unit = { doTask(savedUstate) @@ -104,6 +114,15 @@ trait Suspension extends Serializable { } } + /** + * Entry point to prepareToSuspend for callers using a static heuristic + * that says the first attempt is likely to block. Skips only that + * attempt; state-patching still runs at the same freeze point as a + * genuine block, so a heuristic false positive costs an avoidable + * suspend/resume, never a correctness problem. + */ + final def suspendWithoutAttempting(ustate: UState): Unit = prepareToSuspend(ustate) + private def prepareToSuspend(ustate: UState): Unit = { val mkl = maybeKnownLengthInBits(ustate) // @@ -111,7 +130,7 @@ trait Suspension extends Serializable { // // As written, we have a bunch of suspensions that occur, but have // specifically known length of zero bits. So nothing being written out. - // In that case, why do we need to split at all? + // TODO: In that case, why do we need to split at all? // val original = ustate.getDataOutputStream if (mkl.isEmpty || (mkl.isDefined && mkl.get > 0)) { @@ -175,18 +194,20 @@ trait Suspension extends Serializable { // // clone the ustate for use when evaluating the expression // - // TODO: Performance - copying this whole state, just for OVC is painful. - // Some sort of copy-on-write scheme would be better. + // This is a targeted partial clone (shallow VariableMap copy, stack + // tops only), not a full deep copy, but still copies the full + // escapeSchemeEVCache/delimiterStack contents unconditionally. + // TODO: Performance: a copy-on-write scheme could avoid that copy. // val didSplit = (ustate.getDataOutputStream ne original) - val cloneUState = ustate.asInstanceOf[UStateMain].cloneForSuspension(original) + val cloneUState = ustate.asInstanceOf[SuspensionCapableUState].cloneForSuspension(original) if (isReadOnly && didSplit) { Assert.invariantFailed("Shouldn't have split. read-only case") } savedUstate_ = cloneUState - ustate.asInstanceOf[UStateMain].addSuspension(this) + ustate.asInstanceOf[SuspensionCapableUState].addSuspension(this) } final def explain(): Unit = { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SuspensionTracker.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SuspensionTracker.scala index b38ba53909..4e063110bf 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SuspensionTracker.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/SuspensionTracker.scala @@ -30,6 +30,13 @@ class SuspensionTracker(suspensionWaitYoung: Int, suspensionWaitOld: Int) { def suspensions: Seq[Suspension] = suspensionsYoung.toSeq ++ suspensionsOld.toSeq + /** + * Total not-yet-done suspensions, without allocating (unlike + * `suspensions`, not safe to call once per node). A throttle signal for + * pacing a no-op-sink traversal against the pending backlog. + */ + def pendingCount: Int = suspensionsYoung.length + suspensionsOld.length + private var count: Int = 0 private var suspensionStatTracked: Int = 0 @@ -47,12 +54,24 @@ class SuspensionTracker(suspensionWaitYoung: Int, suspensionWaitOld: Int) { * suspensions, we attempt to evaluate them first, with the hope that their * resolution might make the young suspensions more likely to evaluate. */ - def evalSuspensions(): Unit = { + def evalSuspensions(): Unit = evalSuspensionsThrottled(buildResolvableOnly = false) + + /** + * A no-op-sink sweep variant: same cadence as evalSuspensions, but with + * buildResolvableOnly=true; a suspension whose canResolveWithoutWriting + * is false is skipped-and-requeued instead of genuinely attempted, since + * this heuristic marks it as usually unresolvable here; it stays pending + * for a later unfiltered sweep either way. + */ + def evalBuildResolvableSuspensions(): Unit = + evalSuspensionsThrottled(buildResolvableOnly = true) + + private def evalSuspensionsThrottled(buildResolvableOnly: Boolean): Unit = { if (count % suspensionWaitOld == 0) { evalSuspensionQueue(suspensionsOld) } if (count % suspensionWaitYoung == 0) { - evalSuspensionQueue(suspensionsYoung) + evalSuspensionQueue(suspensionsYoung, buildResolvableOnly = buildResolvableOnly) while (suspensionsYoung.nonEmpty) { suspensionsOld.enqueue(suspensionsYoung.dequeue()) } @@ -65,6 +84,20 @@ class SuspensionTracker(suspensionWaitYoung: Int, suspensionWaitOld: Int) { } } + /** + * Attempts every currently-tracked suspension once, bypassing throttling, + * without treating remaining blocks as an error. Intended for a caller + * whose own traversal has fully finished and needs an already-resolvable + * suspension to actually resolve before it can continue. + */ + def evalSuspensionsUnthrottled(): Unit = { + evalSuspensionQueue(suspensionsOld) + evalSuspensionQueue(suspensionsYoung) + while (suspensionsYoung.nonEmpty) { + suspensionsOld.enqueue(suspensionsYoung.dequeue()) + } + } + /** * Evaluates all suspensions until either they are all evaluated or a * deadlock is detected. This moves all young suspensions to the old queue, @@ -94,23 +127,34 @@ class SuspensionTracker(suspensionWaitYoung: Int, suspensionWaitOld: Int) { } /** - * Attempt to evaluate suspensions on the provie queue. Keep repeating the - * evaluates as long as some progress is being made. Suspensions that - * evaluate sucessfully are removed from the queue. Once suspensions make no - * further progress and are all blocked, we return. Blocked suspensions put - * back on the same queue. + * Attempt to evaluate suspensions on the provided queue. Keep repeating + * the evaluates as long as some progress is being made. Suspensions that + * evaluate successfully are removed from the queue. Once suspensions make + * no further progress and are all blocked, we return. Blocked suspensions + * are put back on the same queue. buildResolvableOnly diverts a + * canResolveWithoutWriting=false suspension into skip-and-requeue instead + * of genuinely attempting it, since this heuristic marks it as usually + * unresolvable here. */ - private def evalSuspensionQueue(queue: Queue[Suspension]): Unit = { + private def evalSuspensionQueue( + queue: Queue[Suspension], + buildResolvableOnly: Boolean = false + ): Unit = { var countOfNotMakingProgress = 0 while (!queue.isEmpty && countOfNotMakingProgress < queue.length) { val s = queue.dequeue() - suspensionStatRuns += 1 - s.runSuspension() - if (!s.isDone) queue.enqueue(s) - if (s.isDone || s.isMakingProgress) { - countOfNotMakingProgress = 0 - } else { + if (buildResolvableOnly && !s.canResolveWithoutWriting) { + queue.enqueue(s) countOfNotMakingProgress += 1 + } else { + suspensionStatRuns += 1 + s.runSuspension() + if (!s.isDone) queue.enqueue(s) + if (s.isDone || s.isMakingProgress) { + countOfNotMakingProgress = 0 + } else { + countOfNotMakingProgress += 1 + } } } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/VariableMap1.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/VariableMap1.scala index 7d8cb037ef..1ffbbfca78 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/VariableMap1.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/VariableMap1.scala @@ -181,8 +181,10 @@ class VariableSuspended(qname: NamedQName, context: VariableRuntimeData) * expressions in a defineVariable are circular, or later during parsing if * newVariableInstance contains a circular expression */ -class VariableCircularDefinition(qname: NamedQName, context: VariableRuntimeData) - extends VariableException( +class VariableCircularDefinition( + qname: NamedQName, + context: VariableRuntimeData +) extends VariableException( qname, context, s"Variable map (runtime): variable $qname is part of a circular definition with other variables" @@ -284,6 +286,29 @@ class VariableMap private ( new VariableMap(vrds, newTable) } + /** + * Used only by BuildState.cloneForSuspension, on a clone just produced + * above, never the live VariableMap. Corrects the frozen snapshot's head + * at vmapIndex to the instance build's recursion currently considers + * active, since a later retry replays only against this frozen clone. + */ + def overrideHeadForSuspensionClone(vmapIndex: Int, instance: VariableInstance): Unit = { + // Replaces rather than prepends: this clone is never pushed/popped + // again via a real NVI scope change, so a stale extra head would + // inflate vTable(vmapIndex).size by one, tripping setVariable's "more + // than one instance in scope" guard later even with no scope actually + // open on this clone. + vTable(vmapIndex) = instance +: vTable(vmapIndex).tail + } + + /** + * The pre-NVI default instance for vmapIndex, used by BuildState + * whenever build's local NVI scope stack for this index is empty. + * newVariableInstance always prepends and only write's later + * removeVariableInstance call ever pops the shared array, so this is always the LAST entry regardless of how many NVI scopes build has pushed (and locally popped) since. + */ + def originalInstanceAt(vmapIndex: Int): VariableInstance = vTable(vmapIndex).last + // For defineVariable's with non-constant expressions for default values, it // is necessary to force the evaluation of the expressions after the // VariableMap has been created and initialized, but before parsing begins. We @@ -361,18 +386,7 @@ class VariableMap private ( referringContext: ThrowsSDE, state: ParseOrUnparseState ): DataValuePrimitive = { - val varQName = vrd.globalQName - vrd.direction match { - case VariableDirection.ParseOnly if (!state.isInstanceOf[PState]) => - state.SDE( - s"Attempting to read variable $varQName which is marked as parseOnly during unparsing" - ) - case VariableDirection.UnparseOnly if (!state.isInstanceOf[UState]) => - state.SDE( - s"Attempting to read variable $varQName which is marked as unparseOnly during parsing" - ) - case _ => // Do nothing - } + checkDirectionForRead(vrd, state) val variable = { // The vrd.vmapIndex cannot be out of range of the vTable, because the vTable size @@ -391,6 +405,43 @@ class VariableMap private ( } varAtIndex } + readVariable(variable, vrd, referringContext, state) + } + + /** + * Direction-eligibility check shared by the live-stack lookup above and + * BuildState's build-local scope lookup (see BuildState.scala); both need + * the same parseOnly/unparseOnly guard regardless of which + * VariableInstance the read ultimately resolves against. + */ + def checkDirectionForRead(vrd: VariableRuntimeData, state: ParseOrUnparseState): Unit = { + val varQName = vrd.globalQName + vrd.direction match { + case VariableDirection.ParseOnly if (!state.isInstanceOf[PState]) => + state.SDE( + s"Attempting to read variable $varQName which is marked as parseOnly during unparsing" + ) + case VariableDirection.UnparseOnly if (!state.isInstanceOf[UState]) => + state.SDE( + s"Attempting to read variable $varQName which is marked as unparseOnly during parsing" + ) + case _ => // Do nothing + } + } + + /** + * Overload of readVariable above that runs the variable-state machine + * against a specific, caller-supplied VariableInstance rather than + * always the shared vTable's head. BuildState overrides its read path + * to target whichever instance build's recursion currently has open, since the shared vTable's head can lag behind build's nesting once more than one newVariableInstance for the same variable is pushed there. + */ + def readVariable( + variable: VariableInstance, + vrd: VariableRuntimeData, + referringContext: ThrowsSDE, + state: ParseOrUnparseState + ): DataValuePrimitive = { + val varQName = vrd.globalQName variable.state match { case VariableRead if (variable.value.isDefined) => variable.value.getNonNullable case VariableDefined | VariableSet if (variable.value.isDefined) => { @@ -422,7 +473,10 @@ class VariableMap private ( } /** - * Assigns a variable and sets the variables state to VariableSet + * Assigns a variable and sets its state to VariableSet, against the + * shared vTable's head. BuildState overrides state.setVariable to + * instead resolve against whichever instance build's recursion + * currently has open, via the instance-targeted overload below. */ def setVariable( vrd: VariableRuntimeData, @@ -430,9 +484,24 @@ class VariableMap private ( referringContext: ThrowsSDE, pstate: ParseOrUnparseState ): Unit = { - val varQName = vrd.globalQName val variableInstances = vTable(vrd.vmapIndex) - val variable = variableInstances.head + setVariable(variableInstances.head, vrd, newValue, referringContext, pstate) + } + + /** + * Overload of setVariable above that runs the variable-set state + * machine against a specific, caller-supplied VariableInstance rather + * than always the shared vTable's head; see BuildState's + * state.setVariable override (BuildState.scala). + */ + def setVariable( + variable: VariableInstance, + vrd: VariableRuntimeData, + newValue: DataValuePrimitive, + referringContext: ThrowsSDE, + pstate: ParseOrUnparseState + ): Unit = { + val varQName = vrd.globalQName variable.state match { case VariableSet => { referringContext.SDE( @@ -471,7 +540,7 @@ class VariableMap private ( * variable. */ case VariableDirection.UnparseOnly | VariableDirection.Both - if (vrd.maybeDefaultValueExpr.isDefined && variableInstances.size > 1) => { + if (vrd.maybeDefaultValueExpr.isDefined && vTable(vrd.vmapIndex).size > 1) => { // Variable has an unparse direction, a default value, and a // newVariableInstance pstate.SDE( diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildState.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildState.scala new file mode 100644 index 0000000000..1852df167c --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildState.scala @@ -0,0 +1,314 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream +import scala.collection.mutable + +import org.apache.daffodil.api +import org.apache.daffodil.io.DirectOrBufferedDataOutputStream +import org.apache.daffodil.io.StringDataInputStreamForUnparse +import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.exceptions.ThrowsSDE +import org.apache.daffodil.lib.util.LocalStack +import org.apache.daffodil.lib.util.MStackOfMaybe +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.Nope +import org.apache.daffodil.runtime1.dpath.UnparserBlocking +import org.apache.daffodil.runtime1.infoset.DIDocument +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.infoset.DataValue.DataValuePrimitive +import org.apache.daffodil.runtime1.infoset.InfosetAccessor +import org.apache.daffodil.runtime1.infoset.InfosetInputter +import org.apache.daffodil.runtime1.processors.DelimiterStackUnparseNode +import org.apache.daffodil.runtime1.processors.EscapeSchemeUnparserHelper +import org.apache.daffodil.runtime1.processors.Suspension +import org.apache.daffodil.runtime1.processors.SuspensionTracker +import org.apache.daffodil.runtime1.processors.TermRuntimeData +import org.apache.daffodil.runtime1.processors.VariableInstance +import org.apache.daffodil.runtime1.processors.VariableRuntimeData +import org.apache.daffodil.runtime1.processors.dfa.DFADelimiter + +/** + * A `UState` subclass that consumes an actual `InfosetInputter` and provides + * live Cursor/TRD/index-stack behavior; this is the "build" side of the + * build/write unparse split. The write-only surface (delimiter stack, + * escape scheme cache, and the scratch buffers used for measuring/escaping + * text) is stubbed to error, since nothing build does should ever touch it; + * build never writes content. + * + * `getDataOutputStream` is NOT stubbed: the suspend path reads it + * unconditionally, even for read-only (never-actually-written) + * suspensions, so `BuildState` constructs an actual DOS wrapping a no-op + * sink purely to satisfy that. + * + * Used only when the `useBuildWritePrefetch` tunable is enabled (default + * false); otherwise unused, and unparsing constructs `UStateMain` + * exclusively as before. + */ +final class BuildState( + private val inputter: InfosetInputter, + sharedCtx: UnparseSharedContext, + diagnosticsArg: Seq[api.Diagnostic], + areDebugging: Boolean +) extends UState( + sharedCtx.variableBox, + diagnosticsArg, + Maybe(sharedCtx.dataProc), + sharedCtx.tunable, + areDebugging + ) + with SuspensionCapableUState + with LiveIndexStacks { + + dState.setMode(UnparserBlocking) + setSharedContext(sharedCtx) + + // Purely to satisfy the unconditional getDataOutputStream read on + // the suspend path (see class doc above). Never actually written + // to: build's suspensions are always read-only + // (maybeKnownLengthInBits == 0), so splitDOS never runs for them. + setDataOutputStream( + DirectOrBufferedDataOutputStream( + new java.io.OutputStream { override def write(b: Int): Unit = () }, + null, + false, + sharedCtx.tunable.outputStreamChunkSizeInBytes, + sharedCtx.tunable.maxByteArrayOutputStreamBufferSizeInBytes, + sharedCtx.tunable.tempFilePath + ) + ) + // Sets up bit order the same way single-pass unparse does, for + // parity with write/single-pass, even though build's own isBuildOnly + // gates mean it never actually needs to check bit order during build. + getDataOutputStream.setPriorBitOrder( + sharedCtx.dataProc.ssrd.elementRuntimeData.defaultBitOrder + ) + + override def isBuildOnly: Boolean = true + + private def writeOnly = + Assert.usageError("BuildState never writes content, so this write-only state doesn't exist") + + override def escapeSchemeEVCache: MStackOfMaybe[EscapeSchemeUnparserHelper] = writeOnly + override def withUnparserDataInputStream: LocalStack[StringDataInputStreamForUnparse] = + writeOnly + override def withByteArrayOutputStream + : LocalStack[(ByteArrayOutputStream, DirectOrBufferedDataOutputStream)] = writeOnly + override def allTerminatingMarkup: List[DFADelimiter] = writeOnly + override def localDelimiters: DelimiterStackUnparseNode = writeOnly + override def pushDelimiters(node: DelimiterStackUnparseNode): Unit = writeOnly + override def popDelimiters(): Unit = writeOnly + + override def advance: Boolean = inputter.advance + override def advanceAccessor: InfosetAccessor = inputter.advanceAccessor + override def inspect: Boolean = inputter.inspect + override def inspectAccessor: InfosetAccessor = inputter.inspectAccessor + override def fini(): Unit = Assert.usageError("Not to be used on UState") + + override def inspectOrError: InfosetAccessor = { + if (inspect) { + inspectAccessor + } else { + Assert.invariantFailed( + "An InfosetEvent was required for building, but no InfosetEvent was available." + ) + } + } + + override def advanceOrError: InfosetAccessor = { + if (advance) { + advanceAccessor + } else { + Assert.invariantFailed( + "An InfosetEvent was required for building, but no InfosetEvent was available." + ) + } + } + + override def isInspectArrayEnd: Boolean = { + if (!inspect) { + false + } else { + inspectAccessor match { + case e if e.isEnd && e.isArray => true + case _ => false + } + } + } + + def currentInfosetNode: DINode = { + if (currentInfosetNodeMaybe.isEmpty) { + null + } else { + currentInfosetNodeMaybe.get + } + } + + def currentInfosetNodeMaybe: Maybe[DINode] = { + if (currentInfosetNodeStack.isEmpty) { + Nope + } else { + currentInfosetNodeStack.top + } + } + + override val currentInfosetNodeStack = new MStackOfMaybe[DINode] + + /** + * Build-local NVI (`dfdl:newVariableInstance`) scope tracking, keyed by + * `vmapIndex`, closing each scope the moment build's own traversal exits + * it. Kept separate from the shared `VariableMap`/`vTable`, which only + * write pops, much later, so the shared array's head can lag behind. + */ + private val nviLocalStacks: mutable.Map[Int, List[VariableInstance]] = mutable.Map.empty + + /** + * Pushes onto the shared vTable exactly as the base implementation + * does, then additionally records the same instance on build's own + * local scope stack. + */ + override def newVariableInstance(vrd: VariableRuntimeData): VariableInstance = { + val nvi = super.newVariableInstance(vrd) + val idx = vrd.vmapIndex + nviLocalStacks(idx) = nvi :: nviLocalStacks.getOrElse(idx, Nil) + nvi + } + + /** + * Pops build's own local scope stack only; the shared vTable is left + * untouched here, since write's own end-of-scope handling pops that, + * later, via the unmodified base `removeVariableInstance`. + */ + override def removeVariableInstance(vrd: VariableRuntimeData): Unit = { + val idx = vrd.vmapIndex + val stack = nviLocalStacks.getOrElse(idx, Nil) + Assert.invariant(stack.nonEmpty) + nviLocalStacks(idx) = stack.tail + } + + /** + * Resolves against whichever NVI instance build currently has open, + * rather than the shared vTable's head (which can lag behind). Falls + * back to the original (pre-NVI) instance once build has locally + * exited every NVI scope for this vmapIndex. + */ + override def getVariable( + vrd: VariableRuntimeData, + referringContext: ThrowsSDE + ): DataValuePrimitive = { + variableMap.checkDirectionForRead(vrd, this) + nviLocalStacks.getOrElse(vrd.vmapIndex, Nil) match { + case head :: _ => variableMap.readVariable(head, vrd, referringContext, this) + case Nil => + variableMap.readVariable( + variableMap.originalInstanceAt(vrd.vmapIndex), + vrd, + referringContext, + this + ) + } + } + + /** + * Mirrors getVariable above: must land on whichever NVI instance build + * currently has open, not the shared vTable's head; write's deferred + * End pop can race ahead of it, tripping a spurious "cannot set + * variable twice" SDE on the wrong scope. + */ + override def setVariable( + vrd: VariableRuntimeData, + newValue: DataValuePrimitive, + referringContext: ThrowsSDE + ): Unit = { + nviLocalStacks.getOrElse(vrd.vmapIndex, Nil) match { + case head :: _ => + variableMap.setVariable(head, vrd, newValue, referringContext, this) + case Nil => + variableMap.setVariable( + variableMap.originalInstanceAt(vrd.vmapIndex), + vrd, + newValue, + referringContext, + this + ) + } + } + + // Shared, not owned; one SuspensionTracker queue, both build and write + // see the same one via sharedCtx. + def suspensionTracker: SuspensionTracker = sharedCtx.suspensionTracker + def addSuspension(se: Suspension): Unit = sharedCtx.suspensionTracker.trackSuspension(se) + + /** + * Uses evalBuildResolvableSuspensions: canResolveWithoutWriting is a + * static, direction-blind heuristic, so a suspension it marks false + * (usually a forward reference, occasionally an already-resolved + * backward one) is skipped here rather than genuinely retried, and + * left pending for a later, unfiltered sweep instead. + */ + def evalSuspensions(isFinal: Boolean): Unit = { + sharedCtx.suspensionTracker.evalBuildResolvableSuspensions() + if (isFinal) sharedCtx.suspensionTracker.requireFinal() + } + def suspensions = sharedCtx.suspensionTracker.suspensions + + /** + * Simpler than the base implementation's suspension clone: no + * escape-scheme/delimiter state to clone. Patches the clone's vTable, + * per open NVI scope, to the SAME instance build's own read resolved + * against, not the shared array's raw head, or every retry would + * target the wrong, stale scope. + */ + override def cloneForSuspension(suspendedDOS: DirectOrBufferedDataOutputStream): UState = { + val clonedBox = sharedCtx.variableBox.cloneForSuspension() + nviLocalStacks.foreach { + case (idx, head :: _) => clonedBox.vmap.overrideHeadForSuspensionClone(idx, head) + case (idx, Nil) => + // Build locally exited every NVI scope for this index; target the + // original instance the read resolved against, not whatever the + // shared array's raw head happens to be. + clonedBox.vmap.overrideHeadForSuspensionClone(idx, variableMap.originalInstanceAt(idx)) + } + val clone = new UStateForSuspension( + this, + suspendedDOS, + clonedBox, + currentInfosetNodeStack.top.get, + arrayIterationIndexStack.top, + occursIndexStack.top, + Nope, + Nope, + sharedCtx.tunable, + areDebugging + ) + clone.setProcessor(processor) + clone + } + + final override def pushTRD(trd: TermRuntimeData): Unit = inputter.pushTRD(trd) + final override def maybeTopTRD(): Maybe[TermRuntimeData] = inputter.maybeTopTRD() + final override def popTRD(trd: TermRuntimeData): TermRuntimeData = { + val poppedTRD = inputter.popTRD() + if (poppedTRD ne trd) + Assert.invariantFailed("TRDs do not match. Expected: " + trd + " got " + poppedTRD) + poppedTRD + } + + final override def documentElement: DIDocument = inputter.documentElement +} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildWriteCoroutines.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildWriteCoroutines.scala new file mode 100644 index 0000000000..3fa16c6b39 --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildWriteCoroutines.scala @@ -0,0 +1,113 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.lib.util.Coroutine +import org.apache.daffodil.lib.util.MainCoroutine + +/* + * The build/write handoff for build/write-prefetch: build and write's + * own content dispatch hand off via Coroutine[T], guaranteeing only one + * of the pair ever runs at a time, via a blocking-queue rendezvous + * between two threads. + */ + +/** Signals build sends to write when resuming it. */ +sealed trait BuildSignal + +/** Build made progress since write last ran; try again. */ +case object MoreTreeAvailable extends BuildSignal + +/** + * Build's top-level recursion has fully returned; no further nodes will + * ever be added. Sent exactly once, as build's final handoff; possibly + * as the very FIRST signal write's coroutine ever receives, if build's + * whole recursion never crossed the prefetch-lead threshold (small + * enough document that write's thread was never spawned before build + * finished). Write's loop must check for this on the result of ANY + * resume, including the one that started its thread, not just as a + * distinct "later" case. + */ +case object BuildFinished extends BuildSignal + +/** + * Build's own thread failed with an exception before ever reaching its + * normal `BuildFinished` handoff (anywhere inside build's own top-level + * recursion, or the invariant checks immediately around it). Write's + * thread is guaranteed to be parked at this exact moment (build and + * write never run concurrently), so this is how build's own exception + * handling wakes it back up instead of leaving it blocked forever. + * Write's own normal finalization assumes a consistently, fully-built + * tree that a genuine abort may not have produced, so this signal skips + * that path entirely; build's own exception, not anything from write's + * side, is what actually gets reported to the caller. + */ +case object BuildAborted extends BuildSignal + +/** Signals write sends to build when yielding control back. */ +sealed trait WriteSignal + +/** + * Write's own recursive dispatch has blocked, needing more tree before + * it can continue; build should keep building (and will be resumed + * again itself once it does, or once build's own recursion finishes and + * sends `BuildFinished` instead). + */ +case object WriteNeedsMore extends WriteSignal + +/** + * Write has finished all remaining work, or failed trying to. Sent + * exactly once, as write's last act: call `resumeFinal`, then return + * from `run()` immediately (its own contract). `error` is the captured + * exception if write's work failed, so it can be re-thrown on build's + * thread and handled uniformly with an exception thrown directly there. + */ +final case class WriteDone(error: Option[Throwable]) extends WriteSignal + +/** + * Build's side of the handoff. Runs on the ORIGINAL calling thread + * (`isMain = true`, so no thread is spawned for it); build already drives + * everything from the caller's thread; only write gets a genuinely new + * thread (`WriteCoroutine` below). + */ +final class BuildCoroutine extends MainCoroutine[WriteSignal] + +/** + * Write's side of the handoff. An actual, separate thread, spawned lazily + * (via `Coroutine`'s own `init()`) the first time build resumes it. + * `runBody` (the caller-supplied build/write driver) must construct + * write's own `UState` as its FIRST action on this thread, not before it + * starts and not on the build thread, since `DataOutputStream` is + * single-thread-affine; constructing it eagerly on build's thread would + * bind it to the wrong one. + * + * `runBody` receives the signal that started this thread as its second + * argument, rather than `run()` silently discarding it, because that + * first signal might already be `BuildFinished`; `runBody` must treat it + * like any later resume result, not assume the first is always + * `MoreTreeAvailable`. It must end by calling + * `resumeFinal(buildCoroutine, WriteDone(...))` (call it, then return + * from `run()` immediately; its own contract). + */ +final class WriteCoroutine(runBody: (WriteCoroutine, BuildSignal) => Unit) + extends Coroutine[BuildSignal] { + override protected def run(): Unit = { + val firstSignal = waitForResume() + runBody(this, firstSignal) + } +} diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala index 27a4381877..be694a4bc7 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UState.scala @@ -81,7 +81,13 @@ abstract class UState( with ThrowsSDE with SavesErrorsAndWarnings { - final override def setVariable( + /** + * Not final: BuildState overrides this to target its build-local NVI + * scope stack, since the shared vTable's head is popped only by write, + * much later, and could otherwise land a build-time set on the wrong, + * already-exited scope. See BuildState.scala. + */ + override def setVariable( vrd: VariableRuntimeData, newValue: DataValuePrimitive, referringContext: ThrowsSDE @@ -89,22 +95,23 @@ abstract class UState( vbox.vmap.setVariable(vrd, newValue, referringContext, this) /** - * For unparsing, this throws a RetryableException in the case where the variable cannot (yet) be read. - * - * @param vrd Identifies the variable to read. - * @param referringContext Where to place blame if there is an error. - * @return The data value of the variable, or throws exceptions if there is no value. + * For unparsing, throws a RetryableException if the variable can't yet + * be read. Not final: BuildState overrides this to resolve reads + * against its build-local NVI scope stack, since the shared vTable's + * head (popped only by write, much later) can still show an already-exited scope. See BuildState.scala. */ - final override def getVariable( + override def getVariable( vrd: VariableRuntimeData, referringContext: ThrowsSDE ): DataValuePrimitive = vbox.vmap.readVariable(vrd, referringContext, this) - final override def newVariableInstance(vrd: VariableRuntimeData): VariableInstance = + // Not final: see BuildState's override. + override def newVariableInstance(vrd: VariableRuntimeData): VariableInstance = variableMap.newVariableInstance(vrd) - final override def removeVariableInstance(vrd: VariableRuntimeData): Unit = + // Not final: see BuildState's override. + override def removeVariableInstance(vrd: VariableRuntimeData): Unit = variableMap.removeVariableInstance(vrd) /** @@ -386,10 +393,10 @@ abstract class UState( // clone the UState so it can no longer change, and pass that clone into // setFinished. val finfo = this match { - case m: UStateMain => m.cloneForSuspension(dos) + case m: SuspensionCapableUState => m.cloneForSuspension(dos) case _ => Assert.invariantFailed( - "State must be a UStateMain when splitting for bit order change" + "State must be SuspensionCapableUState when splitting for bit order change" ) } @@ -404,9 +411,48 @@ abstract class UState( def documentElement: DIDocument - final val releaseUnneededInfoset: Boolean = !areDebugging && tunable.releaseUnneededInfoset + // Build must never free infoset nodes: build runs ahead of write, so + // freeChildIfNoLongerNeeded would null out a child reference write + // hasn't read yet. Write still frees as normal once done with a node + // (see DIArray/DIComplex.freeChildIfNoLongerNeeded). + final val releaseUnneededInfoset: Boolean = + !isBuildOnly && !areDebugging && tunable.releaseUnneededInfoset def delimitedParseResult = Nope + + // UStateMain owns the real one; UStateForSuspension delegates to its + // mainUState so that a Suspension can always reach its tracker via + // savedUstate, even after it's been cloned off for suspension. + def suspensionTracker: SuspensionTracker + + // Optional reference to the shared build/write lead counter. Defaults + // unset (a no-op for every existing call site); only BuildState sets it + // (in its constructor), and only a write-side UState that opts in (via + // setSharedContext) reads it. + private var sharedContextMaybe: Maybe[UnparseSharedContext] = Nope + final def setSharedContext(ctx: UnparseSharedContext): Unit = sharedContextMaybe = One(ctx) + final def sharedContext: Maybe[UnparseSharedContext] = sharedContextMaybe + + // True for BuildState and any UStateForSuspension cloned from one (see + // override below). Unparsers gate content-writing on this rather than + // `isInstanceOf[BuildState]`, which a suspension continuation running + // on a UStateForSuspension would miss despite being build-side. + def isBuildOnly: Boolean = false +} + +/** + * Mixed in by any `UState` that can create and track its own + * `Suspension`s - `UStateMain` and `BuildState`, since build creates + * value-only (OVC/setVariable) suspensions too. `Suspension.suspend()` + * casts through this trait rather than hardcoding `UStateMain` directly, + * so a `BuildState`-created suspension resolves correctly instead of + * throwing `ClassCastException`. + */ +trait SuspensionCapableUState { self: UState => + def suspensions: Seq[Suspension] + def addSuspension(se: Suspension): Unit + def evalSuspensions(isFinal: Boolean): Unit + def cloneForSuspension(suspendedDOS: DirectOrBufferedDataOutputStream): UState } /** @@ -418,7 +464,7 @@ abstract class UState( * memory. */ final class UStateForSuspension( - val mainUState: UStateMain, + val mainUState: UState with SuspensionCapableUState, val dataOutputStream: DirectOrBufferedDataOutputStream, vbox: VariableBox, override val currentInfosetNode: DINode, @@ -430,6 +476,8 @@ final class UStateForSuspension( areDebugging: Boolean ) extends UState(vbox, mainUState.diagnostics, mainUState.dataProc, tunable, areDebugging) { + override def isBuildOnly: Boolean = mainUState.isBuildOnly + _dataOutputStream = dataOutputStream dState.setMode(UnparserBlocking) dState.setCurrentNode(thisElement.asInstanceOf[DINode]) @@ -443,6 +491,7 @@ final class UStateForSuspension( override def getEncoder(cs: BitsCharset): BitsCharsetEncoder = mainUState.getEncoder(cs) override def suspensions = mainUState.suspensions + override val suspensionTracker = mainUState.suspensionTracker // override def charBufferDataOutputStream = mainUState.charBufferDataOutputStream override def withUnparserDataInputStream = mainUState.withUnparserDataInputStream @@ -503,7 +552,44 @@ final class UStateForSuspension( } } -final class UStateMain private ( +/** + * Live (stack-backed) array-iteration/occurs/group/child index tracking, + * shared by UStateMain and BuildState: each stack starts seeded with 1L, + * and moveOverOne*Only bumps its top by one as navigation advances. + * UStateForSuspension needs none of this; it stubs the stacks to die and + * tracks arrayIterationPos/occursPos as plain frozen Longs captured at + * suspension time instead (see its overrides above), so this lives in a + * mixin rather than directly on UState. + */ +trait LiveIndexStacks { self: UState => + override val arrayIterationIndexStack = MStackOfLong() + arrayIterationIndexStack.push(1L) + override def moveOverOneArrayIterationIndexOnly(): Unit = + arrayIterationIndexStack.setTop(arrayIterationIndexStack.top + 1) + override def arrayIterationPos = arrayIterationIndexStack.top + + override val occursIndexStack = MStackOfLong() + occursIndexStack.push(1L) + override def moveOverOneOccursIndexOnly(): Unit = + occursIndexStack.setTop(occursIndexStack.top + 1) + override def occursPos = occursIndexStack.top + + override val groupIndexStack = MStackOfLong() + groupIndexStack.push(1L) + override def moveOverOneGroupIndexOnly(): Unit = + groupIndexStack.setTop(groupIndexStack.top + 1) + override def groupPos = groupIndexStack.top + + // TODO: it doesn't look anything is actually reading the value of childindex + // stack. Can we get rid of it? + override val childIndexStack = MStackOfLong() + childIndexStack.push(1L) + override def moveOverOneElementChildOnly(): Unit = + childIndexStack.setTop(childIndexStack.top + 1) + override def childPos = childIndexStack.top +} + +final class UStateMain private[unparsers] ( private val inputter: InfosetInputter, outStream: java.io.OutputStream, vbox: VariableBox, @@ -511,7 +597,9 @@ final class UStateMain private ( dataProcArg: DataProcessor, tunable: DaffodilTunables, areDebugging: Boolean -) extends UState(vbox, diagnosticsArg, One(dataProcArg), tunable, areDebugging) { +) extends UState(vbox, diagnosticsArg, One(dataProcArg), tunable, areDebugging) + with SuspensionCapableUState + with LiveIndexStacks { dState.setMode(UnparserBlocking) @@ -553,8 +641,10 @@ final class UStateMain private ( // MStack, since the escape scheme cache logic requires an MStack. We // reallyjust need the top for cloning for suspensions, but that // requires changes to how the escape schema cache is accessed, which - // isn't a trivial change. - val esClone = new MStackOfMaybe[EscapeSchemeUnparserHelper]() + // isn't a trivial change. Sized to the source's actual depth since + // nothing ever pushes onto a suspension's cloned escapeSchemeEVCache + // after this point. + val esClone = new MStackOfMaybe[EscapeSchemeUnparserHelper](escapeSchemeEVCache.length) esClone.copyFrom(escapeSchemeEVCache) Maybe(esClone) } else { @@ -563,8 +653,11 @@ final class UStateMain private ( val ds = if (!delimiterStack.isEmpty) { // If there are any delimiters, then we need to clone them all since - // they may be needed for escaping - val dsClone = new MStackOf[DelimiterStackUnparseNode]() + // they may be needed for escaping. Sized to the source's actual + // depth: pushDelimiters/popDelimiters both die on this clone (see + // below), so it's read-only for the rest of the suspension's life + // and can never grow past this depth. + val dsClone = new MStackOf[DelimiterStackUnparseNode](delimiterStack.length) dsClone.copyFrom(delimiterStack) Maybe(dsClone) } else { @@ -665,29 +758,6 @@ final class UStateMain private ( override val currentInfosetNodeStack = new MStackOfMaybe[DINode] - override val arrayIterationIndexStack = MStackOfLong() - arrayIterationIndexStack.push(1L) - override def moveOverOneArrayIterationIndexOnly() = - arrayIterationIndexStack.setTop(arrayIterationIndexStack.top + 1) - override def arrayIterationPos = arrayIterationIndexStack.top - - override val occursIndexStack = MStackOfLong() - occursIndexStack.push(1L) - override def moveOverOneOccursIndexOnly() = occursIndexStack.setTop(occursIndexStack.top + 1) - override def occursPos = occursIndexStack.top - - override val groupIndexStack = MStackOfLong() - groupIndexStack.push(1L) - override def moveOverOneGroupIndexOnly() = groupIndexStack.setTop(groupIndexStack.top + 1) - override def groupPos = groupIndexStack.top - - // TODO: it doesn't look anything is actually reading the value of childindex - // stack. Can we get rid of it? - override val childIndexStack = MStackOfLong() - childIndexStack.push(1L) - override def moveOverOneElementChildOnly() = childIndexStack.setTop(childIndexStack.top + 1) - override def childPos = childIndexStack.top - override lazy val escapeSchemeEVCache = new MStackOfMaybe[EscapeSchemeUnparserHelper] val delimiterStack = new MStackOf[DelimiterStackUnparseNode]() @@ -707,9 +777,20 @@ final class UStateMain private ( * All the other clones used for outputValueCalc, those never * need to add any. */ - private val suspensionTracker = + private val ownSuspensionTracker = new SuspensionTracker(tunable.unparseSuspensionWaitYoung, tunable.unparseSuspensionWaitOld) + // When sharedContext is set, route through the SAME tracker the other + // side of that split uses, or these suspensions would queue where + // nothing ever drains them. + override def suspensionTracker: SuspensionTracker = { + if (sharedContext.isDefined) { + sharedContext.get.suspensionTracker + } else { + ownSuspensionTracker + } + } + def addSuspension(se: Suspension): Unit = { suspensionTracker.trackSuspension(se) } @@ -772,4 +853,28 @@ object UState { ) newState } + + /** + * Like `createInitialUState`, but takes an already-constructed + * `VariableBox` directly instead of copying it fresh: in the two-phase + * build-then-write path, write must see the SAME `VariableBox` build + * mutated, not an independent copy frozen at the pre-build state. + */ + def createInitialUStateForSharedVariables( + outStream: java.io.OutputStream, + dataProc: DFDL.DataProcessor, + inputter: InfosetInputter, + vbox: VariableBox, + areDebugging: Boolean + ): UStateMain = { + new UStateMain( + inputter, + outStream, + vbox, + Nil, + dataProc.asInstanceOf[DataProcessor], + dataProc.tunables, + areDebugging + ) + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContext.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContext.scala new file mode 100644 index 0000000000..5a97e113c4 --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContext.scala @@ -0,0 +1,215 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.lib.exceptions.Assert +import org.apache.daffodil.lib.iapi.DaffodilTunables +import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.lib.util.Maybe.* +import org.apache.daffodil.runtime1.infoset.DIDocument +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.processors.DataProcessor +import org.apache.daffodil.runtime1.processors.SuspensionTracker +import org.apache.daffodil.runtime1.processors.VariableBox + +/** + * What build and write genuinely need to share by reference: the + * `SuspensionTracker` (one queue, two callers; a suspension build + * creates gets resolved once write finishes) and the `VariableBox` + * (shared `VariableInstance` objects, so OVC/setVariable results computed + * during build are visible to write). + * + * Also owns the lead counter: how far build is ahead of write, + * incremented once per node build constructs and decremented once per + * node write finishes. Once `leadExceedsPrefetchLimit` or the + * pending-suspension backlog exceeds `pendingSuspensionTripLimit` + * (below), build must resume write before continuing its own + * recursion, bounding how far ahead it may run. + */ +final class UnparseSharedContext( + val rootDoc: DIDocument, + val variableBox: VariableBox, + val suspensionTracker: SuspensionTracker, + val dataProc: DataProcessor, + val tunable: DaffodilTunables, + val prefetchLimit: Long +) { + private var buildLead: Long = 0 + + def incrementLead(): Unit = buildLead += 1 + + def decrementLead(): Unit = { + buildLead -= 1 + Assert.invariant(buildLead >= 0) + } + + def currentLead: Long = buildLead + + def leadExceedsPrefetchLimit: Boolean = buildLead > prefetchLimit + + /** + * A second, independent limit on how far build may run ahead of write, + * alongside prefetchLimit: a suspension can be created without moving + * the lead counter, so the lead alone doesn't bound how many pile up + * pending. + */ + def pendingSuspensionTripLimit: Long = tunable.unparsePendingSuspensionTripLimit + + private var buildCoroutine_ : BuildCoroutine = null + private var writeCoroutine_ : WriteCoroutine = null + + def setCoroutines(bc: BuildCoroutine, wc: WriteCoroutine): Unit = { + buildCoroutine_ = bc + writeCoroutine_ = wc + } + def buildCoroutine: BuildCoroutine = buildCoroutine_ + def writeCoroutine: WriteCoroutine = writeCoroutine_ + + private var recordedWriteDone: Maybe[WriteDone] = Nope + + /** + * Resumes writeCoroutine with `signal`, unless write already finished + * (a second resume would park forever), in which case the cached + * WriteDone is returned instead. Always go through this, never resume + * writeCoroutine directly. + */ + def resumeWrite(signal: BuildSignal): WriteSignal = { + if (recordedWriteDone.isDefined) { + recordedWriteDone.get + } else { + val result = buildCoroutine.resume(writeCoroutine, signal) + result match { + case wd: WriteDone => recordedWriteDone = One(wd) + case _ => + } + result + } + } + + /** + * Wakes write's thread (spawning it first if never started) after + * build's own thread failed before reaching the normal resumeWrite + * handoff, so it can clean up instead of hanging forever. A no-op if + * write already finished, or if setCoroutines was never reached. + */ + def abortWrite(): Unit = { + if (recordedWriteDone.isEmpty && writeCoroutine_ != null) { + buildCoroutine.resumeFinal(writeCoroutine, BuildAborted) + } + } + + /** + * True if `child` may safely be written now: complex/array existing is + * enough; simple needs a value, except hidden/nilled elements (never + * given one) and OVC (deferred via its own Suspension; must not block, + * or an OVC depending on a later sibling's write-time property would deadlock). + */ + private def isChildReady(child: DINode): Boolean = { + if (child.isSimple) { + val s = child.asSimple + child.isHidden || s.isNilled || s.hasValue || s.erd.dpathElementCompileInfo.isOutputValueCalc + } else { + // complex/array, hidden or not; existing is enough + true + } + } + + /** + * True once build's recursion has fully returned. Sticky: once true, + * awaitChild must never again resume buildCoroutine; build's thread has + * moved on to waiting for write's ultimate WriteDone, not another + * WriteNeedsMore cycle, so this can't be a per-call local flag. + */ + private var buildFinished: Boolean = false + + /** + * Records a signal write's coroutine received, without itself resuming + * anyone. Needed for the signal that STARTS write's thread, which can + * legitimately already be BuildFinished; must be recorded before any + * awaitChild call, or tryMakeProgress would wrongly resume build again. + */ + def observeBuildSignal(signal: BuildSignal): Unit = { + if (signal == BuildFinished) buildFinished = true + else if (signal == BuildAborted) throw new BuildAbortedException + } + + /** + * One attempt at unblocking write: resumes build if not finished yet, + * else retries suspensions directly. False means no progress was made; + * callers must throw AwaitChildStalledException rather than loop again, + * deferring to evalSuspensions(isFinal = true) for the real diagnosis. + */ + private def tryMakeProgress(): Boolean = { + if (!buildFinished) { + observeBuildSignal(writeCoroutine.resume(buildCoroutine, WriteNeedsMore)) + true + } else { + val before = suspensionTracker.suspensions.length + suspensionTracker.evalSuspensionsUnthrottled() + suspensionTracker.suspensions.length < before + } + } + + /** + * Blocks (parks write's thread) until `parent.child(index)` both exists + * and is ready (isChildReady above), throwing AwaitChildStalledException + * once tryMakeProgress reports no further progress is possible. + */ + def awaitChild(parent: DINode, index: Int): DINode = { + while (index >= parent.numChildren || !isChildReady(parent.child(index))) { + if (!tryMakeProgress()) throw new AwaitChildStalledException + } + parent.child(index) + } + + /** + * Blocks until `parent.child(index)` exists, or `parent.isFinal` with no + * child there (none ever will be); for callers asking "is there more + * data at all" rather than awaiting a child known to be coming. A true + * result may still need awaitChild afterward for value readiness. + */ + def childExistsOrFinal(parent: DINode, index: Int): Boolean = { + while (index >= parent.numChildren) { + if (parent.isFinal) return false + if (!tryMakeProgress()) throw new AwaitChildStalledException + } + true + } +} + +/** + * Thrown when write can make no further progress after build has + * finished and an unthrottled suspension retry made no headway. Caught + * only by the top-level coroutine driver, which still runs its normal + * finalization (the genuine, diagnostic-producing final suspension + * drain) rather than treating this as the final outcome itself. + */ +final class AwaitChildStalledException extends Exception + +/** + * Thrown the moment write's thread observes that build's own thread + * failed with an exception before reaching its normal completion, either + * as the very first signal write's thread ever receives, or as the + * result of a resume from deep inside a block on a pending child. Caught + * only by the top-level coroutine driver, which skips its own normal + * finalization entirely (those invariants and the final suspension drain + * assume a consistently, fully-built tree that a genuine abort may not + * have produced) and just cleans up write's own resources; build's own + * exception, not this one, is what actually gets reported. + */ +final class BuildAbortedException extends Exception diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala index 592544ff0d..87d9aacb21 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/runtime1/processors/unparsers/Unparser.scala @@ -20,7 +20,12 @@ package org.apache.daffodil.runtime1.processors.unparsers import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.runtime1.dsom.RuntimeSchemaDefinitionError +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.* +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase +import org.apache.daffodil.unparsers.runtime1.NewVariableInstanceStartUnparser +import org.apache.daffodil.unparsers.runtime1.SetVariableUnparser +import org.apache.daffodil.unparsers.runtime1.WriteUnparser sealed trait Unparser extends Processor { @@ -49,22 +54,28 @@ sealed trait Unparser extends Processor { // // So this is a temporary fix, until we can figure out where else to do this. // - this match { - // bit order only applies to primitives, not combinators, nor "noData" unparsers. - case af: AlignmentPrimUnparser => // ok. Don't check bitOrder before Aligning. - case u: PrimUnparser => { - u.context match { - case trd: TermRuntimeData => { - ustate.bitOrder // asking for bitOrder checks bit order changes. - // this splits DOS on bitOrder changes if absoluteBitPos not known + // Skipped for build: asking bitOrder can trigger a DOS split via + // cloneForSuspension, which BuildState can't do and would trip an + // invariant. Build reaches this wrapper for SeqCompUnparser siblings + // whose own internal gates aren't enough, so this must be skipped too. + if (!ustate.isBuildOnly) { + this match { + // bit order only applies to primitives, not combinators, nor "noData" unparsers. + case af: AlignmentPrimUnparser => // ok. Don't check bitOrder before Aligning. + case u: PrimUnparser => { + u.context match { + case trd: TermRuntimeData => { + ustate.bitOrder // asking for bitOrder checks bit order changes. + // this splits DOS on bitOrder changes if absoluteBitPos not known + } + case rd: RuntimeData => + Assert.invariantFailed( + "Primitive unparser " + u + " has non-Term runtime data: " + rd + ) } - case rd: RuntimeData => - Assert.invariantFailed( - "Primitive unparser " + u + " has non-Term runtime data: " + rd - ) } + case _ => // ok } - case _ => // ok } try { unparse(ustate) @@ -72,7 +83,11 @@ sealed trait Unparser extends Processor { ustate.resetFormatInfoCaches() } if (ustate.dataProc.isDefined) ustate.dataProc.get.after(ustate, this) - ustate.setMaybeProcessor(savedProc) + // Restore the prior processor only if one existed. Nope means this is + // the first unparse1 call on a freshly cloned suspension UState, which + // starts with none; resetting to Nope would discard the only context + // it will ever have, since runSuspension's later setFinished() needs one. + if (savedProc.isDefined) ustate.setMaybeProcessor(savedProc) } def UE(ustate: UState, s: String, args: Any*) = { @@ -147,7 +162,8 @@ final class ErrorUnparser(override val context: TermRuntimeData = null) final class SeqCompUnparser(context: RuntimeData, val childUnparsers: Array[Unparser]) extends CombinatorUnparser(context) - with ToBriefXMLImpl { + with ToBriefXMLImpl + with WriteUnparser { override val runtimeDependencies = Array() @@ -164,6 +180,37 @@ final class SeqCompUnparser(context: RuntimeData, val childUnparsers: Array[Unpa } } + /** + * SeqCompUnparser can wrap any `WriteUnparser` (sequence/choice/ + * hidden-group/delimiter-stack), so it walks children one at a time + * like `unparse()` above: each runs synchronously if not itself a + * WriteUnparser, else recurses into its own writeContent. + */ + override def writeContent(containerNode: DINode, ustate: UState): Unit = { + var i = 0 + while (i < childUnparsers.length) { + childUnparsers(i) match { + case _: NewVariableInstanceStartUnparser | _: SetVariableUnparser => + // Skip: build's unconditional recursion already ran these once + // (no isBuildOnly gate on either; see their own doc comments). + // Re-running on write would create a second VariableInstance or + // re-evaluate setVariable ("cannot set variable twice"). + () + case elemUnp: ElementUnparserBase => + // An element never appears here directly (SeqCompUnparser only + // wraps alignment/capture-length prims alongside a group's body + // unparser); plain unparse1 since writeContent expects an + // already-existing child node, not this shared containerNode. + elemUnp.unparse1(ustate) + case wu: WriteUnparser => + wu.writeContent(containerNode, ustate) + case cu => + cu.unparse1(ustate) + } + i += 1 + } + } + override def toString: String = { val strings = childUnparsers.map { _.toString } strings.mkString(" ~ ") diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala index 9cfff9e970..d0388801cd 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ChoiceAndOtherVariousUnparsers.scala @@ -19,6 +19,7 @@ package org.apache.daffodil.unparsers.runtime1 import scala.jdk.CollectionConverters.* +import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.lib.util.MaybeInt @@ -78,13 +79,116 @@ class ChoiceCombinatorUnparser( choiceBranchMap: ChoiceBranchMap, choiceLengthInBits: MaybeInt ) extends CombinatorUnparser(mgrd) - with ToBriefXMLImpl { + with ToBriefXMLImpl + with WriteUnparser { override def nom = "Choice" override val runtimeDependencies = Array() override def childProcessors = choiceBranchMap.childProcessors + /** + * Resolves which branch an already-built child belongs to by keying off the child's ERD (blocking via childExistsOrFinal/awaitChild until known), same as choiceBranchMap production code, then recurses into the resolved branch. + * Self-manages advancing/freeing its own resolved tree-child position and mirrors the choiceLengthInBits filler from unparse(); covers the visible-choice path only, withinHiddenNest's default branch is separate below. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = { + val sharedCtx = state.sharedContext.get + val complex = containerNode.asComplex + val childIndex = state.childIndexStack.top.toInt + + val (maybeChildUnparser, resolvedChildIndex): (Maybe[Unparser], Int) = + if (state.withinHiddenNest) { + // A hidden choice's branch is always the single deterministic + // default one (DFDL requires its outcome to be schema-determined, + // not data-driven), so no key/event peek is needed. The tree child at this position, if any, IS this default branch's own content, not something to skip past. + val idx = if (sharedCtx.childExistsOrFinal(complex, childIndex)) { + childIndex + } else { + -1 + } + (Maybe.toMaybe(choiceBranchMap.defaultUnparser), idx) + } else { + // Mirrors unparse()'s peek at the next infoset EVENT: hidden elements + // never produce events, so choiceBranchMap's keys are built from the + // first REPRESENTED child. Build still materializes a branch's leading hidden group as actual tree children, so skip past those first. + var idx = childIndex + while (sharedCtx.childExistsOrFinal(complex, idx) && complex.child(idx).isHidden) { + idx += 1 + } + if (idx >= complex.numChildren) { + // Build is done and no child ever showed up at this position: the + // choice resolved to a branch with no infoset footprint at all + // (e.g. an empty sequence branch, or an absent defaultable + // element). No event/child to key off, so fall back to the default/unmapped branch, mirroring unparse()'s own fallback. + (Maybe.toMaybe(choiceBranchMap.defaultUnparser), -1) + } else { + val child = sharedCtx.awaitChild(complex, idx) + val key: ChoiceBranchEvent = ChoiceBranchStartEvent(child.erd.namedQName) + val fromTable = choiceBranchMap.lookupTable.get(key) + if (fromTable != null) { + // An actual match; this tree position genuinely belongs to this + // choice. + (One(fromTable), idx) + } else { + // No branch key matches this child, so it must belong to a + // sibling term after this choice: this choice resolved to a + // branch with no infoset footprint here and consumes no tree position, mirroring unparse()'s own no-matching-event fallback. + (Maybe.toMaybe(choiceBranchMap.defaultUnparser), -1) + } + } + } + if (maybeChildUnparser.isEmpty) { + // A real UnparseError, not an internal assertion, mirroring unparse()'s + // own diagnostic: a choice with no default and no match for what's actually in the tree is malformed input, not an invariant violation. + UnparseError( + One(mgrd.schemaFileLocation), + One(state.currentLocation), + "No matching or default choice branch found." + ) + } + val child = if (resolvedChildIndex >= 0) { + complex.child(resolvedChildIndex) + } else { + // A resumable group or ChoiceBranchEmptyUnparser needs no tree child, + // so null is fine, but the unmapped default can also be a bare + // ElementUnparserBase, which DOES need one: writeContent(null, state) + // on that would NPE deep inside it instead of failing clearly here. + if (maybeChildUnparser.get.isInstanceOf[ElementUnparserBase]) { + Assert.invariantFailed( + "Choice resolved to a simple-element default branch with no resolved tree position." + ) + } + null + } + + // True when the resolved branch is itself a resumable group, whose own + // dispatch already advances/frees whatever tree positions it consumes. + // This choice must not also advance/free in that case, or it double-advances past the branch's last child, skipping the following sibling term. + var innerSelfManagesPosition = false + withChoiceLengthFiller(state) { + maybeChildUnparser.get match { + case elemUnp: ElementUnparserBase => elemUnp.writeContent(child, state) + case wu: WriteUnparser => + innerSelfManagesPosition = true + wu.writeContent(containerNode, state) + case emptyUnp: ChoiceBranchEmptyUnparser => + // A branch that optimized to nothing (e.g. a sequence containing + // only an assert); runs its (no-op) unparse, same as any other + // synchronous branch. + emptyUnp.unparse1(state) + case other => + Assert.usageError(s"unhandled choice branch unparser type: $other") + } + } + + // Nothing to advance/free when the branch had no infoset footprint + // (resolvedChildIndex == -1), nor when it's itself a resumable group (innerSelfManagesPosition), which already did so for its own positions. + if (resolvedChildIndex >= 0 && !innerSelfManagesPosition) { + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(resolvedChildIndex, state.releaseUnneededInfoset) + } + } + def unparse(state: UState): Unit = { if (state.withinHiddenNest) { val branchForUnparseIfHidden = choiceBranchMap.defaultUnparser @@ -93,11 +197,8 @@ class ChoiceCombinatorUnparser( state.pushTRD(mgrd) val event: InfosetAccessor = state.inspectOrError val key: ChoiceBranchEvent = event match { - // - // The ChoiceBranchStartEvent(...) is not a case class constructor. It is a - // hash-table lookup for a cached value. This avoids constructing these - // objects over and over again. - // + // ChoiceBranchStartEvent(...) here is a hash-table lookup for a cached + // value, not a case class constructor, avoiding reconstructing these objects repeatedly. case e if e.isStart && e.isElement => ChoiceBranchStartEvent(e.erd.namedQName) case e if e.isEnd && e.isElement => ChoiceBranchEndEvent(e.erd.namedQName) case e if e.isStart && e.isArray => ChoiceBranchStartEvent(e.erd.namedQName) @@ -121,22 +222,32 @@ class ChoiceCombinatorUnparser( val childUnparser = maybeChildUnparser.get state.popTRD(mgrd) state.pushTRD(childUnparser.context.asInstanceOf[TermRuntimeData]) - if (choiceLengthInBits.isDefined) { - val suspendableOp = - new ChoiceUnusedUnparserSuspendableOperation(mgrd, choiceLengthInBits.get) - val choiceUnusedUnparser = - new ChoiceUnusedUnparser(mgrd, choiceLengthInBits.get, suspendableOp) - - suspendableOp.captureDOSStartForChoiceUnused(state) - childUnparser.unparse1(state) - suspendableOp.captureDOSEndForChoiceUnused(state) - choiceUnusedUnparser.unparse(state) - } else { + withChoiceLengthFiller(state) { childUnparser.unparse1(state) } state.popTRD(childUnparser.context.asInstanceOf[TermRuntimeData]) } } + + /** + * Wraps runChosenBranch with the dfdl:choiceLength "unused region" filler (no-op if choiceLengthInBits isn't set), capturing DOS position before/after via ChoiceUnusedUnparser. + * The state.setProcessor calls are redundant for unparse()'s call site but required for writeContent's, which bypasses unparse1 (and its equivalent setProcessor) entirely. + */ + private def withChoiceLengthFiller(state: UState)(runChosenBranch: => Unit): Unit = { + if (choiceLengthInBits.isEmpty) { + runChosenBranch + } else { + val suspendableOp = + new ChoiceUnusedUnparserSuspendableOperation(mgrd, choiceLengthInBits.get) + val unusedUnparser = new ChoiceUnusedUnparser(mgrd, choiceLengthInBits.get, suspendableOp) + state.setProcessor(ChoiceCombinatorUnparser.this) + suspendableOp.captureDOSStartForChoiceUnused(state) + runChosenBranch + state.setProcessor(ChoiceCombinatorUnparser.this) + suspendableOp.captureDOSEndForChoiceUnused(state) + unusedUnparser.unparse(state) + } + } } class DelimiterStackUnparser( @@ -145,7 +256,20 @@ class DelimiterStackUnparser( terminatorOpt: Maybe[TerminatorUnparseEv], ctxt: TermRuntimeData, bodyUnparser: Unparser -) extends CombinatorUnparser(ctxt) { +) extends CombinatorUnparser(ctxt) + with WriteUnparser { + + /** + * Same delimiter-scope push/eval/pop as unparse() below, but the pop happens only once the body's own writeContent (if any) returns, since a pending pause still needs the stack for separator writing. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop( + containerNode, + bodyUnparser, + state, + setup = pushDelimiterScope, + teardown = (state, _) => state.popDelimiters() + ) override def nom = "DelimiterStack" override def toBriefXML(depthLimit: Int = -1): String = { @@ -164,7 +288,19 @@ class DelimiterStackUnparser( (initiatorOpt.toList ++ separatorOpt.toList ++ terminatorOpt.toList).toArray def unparse(state: UState): Unit = { - // Evaluate Delimiters + // The delimiter stack is write-only state; build never writes delimiter + // text (DelimiterTextUnparser's writers are gated on isBuildOnly), so + // recursing into bodyUnparser without push/pop is what lets build navigate a separated sequence's children instead of hitting its pushDelimiters stub. + if (state.isBuildOnly) { + bodyUnparser.unparse1(state) + } else { + pushDelimiterScope(state) + bodyUnparser.unparse1(state) + state.popDelimiters() + } + } + + private def pushDelimiterScope(state: UState): Unit = { val init = if (initiatorOpt.isDefined) initiatorOpt.get.evaluate(state) else EmptyDelimiterStackUnparseNode.empty @@ -174,14 +310,7 @@ class DelimiterStackUnparser( val term = if (terminatorOpt.isDefined) terminatorOpt.get.evaluate(state) else EmptyDelimiterStackUnparseNode.empty - - val node = DelimiterStackUnparseNode(init, sep, term) - - state.pushDelimiters(node) - - bodyUnparser.unparse1(state) - - state.popDelimiters() + state.pushDelimiters(DelimiterStackUnparseNode(init, sep, term)) } } @@ -189,25 +318,45 @@ class DynamicEscapeSchemeUnparser( escapeScheme: EscapeSchemeUnparseEv, ctxt: TermRuntimeData, bodyUnparser: Unparser -) extends CombinatorUnparser(ctxt) { +) extends CombinatorUnparser(ctxt) + with WriteUnparser { override def nom = "EscapeSchemeStack" override def childProcessors = Vector(bodyUnparser) override val runtimeDependencies = Array(escapeScheme) - def unparse(state: UState): Unit = { - // evaluate the dynamic escape scheme in the correct scope. the resulting - // value is cached in the Evaluatable (since it is manually cached) and - // future parsers/evaluatables that use this escape scheme will use that - // cached value. - escapeScheme.newCache(state) - escapeScheme.evaluate(state) + /** + * Same escape-scheme-cache push/eval/pop as unparse() below, but the pop happens only once the body's own writeContent (if any) returns, since a pending pause still needs the cache for delimiter writing. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop( + containerNode, + bodyUnparser, + state, + setup = cacheEscapeScheme, + teardown = (state, _) => escapeScheme.invalidateCache(state) + ) - // Unparse + def unparse(state: UState): Unit = { + // The escape scheme cache is write-only state; build never writes + // escaped text itself, so recursing into bodyUnparser without + // caching/invalidating it is what lets build navigate a dynamically-escaped element's children instead of hitting its escapeSchemeEVCache stub. + if (state.isBuildOnly) { + bodyUnparser.unparse1(state) + return + } + cacheEscapeScheme(state) bodyUnparser.unparse1(state) - - // invalidate the escape scheme cache escapeScheme.invalidateCache(state) } + + // Evaluates the dynamic escape scheme in the correct scope; the result is + // cached in the Evaluatable (since it is manually cached), so future + // unparsers/evaluatables that use this escape scheme reuse that cached + // value. + private def cacheEscapeScheme(state: UState): Unit = { + escapeScheme.newCache(state) + escapeScheme.evaluate(state) + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/DelimiterUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/DelimiterUnparsers.scala index b91e4ce338..d9ecf33b09 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/DelimiterUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/DelimiterUnparsers.scala @@ -49,6 +49,11 @@ class DelimiterTextUnparser( } def unparse(state: UState): Unit = { + // Write-only, like ElementUnparserBase's doBeforeContentUnparser/ + // doAfterContentUnparser gate. Also required for safety: build never + // pushes a delimiter stack node, so state.localDelimiters would hit + // BuildState's writeOnly stub (Assert.usageError) otherwise. + if (state.isBuildOnly) return Logger.log.debug( s"Unparsing starting at bit position: ${state.getDataOutputStream.maybeAbsBitPos0b}" diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala index c1343654ba..f1ec6b50ee 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ElementUnparser.scala @@ -24,6 +24,7 @@ import org.apache.daffodil.lib.util.MaybeULong import org.apache.daffodil.runtime1.dpath.SuspendableExpression import org.apache.daffodil.runtime1.dsom.CompiledExpression import org.apache.daffodil.runtime1.infoset.DIComplex +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.infoset.DISimple import org.apache.daffodil.runtime1.infoset.DataValue.DataValuePrimitive import org.apache.daffodil.runtime1.infoset.RetryableException @@ -31,6 +32,7 @@ import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.Evaluatable import org.apache.daffodil.runtime1.processors.UnparseTargetLengthInBitsEv import org.apache.daffodil.runtime1.processors.unparsers.* +import org.apache.daffodil.runtime1.processors.unparsers.MoreTreeAvailable /** * Elements that, when unparsing, have no length specified. @@ -65,6 +67,49 @@ sealed trait RepMoveMixin { } } +/** + * The build-side hookup for bounded lookahead: called once per node from + * unparseBegin (regular and OVC strategies alike). Increments the shared + * lead counter, and once it exceeds the prefetch limit, resumes write's + * coroutine and blocks until it yields back before continuing build's own + * recursion. A no-op for single-pass, where state.sharedContext is never + * set. + */ +private object BuildWriteLeadHookup { + + /** + * Must be called from unparseBegin, not unparseEnd: an ancestor's + * increment must happen before any descendant's content is built. A + * write cascade triggered here can reach that ancestor before + * build's own recursive call into it returns; incrementing in + * unparseEnd instead would underflow the counter. + */ + def afterNodeAdded(state: UState): Unit = { + // Gated on isBuildOnly, not just sharedContext.isDefined: write's + // own UState also has sharedContext set, so gating on that alone + // would run build-only bookkeeping (incl. ctx.resumeWrite, which + // assumes the caller is build's thread) from write's side too, if + // unparseBegin were ever reached there. + if (state.isBuildOnly && state.sharedContext.isDefined) { + val ctx = state.sharedContext.get + ctx.incrementLead() + // leadExceedsPrefetchLimit alone doesn't bound the retained + // memory of pending suspensions, so pendingSuspensionTripLimit + // separately caps total pendingCount; resuming write here can + // also resolve suspensions immediately. + if ( + ctx.leadExceedsPrefetchLimit || + ctx.suspensionTracker.pendingCount > ctx.pendingSuspensionTripLimit + ) { + // resumeWrite may legitimately send WriteDone here, not just + // WriteNeedsMore, and records that so nothing tries to resume an + // already-finished coroutine again later. + ctx.resumeWrite(MoreTreeAvailable) + } + } + } +} + /** * The unparser used for an element that has inputValueCalc. * @@ -126,7 +171,8 @@ sealed abstract class ElementUnparserBase( val eReptypeUnparser: Maybe[Unparser] ) extends CombinatorUnparser(erd) with RepMoveMixin - with ElementUnparserStartEndStrategy { + with ElementUnparserStartEndStrategy + with WriteUnparser { final override def childProcessors = (eBeforeUnparser.toList ++ eUnparser.toList ++ eAfterUnparser.toList ++ eReptypeUnparser.toList ++ setVarUnparsers.toList).toVector @@ -161,21 +207,118 @@ sealed abstract class ElementUnparserBase( } } - protected def doBeforeContentUnparser(state: UState): Unit = { - if (eBeforeUnparser.isDefined) + /** + * Padding is always a synchronous, write-only concern, so build + * (which never writes actual output) unconditionally skips it; + * unlike runContentUnparser below, there's no group case to + * preserve here. + */ + private[runtime1] def doBeforeContentUnparser(state: UState): Unit = { + if (!state.isBuildOnly && eBeforeUnparser.isDefined) eBeforeUnparser.get.unparse1(state) } - protected def doAfterContentUnparser(state: UState): Unit = { - if (eAfterUnparser.isDefined) + private[runtime1] def doAfterContentUnparser(state: UState): Unit = { + if (!state.isBuildOnly && eAfterUnparser.isDefined) eAfterUnparser.get.unparse1(state) } - protected def runContentUnparser(state: UState): Unit = { + /** + * Registers whatever suspensions this element's content depends on + * (dfdl:length, dfdl:outputValueCalc) before content bytes get + * written. No-op by default; overridden by the SpecifiedLength + * unparsers. Called unconditionally from writeContent too, before + * its group-unparser check, so this isn't skipped for + * delimited/escape-scheme-wrapped elements. + */ + private[runtime1] def contentSetup(state: UState): Unit = () + + /** + * eReptypeUnparser is a transient concern like padding, always + * skipped for build; eUnparser for a complex element IS the + * model-group unparser build recurses into. Gated on + * `erd.isSimpleType`, not `WriteUnparser`-ness (delimiter/escape + * wrapping makes a simple element's eUnparser one too, and running + * it would corrupt one-time state write relies on). + */ + private[runtime1] def dispatchContentUnparser(state: UState): Unit = { if (eReptypeUnparser.isDefined) { - eReptypeUnparser.get.unparse1(state) - } else if (eUnparser.isDefined) - eUnparser.get.unparse1(state) + if (!state.isBuildOnly) eReptypeUnparser.get.unparse1(state) + } else if (eUnparser.isDefined) { + if (state.isBuildOnly && erd.isSimpleType) { + // build skips simple-element value-writing; the group-recursion + // case (erd.isComplexType) falls through to the else branch below + } else { + eUnparser.get.unparse1(state) + } + } + } + + private[runtime1] def runContentUnparser(state: UState): Unit = { + contentSetup(state) + dispatchContentUnparser(state) + } + + /** + * Writes this element's content against an already-built + * `containerNode`, without consuming any InfosetInputter events: + * dispatches to whatever `eUnparser` turns out to be, a group + * unparser or a plain simple-element value-writer. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = { + state.currentInfosetNodeStack.push(One(containerNode)) + state.childIndexStack.push(0L) + doBeforeContentUnparser(state) + // contentSetup can suspend, and suspending reads state.processor. + // That's normally set by the ordinary unparse dispatch, which this + // call bypasses entirely, so it's set explicitly here to match. + state.setProcessor(this) + // Must run before dispatchContentUnparser, unconditionally: an + // element whose eUnparser is group-wrapped (delimiter stack, escape + // scheme, specified-length prefix) delegates straight to that + // wrapper and never reaches dispatchContentUnparser, so without + // this call up front such an element would never get its setup run + // from write's side. + contentSetup(state) + // eReptypeUnparser takes priority over eUnparser here too, mirroring + // dispatchContentUnparser's own if/else-if: otherwise a repType'd + // string element whose raw (unused) eUnparser happens to be + // group-wrapped (e.g. delimiter-wrapped text) would incorrectly + // delegate to that raw content's dispatch, skipping the repType + // conversion entirely. + if ( + eReptypeUnparser.isEmpty && eUnparser.isDefined && eUnparser.get + .isInstanceOf[WriteUnparser] + ) { + eUnparser.get.asInstanceOf[WriteUnparser].writeContent(containerNode, state) + } else { + dispatchContentUnparser(state) + } + // The after-content (padding/fill) region depends on the content + // having been written, so it must run only once the (possibly + // nested) content dispatch above has fully returned; true for both + // the simple-element and group-content cases. + doAfterContentUnparser(state) + // A SIMPLE node's content-conversion unparser only just ran, above + // (build skipped it); complex/array nodes were already finalized in + // build's unparseEnd, so skip re-finalizing here or setFinal()'s + // !isFinal assert aborts. An OVC node may still be valueless + // (isChildReady allows write to reach later siblings without one), + // so wait for hasValue rather than asserting it via setFinal(). + if (containerNode.isSimple && !containerNode.isFinal && containerNode.asSimple.hasValue) + containerNode.setFinal() + // Write-side half of the shared lead counter, matching build's own + // once-per-node increment; no-op unless a write-side UState has + // opted in via setSharedContext. + if (state.sharedContext.isDefined) state.sharedContext.get.decrementLead() + // Content-length/value-length suspensions can only unblock once + // write has actually written the bytes, so this must run here too, + // not just build's unparseEnd, or such suspensions pile up + // unresolved until the final isFinal=true call instead of resolving + // as data becomes available. + state.asInstanceOf[SuspensionCapableUState].evalSuspensions(isFinal = false) + state.childIndexStack.pop() + state.currentInfosetNodeStack.pop } override def unparse(state: UState): Unit = { @@ -188,11 +331,8 @@ sealed abstract class ElementUnparserBase( doBeforeContentUnparser(state) - // - // We must push the TermRuntimeData for all model-groups. - // The starting point for this is the model-group of a complex type. - // Simple types don't have model groups, so no pushing those. - // + // We must push the TermRuntimeData for all model-groups, starting from + // the complex type's model-group; simple types have none to push. if (erd.isComplexType) state.pushTRD(erd.optComplexTypeModelGroupRuntimeData.get) @@ -298,11 +438,16 @@ class ElementSpecifiedLengthUnparser( override val runtimeDependencies = maybeTargetLengthEv.toArray - override def runContentUnparser(state: UState): Unit = { - computeTargetLength( - state - ) // must happen before run() so that we can take advantage of knowing the length - super.runContentUnparser(state) // setup unparsing, which will block for no valu + // Must happen before dispatchContentUnparser so we can take + // advantage of knowing the length. Build already runs this + // unconditionally on the shared infoset tree; write's writeContent + // call must not run it again, or it would create an independent + // Suspension for the same target length instead of reusing + // whatever build already resolved. + override private[runtime1] def contentSetup(state: UState): Unit = { + if (!(state.sharedContext.isDefined && !state.isBuildOnly)) { + computeTargetLength(state) + } } } @@ -324,12 +469,12 @@ class ElementOVCSpecifiedLengthUnparserSuspendableExpression( val diSimple = state.currentInfosetNode.asSimple diSimple.setDataValue(v) - - // - // These are now done in the main unparse, but they will - // suspend if they cannot be evaluated because there is not data value yet. - // - // callingUnparser.computeSetVariables(state) + // Do NOT setFinal here: a chained SimpleTypeRetryUnparser's own + // suspension may still be pending, and its continuation's + // overwriteDataValue call asserts !isFinal, which finalizing here + // first would trip. writeContent's own hasValue-guarded setFinal + // safely covers the common case where everything resolved + // synchronously. } override protected def maybeKnownLengthInBits(ustate: UState): MaybeULong = MaybeULong(0L) @@ -360,14 +505,43 @@ class ElementOVCSpecifiedLengthUnparser( private def suspendableExpression = new ElementOVCSpecifiedLengthUnparserSuspendableExpression(this, expr) + // True when this OVC expression references dfdl:valueLength/contentLength. + // Not a guarantee the attempt below would fail; a reference to an + // already-finished sibling's length could resolve immediately even + // during build. Just a static, direction-blind heuristic: most such + // references are forward (the common case this optimizes for), so an + // uncommon backward reference just takes an avoidable suspend/retry. + private val requiresWriteTimeValue: Boolean = + !SuspendableExpression.canResolveWithoutWriting(expr) + Assert.invariant(context.dpathElementCompileInfo.isOutputValueCalc) - override def runContentUnparser(state: UState): Unit = { - computeTargetLength( - state - ) // must happen before run() so that we can take advantage of knowing the length - suspendableExpression.run(state) // run the expression. It might or might not have a value. - super.runContentUnparser(state) // setup unparsing, which will block for no valu + // Build always registers this OVC's suspension unconditionally + // (needed for NVI-scope visibility); write's writeContent call must + // not register it again -- a `!hasValue` guard alone is + // insufficient, since a forward reference to a later sibling's + // write-time-only property leaves hasValue false for both, so both + // would create independent Suspensions and crash ("cannot set + // value/length twice"). + override private[runtime1] def contentSetup(state: UState): Unit = { + if (!(state.sharedContext.isDefined && !state.isBuildOnly)) { + // Must happen before dispatchContentUnparser so we can take + // advantage of knowing the length. + computeTargetLength(state) + if (!state.currentInfosetNode.asSimple.hasValue) { + // Runs at build's normal structural point so its NVI-scope + // patching still applies; deferring this to write instead of + // just skipping the wasted attempt could silently resolve + // dfdl:newVariableInstance-scoped variables to the wrong + // iteration under prefetch. + if (state.sharedContext.isDefined && requiresWriteTimeValue) { + suspendableExpression.suspendWithoutAttempting(state) + } else { + // run the expression. It might or might not have a value. + suspendableExpression.run(state) + } + } + } } } @@ -484,6 +658,12 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart // is pushing and popping to match the events. This provides the proper // context for evaluation of expressions. state.currentInfosetNodeStack.push(One(newElem)) + + // Must happen here (begin), not in unparseEnd: an ancestor's lead + // increment must happen before any descendant's content is built, + // or a write triggered here could reach that ancestor before + // build's own recursive call into it returns, underflowing the counter. + BuildWriteLeadHookup.afterNodeAdded(state) } } @@ -530,15 +710,18 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart } } - // cur is finished, mark it as final and free if possible. Note that we - // need the container and not the parent of the current element to free - // it. This way if this element is in an array, we free this element - // from the array. We also do not set hidden IVC elements as - // final--although we allow hidden IVC elements when unparsing, they - // never get a value so we can't set them as final without breaking - // assertions. Nothing can access hidden IVC elements, so this should - // not break anything - if (!state.withinHiddenNest || erd.isRepresented) cur.setFinal() + // cur is finished: mark it final and free via its container (not + // parent, so an array-member frees from the array), except hidden IVC + // elements (never get a value, so setFinal would trip assertions) and, + // for build/write-prefetch, SIMPLE elements: build skips their + // content-conversion unparser, so finalizing here would make + // write's later overwriteDataValue call trip its `!isFinal` + // invariant. Complex/array nodes are unaffected, since write + // never calls overwriteDataValue on those. + if ( + (!state.withinHiddenNest || erd.isRepresented) && + !(state.isBuildOnly && cur.isSimple) + ) cur.setFinal() val curContainer = if (cur.erd.isArray) cur.diParent.maybeLastChild.get else cur.diParent @@ -558,7 +741,7 @@ sealed trait RegularElementUnparserStartEndStrategy extends ElementUnparserStart move(state) - state.asInstanceOf[UStateMain].evalSuspensions(isFinal = false) + state.asInstanceOf[SuspensionCapableUState].evalSuspensions(isFinal = false) } } @@ -626,6 +809,12 @@ trait OVCStartEndStrategy extends ElementUnparserStartEndStrategy { parentComplex.addChild(ovcElem, state.tunable) state.currentInfosetNodeStack.push(One(ovcElem)) + + // Must happen here (begin), not in unparseEnd: an ancestor's lead + // increment must happen before any descendant's content is built, + // or a write triggered here could reach that ancestor before + // build's own recursive call into it returns, underflowing the counter. + BuildWriteLeadHookup.afterNodeAdded(state) } protected final override def unparseEnd(state: UState): Unit = { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ExpressionEvaluatingUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ExpressionEvaluatingUnparsers.scala index 180b96f71f..659ef0dc6d 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ExpressionEvaluatingUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/ExpressionEvaluatingUnparsers.scala @@ -100,6 +100,10 @@ class NewVariableInstanceStartUnparser(vrd: VariableRuntimeData, trd: TermRuntim override def childProcessors = Vector() override def unparse(state: UState) = { + // Runs unconditionally during build, not write-gated: a suspension's + // shallow-copy clone only sees a scope pushed before it was cloned, so + // deferring this push to write would leave that suspension permanently + // wired to the wrong VariableInstance. The pop is deferred to write. val nvi = state.newVariableInstance(vrd) if (vrd.maybeDefaultValueExpr.isDefined) { @@ -124,5 +128,11 @@ class NewVariableInstanceEndUnparser(vrd: VariableRuntimeData, trd: TermRuntimeD override def childProcessors = Vector() - override def unparse(state: UState) = state.removeVariableInstance(vrd) + override def unparse(state: UState) = { + // Polymorphic pop: build pops only its own build-local NVI stack, + // never the shared vTable; single-pass/write pop the shared vTable + // instead. Only write pops the vTable, since it must stay alive + // until write-time value-setting is done. + state.removeVariableInstance(vrd) + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala index 070e787290..1a4eff66a9 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/HiddenGroupCombinatorUnparser.scala @@ -17,6 +17,7 @@ package org.apache.daffodil.unparsers.runtime1 +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.ModelGroupRuntimeData import org.apache.daffodil.runtime1.processors.unparsers.* @@ -27,12 +28,26 @@ import org.apache.daffodil.runtime1.processors.unparsers.* * we unwind from the refs, we'll decrement. */ class HiddenGroupCombinatorUnparser(ctxt: ModelGroupRuntimeData, bodyUnparser: Unparser) - extends CombinatorUnparser(ctxt) { + extends CombinatorUnparser(ctxt) + with WriteUnparser { override def childProcessors = Vector(bodyUnparser) override val runtimeDependencies = Array() + // The hidden-depth counter must stay incremented across any pauses, since + // anything the body writes needs it (e.g. choice-branch resolution + // branches on state.withinHiddenNest, and RepType conversion asserts + // it's never true). + override def writeContent(containerNode: DINode, start: UState): Unit = + writeWithPushPop( + containerNode, + bodyUnparser, + start, + setup = _.incrementHiddenDef(), + teardown = (start, _) => start.decrementHiddenDef() + ) + def unparse(start: UState): Unit = { try { start.incrementHiddenDef() diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala index 3856c0b272..44f1273678 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/LayeredSequenceUnparser.scala @@ -17,6 +17,7 @@ package org.apache.daffodil.unparsers.runtime1 +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.layers.LayerDriver import org.apache.daffodil.runtime1.processors.SequenceRuntimeData import org.apache.daffodil.runtime1.processors.unparsers.* @@ -28,8 +29,48 @@ class LayeredSequenceUnparser( override def nom = "LayeredSequence" + private def handleLayerThrowable(layerDriver: LayerDriver, t: Throwable): Unit = { + if (layerDriver ne null) { + layerDriver.handleThrowable(t) + } else { + LayerDriver.handleThrowableWithoutLayer(t) + } + } + + // Same setup/teardown as unparse() below, via withLayerTransform. Without + // this override, write-side dispatch would treat this as a plain + // WriteUnparser (inherited), bypassing the layer transform entirely: raw + // bytes, no compression/checksum, no layer error handling. + override def writeContent(containerNode: DINode, state: UState): Unit = { + // Needed for the same reason as ChoiceCombinatorUnparser's writeContent: + // setFinished/cloneForSuspension reach state.bitOrder/state.processor, + // normally set by Unparser.unparse1's wrapper, which this + // recursive-dispatch code bypasses. + state.setProcessor(LayeredSequenceUnparser.this) + withLayerTransform(state) { + LayeredSequenceUnparser.super.writeContent(containerNode, state) + } + } + override def unparse(state: UState): Unit = { + // Layers are write-only: build only needs to navigate into the layer + // body to build the tree, not run the byte transform, so this skips + // withLayerTransform entirely. Without it, build's no-op-sink DOS + // gets `addBuffered()` splits that never merge back, and `cloneForSuspension` below would cast it to UStateMain, which BuildState isn't. + if (state.isBuildOnly) { + super.unparse(state) + return + } + withLayerTransform(state) { + super.unparse(state) + } + } + // Splits off a buffered DOS for the layer to flush through, so fragment + // bits/bitOrder on the original DOS can't affect how the layer flushes + // bytes, then runs the layer driver's transform around runBody. The + // `finally` restoration ensures later writes always reach `layerFollowingDOS`, win or lose. + private def withLayerTransform(state: UState)(runBody: => Unit): Unit = { val originalDOS = state.getDataOutputStream // create a new buffered DOS that this layer will flush to when the layer @@ -74,7 +115,7 @@ class LayeredSequenceUnparser( // unparse the layer body into layerDOS state.setDataOutputStream(layerDOS) - super.unparse(state) + runBody // now we're done unparsing the layer recursively. // While doing that unparsing, the data output stream may have been split, so the // DOS in the state may no longer be the layerDOS. @@ -95,8 +136,14 @@ class LayeredSequenceUnparser( // layer stack is potentially still needed, so // nothing can be cleaned up at this point. } catch { - case t: Throwable if (layerDriver ne null) => layerDriver.handleThrowable(t) - case t: Throwable => LayerDriver.handleThrowableWithoutLayer(t) + // Pure write-side control-flow signals, unrelated to the layer + // itself; rewrapping either into a LayerFatalException would defeat + // the write-side's own handling (a stalled-write deadlock + // diagnostic, or build's abort cleanup) with a raw "layer failed" + // exception. + case e: AwaitChildStalledException => throw e + case e: BuildAbortedException => throw e + case t: Throwable => handleLayerThrowable(layerDriver, t) // otherwise we have no layer driver, so we were unable to load the layer. // just let that propagate. } finally { diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala index 1d31e39d1e..851b153561 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/NilEmptyCombinatorUnparsers.scala @@ -19,6 +19,7 @@ package org.apache.daffodil.unparsers.runtime1 import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.util.Maybe +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.unparsers.* @@ -52,7 +53,8 @@ case class ComplexNilOrContentUnparser( ctxt: ElementRuntimeData, nilUnparser: Unparser, contentUnparser: Unparser -) extends CombinatorUnparser(ctxt) { +) extends CombinatorUnparser(ctxt) + with WriteUnparser { override val runtimeDependencies = Array() @@ -66,4 +68,17 @@ case class ComplexNilOrContentUnparser( else contentUnparser.unparse1(state) } + + // Without this override, WriteUnparser dispatch would call + // contentUnparser.unparse1 synchronously on write's state, but it can + // itself be a resumable group unparser expecting live InfosetInputter + // events that don't exist yet (see SpecifiedLengthExplicitImplicitUnparser). + override def writeContent(containerNode: DINode, state: UState): Unit = { + val bodyUnparser = if (containerNode.asComplex.isNilled) { + nilUnparser + } else { + contentUnparser + } + writeWithPushPop(containerNode, bodyUnparser, state) + } } diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala index 7c451f1ddb..bb3279fe94 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SeparatedSequenceUnparsers.scala @@ -26,6 +26,9 @@ import org.apache.daffodil.lib.schema.annotation.props.gen.SeparatorPosition import org.apache.daffodil.lib.schema.annotation.props.gen.SeparatorPosition.* import org.apache.daffodil.lib.util.Maybe import org.apache.daffodil.lib.util.MaybeInt +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.runtime1.infoset.DIComplex +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.ModelGroupRuntimeData import org.apache.daffodil.runtime1.processors.SequenceRuntimeData @@ -120,7 +123,8 @@ class OrderedSeparatedSequenceUnparser( sepMtaUnparserMaybe: Maybe[Unparser], sep: Unparser, childUnparsers: Array[SequenceChildUnparser with Separated] -) extends OrderedSequenceUnparserBase(rd) { +) extends OrderedSequenceUnparserBase(rd) + with WriteUnparser { // Sequences of nothing (no initiator, no terminator, nothing at all) should // have been optimized away Assert.invariant(childUnparsers.length > 0) @@ -129,6 +133,435 @@ class OrderedSeparatedSequenceUnparser( override def childProcessors = childUnparsers.toVector + /** + * In-flight separator-suppression state for one writeContent call: + * whether any term has already written something (`wroteAny`), whichever + * separator is currently deferred pending its term's content + * (`pendingPostfixSeparatorAtTerm` for ssp Never, + * `pendingSuppressibleOp` for AnyEmpty/TrailingEmpty(Strict)), and the + * end-of-sequence TrailingEmpty(Strict) queue (`trailingSuspendedOps`). + */ + private class SeparatorSuppressionState(state: UState) { + + private var wroteAny = false + + // for spos == Postfix, the separator for a represented term must come + // AFTER its content is written, not before. Set by beforeSeparator, + // cleared by afterSeparator once that term's content has been written. + private var pendingPostfixSeparatorAtTerm = false + + // For ssp AnyEmpty/TrailingEmpty(Strict): the in-flight suspension + // beforeSeparator started for the current term, completed by + // afterSeparator. Mirrors pendingPostfixSeparatorAtTerm's role for ssp + // Never, but as a suspension object rather than a flag. + private var pendingSuppressibleOp: SuppressableSeparatorUnparserSuspendableOperation = null + + // For ssp TrailingEmpty/TrailingEmptyStrict: separators deferred until + // the sequence's own end (DFDL requires trailing through the whole + // sequence, not just the local group), matching + // unparseWithSuppression's end-of-loop resolution. + private val trailingSuspendedOps = + scala.collection.mutable.Buffer[SuppressableSeparatorUnparserSuspendableOperation]() + + /** + * ssp=never never omits separators based on content, so a term with + * fewer than maxOccurs actual occurrences still gets one separator per + * missing occurrence, mirroring unparseWithNoSuppression's own step. + * Other policies decide via content length instead; irrelevant here. + */ + def writeExtraSeparatorsIfNeverSuppressed( + rep: RepeatingChildUnparser, + numOccurrences: Long + ): Unit = { + if (ssp != Never) return + // Uses erd.maxOccurs, not rep.maxRepeats(state): for + // occursCountKind="expression"/"parsed", maxRepeats(state) is + // Long.MaxValue, which would loop until OOM. An unbounded array's + // maxOccurs is -1, so sepsNeeded goes negative and this loop no-ops. + val sepsNeeded = rep.erd.maxOccurs - numOccurrences + if (sepsNeeded <= 0) return + val numExtraSeps = if ((spos eq Infix) && !wroteAny) { + sepsNeeded - 1 + } else { + sepsNeeded + } + var n = numExtraSeps + while (n > 0) { + unparseJustSeparator(state) + n -= 1 + } + // Deliberately NOT wroteAny = true: unparseWithNoSuppression's own + // identical loop never advances state.groupPos either, so a + // following Infix term must still see this as unwritten, or it + // would wrongly get its own separator too. + } + + /** + * occursCountKind="implicit" is positional, so a non-trailing bounded + * array/optional still needs speculative missing-occurrence separators, + * or a later term shifts position. stacksAlreadyPushed: true only right + * after an actual occurrence loop already positioned the stacks. + */ + def writePositionallyRequiredSepsIfSuppressed( + rep: RepeatingChildUnparser with Separated, + numOccurrences: Long, + stacksAlreadyPushed: Boolean + ): Unit = { + if (ssp == Never) return + if ( + (rep.ock ne OccursCountKind.Implicit) || + !rep.isPositional || !rep.isBoundedMax || + (rep.isDeclaredLast && rep.isPotentiallyTrailing) + ) return + // safe: isBoundedMax being true guarantees maxRepeats(state) is the + // actual, finite erd.maxOccurs, never Long.MaxValue (see + // MinMaxRepeatsMixin.isBoundedMax). + val maxReps = rep.maxRepeats(state) + if (numOccurrences >= maxReps) return + if (!stacksAlreadyPushed) { + state.arrayIterationIndexStack.push(1L) + state.occursIndexStack.push(1L) + } + var n = numOccurrences + while (n < maxReps) { + beforeSeparator(rep.erd, rep.isKnownStaticallyNotToSuppressSeparator) + afterSeparator() + n += 1 + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + } + if (!stacksAlreadyPushed) { + state.arrayIterationIndexStack.pop() + state.occursIndexStack.pop() + } + } + + /** + * Called before a term/occurrence's content is written, once known + * to actually be written. For ssp Never (or staticallyNotSuppressible), + * Prefix/Infix write immediately and Postfix defers; otherwise + * Prefix/Infix speculatively unparse a suppressible separator. + */ + def beforeSeparator( + trd: TermRuntimeData, + staticallyNotSuppressible: Boolean + ): Unit = { + if (staticallyNotSuppressible || (ssp eq Never)) { + spos match { + case Prefix => unparseJustSeparator(state) + case Infix => if (wroteAny) unparseJustSeparator(state) + case Postfix => pendingPostfixSeparatorAtTerm = true + } + wroteAny = true + return + } + ssp match { + case Never => + Assert.invariantFailed("handled above") + case AnyEmpty | TrailingEmpty | TrailingEmptyStrict => + spos match { + case Prefix | Infix => + if ((spos eq Infix) && !wroteAny) { + // no separator possible; hence, no suppression (mirrors + // unparseOneWithSuppression's `state.groupPos == 1` case) + } else { + val suspendableOp = + new SuppressableSeparatorUnparserSuspendableOperation( + sepMtaAlignmentMaybe, + sep, + trd + ) + val suppressableSep = SuppressableSeparatorUnparser(sep, trd, suspendableOp) + suppressableSep.unparse1(state) + pendingSuppressibleOp = suspendableOp + } + case Postfix => + val suspendableOp = + new SuppressableSeparatorUnparserSuspendableOperation( + sepMtaAlignmentMaybe, + sep, + trd + ) + suspendableOp.captureDOSForStartOfSeparatedRegionBeforePostfixSeparator(state) + pendingSuppressibleOp = suspendableOp + } + } + wroteAny = true + } + + /** + * Completes whatever beforeSeparator started for a term. Handles ssp + * Never (pendingPostfixSeparatorAtTerm, immediate write) and + * AnyEmpty/TrailingEmpty(Strict) (pendingSuppressibleOp); exactly one + * is ever set, matching beforeSeparator's own ssp branch. + */ + def afterSeparator(): Unit = { + if (pendingPostfixSeparatorAtTerm) { + pendingPostfixSeparatorAtTerm = false + unparseJustSeparator(state) + } + if (pendingSuppressibleOp ne null) { + val op = pendingSuppressibleOp + pendingSuppressibleOp = null + ssp match { + case AnyEmpty => + spos match { + case Prefix | Infix => + op.captureStateAtEndOfPotentiallyZeroLengthRegionFollowingTheSeparator(state) + case Postfix => + op.captureDOSForEndOfSeparatedRegionBeforePostfixSeparator(state) + SuppressableSeparatorUnparser(sep, op.rd, op).unparse1(state) + op.captureStateAtEndOfPotentiallyZeroLengthRegionFollowingTheSeparator(state) + } + case TrailingEmpty | TrailingEmptyStrict => + spos match { + case Prefix | Infix => + trailingSuspendedOps += op + case Postfix => + op.captureDOSForEndOfSeparatedRegionBeforePostfixSeparator(state) + SuppressableSeparatorUnparser(sep, op.rd, op).unparse1(state) + trailingSuspendedOps += op + } + case Never => + Assert.invariantFailed("pendingSuppressibleOp should never be set for ssp Never") + } + } + } + + // The term produced zero occurrences (a different term's child took + // this position, or build finished with nothing more coming); no + // tree node was added, so position doesn't move, but it may still owe + // separators, like an actual occurrence loop's own end-of-term handling. + def zeroOccurrences(rep: RepeatingChildUnparser with Separated): Unit = { + if (ssp eq Never) { + writeExtraSeparatorsIfNeverSuppressed(rep, 0) + } else { + writePositionallyRequiredSepsIfSuppressed(rep, 0, stacksAlreadyPushed = false) + } + } + + // ssp TrailingEmpty(Strict): now that nothing at all remains in this + // sequence, every deferred separator's "after" boundary is this exact + // point. Resolve them all, mirroring unparseWithSuppression's identical + // end-of-loop step. + def resolveTrailingSuspended(): Unit = { + if ((ssp eq TrailingEmpty) || (ssp eq TrailingEmptyStrict)) { + trailingSuspendedOps.foreach { + _.captureStateAtEndOfPotentiallyZeroLengthRegionFollowingTheSeparator(state) + } + trailingSuspendedOps.clear() + } + } + } + + /** + * Walks childUnparsers positionally against an already-built + * containerNode, writing each term's content/separator per spos; blocks + * via childExistsOrFinal/awaitChild wherever a needed child isn't ready. + * childIndexStack.top tracks position (an array is ONE slot). + */ + override def writeContent(containerNode: DINode, state: UState): Unit = { + val sharedCtx = state.sharedContext.get + val complex = containerNode.asComplex + val sepState = new SeparatorSuppressionState(state) + + var index = 0 + while (index < childUnparsers.length) { + childUnparsers(index) match { + case rep: RepeatingChildUnparser => + writeRepeatingTerm(rep, complex, sharedCtx, state, sepState) + case cu => + writeRequiredTerm(cu, complex, sharedCtx, state, sepState) + } + index += 1 + } + + sepState.resolveTrailingSuspended() + } + + /** + * Writes an array/optional term: either its full occurrence loop (an + * actual `DIArray`, or a scalar optional's single occurrence) or, if the + * term produced no occurrences at all, its zero-occurrences separator + * bookkeeping. + */ + private def writeRepeatingTerm( + rep: RepeatingChildUnparser, + complex: DIComplex, + sharedCtx: UnparseSharedContext, + state: UState, + sepState: SeparatorSuppressionState + ): Unit = { + val idx = state.childIndexStack.top.toInt + if (sharedCtx.childExistsOrFinal(complex, idx)) { + val next = complex.child(idx) + if (next.erd eq rep.erd) { + next match { + case arrayNode: DIArray => + val repSep = rep.asInstanceOf[RepeatingChildUnparser with Separated] + // A dfdl:occursIndex() expression in the occurrence's content + // reads state.occursIndexStack.top, which must track it or + // every occurrence would evaluate as if it were the first. + state.arrayIterationIndexStack.push(1L) + state.occursIndexStack.push(1L) + var arrayOcc = 0 + while (sharedCtx.childExistsOrFinal(arrayNode, arrayOcc)) { + val occNode = sharedCtx.awaitChild(arrayNode, arrayOcc) + // Postfix's separator comes after this occurrence's + // content; afterSeparator runs once it's written. + sepState.beforeSeparator( + repSep.erd, + repSep.isKnownStaticallyNotToSuppressSeparator + ) + repSep.childUnparser + .asInstanceOf[ElementUnparserBase] + .writeContent(occNode, state) + sepState.afterSeparator() + arrayNode.freeChildIfNoLongerNeeded(arrayOcc, state.releaseUnneededInfoset) + arrayOcc += 1 + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + } + // this whole array term is done; it occupies exactly one + // slot among containerNode's own children, regardless of + // how many occurrences it held + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + state.moveOverOneElementChildOnly() + if (ssp eq Never) { + sepState.writeExtraSeparatorsIfNeverSuppressed(repSep, arrayOcc) + } else { + sepState.writePositionallyRequiredSepsIfSuppressed( + repSep, + arrayOcc, + stacksAlreadyPushed = true + ) + } + state.arrayIterationIndexStack.pop() + state.occursIndexStack.pop() + case scalarOptional => + val repSep = rep.asInstanceOf[RepeatingChildUnparser with Separated] + val readyChild = sharedCtx.awaitChild(complex, idx) + // Same push as a true array's entry (see above); a + // scalar optional is still a RepeatingChildUnparser (just + // one that can never hold more than one occurrence). + state.arrayIterationIndexStack.push(1L) + state.occursIndexStack.push(1L) + // Postfix's separator comes after this element's + // content; afterSeparator runs once it's written. + sepState.beforeSeparator( + repSep.erd, + repSep.isKnownStaticallyNotToSuppressSeparator + ) + repSep.childUnparser + .asInstanceOf[ElementUnparserBase] + .writeContent(readyChild, state) + sepState.afterSeparator() + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + // Matches the DIArray arm's own per-occurrence advancement: + // occursIndexStack.top must reflect the next (missing) + // occurrence's index before writePositionallyRequiredSepsIfSuppressed's + // loop runs, or its first separator would see index 1, not 2. + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + if (ssp eq Never) { + sepState.writeExtraSeparatorsIfNeverSuppressed(repSep, 1) + } else { + sepState.writePositionallyRequiredSepsIfSuppressed( + repSep, + 1, + stacksAlreadyPushed = true + ) + } + state.arrayIterationIndexStack.pop() + state.occursIndexStack.pop() + } + } else { + // this term produced zero occurrences: a different term's + // child appeared in this tree position instead + sepState.zeroOccurrences(rep.asInstanceOf[RepeatingChildUnparser with Separated]) + } + } else { + // no more children will ever come; this optional/array term is absent + sepState.zeroOccurrences(rep.asInstanceOf[RepeatingChildUnparser with Separated]) + } + } + + /** + * Writes a required term (exactly one occurrence): a simple/complex + * element, a nested bare group, or a statement-only term with no tree + * child of its own. Includes its own separator bookkeeping if represented. + */ + private def writeRequiredTerm( + cu: SequenceChildUnparser with Separated, + complex: DIComplex, + sharedCtx: UnparseSharedContext, + state: UState, + sepState: SeparatorSuppressionState + ): Unit = { + cu.childUnparser match { + case nvi: NewVariableInstanceEndUnparser => + // Always runs, with no isBuildOnly gate: build's own recursion + // already popped BuildState's local NVI scope stack; write's call + // performs the polymorphically-different pop of the shared vTable, + // which only write's call ever does. No tree child, no separator. + nvi.unparse1(state) + case align: AlignmentPrimUnparser => + // Padding has no non-idempotent side effect (unlike setVariable/assert + // below), so re-running it here is required, not forbidden: build's + // own recursion only ran it against build's no-op-sink DOS. Always + // represented, and never suppressible since its content is deterministic. + if (cu.trd.isRepresented) { + sepState.beforeSeparator(cu.trd, staticallyNotSuppressible = true) + } + align.unparse1(state) + if (cu.trd.isRepresented) { + sepState.afterSeparator() + } + case statementOnly if !statementOnly.isInstanceOf[WriteUnparser] => + // No tree child, so no separator, and nothing to wait on. Build's own + // unconditional recursion already ran this term exactly once; running + // it again would re-evaluate a dfdl:setVariable/assert/discriminator + // expression (e.g. "cannot set variable twice"). + () + case _ => + val idx = state.childIndexStack.top.toInt + // A nested bare group (e.g. a choice) may resolve to a branch with no + // infoset footprint at all; once build is done and no child showed up, + // let the group's own dispatch decide. A plain element term always + // needs an actual child, so it always waits unconditionally. + val isGroupTerm = !cu.childUnparser.isInstanceOf[ElementUnparserBase] && + cu.childUnparser.isInstanceOf[WriteUnparser] + if (isGroupTerm) { + if (sharedCtx.childExistsOrFinal(complex, idx)) { + sharedCtx.awaitChild(complex, idx) + } + } else { + sharedCtx.awaitChild(complex, idx) + } + // A non-represented term gets no separator, so wroteAny must not + // flip true. Suppression only applies to terms whose presence is + // uncertain (array/optional, or a bare group not statically known + // to need it), never to a plain element's own content length. + val useSuppression = isGroupTerm && !cu.isKnownStaticallyNotToSuppressSeparator + if (cu.trd.isRepresented) { + sepState.beforeSeparator(cu.trd, staticallyNotSuppressible = !useSuppression) + } + cu.childUnparser match { + case elemUnp: ElementUnparserBase => + elemUnp.writeContent(complex.child(idx), state) + sepState.afterSeparator() + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + case wu: WriteUnparser => + wu.writeContent(complex, state) + sepState.afterSeparator() + case other => + Assert.usageError(s"unhandled sequence term unparser type: $other") + } + } + } + /** * Unparses one occurrence with associated separator (non-suppressable). */ diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLength2.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLength2.scala index e5efbdd00b..8cf5d13b9d 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLength2.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLength2.scala @@ -228,12 +228,19 @@ class CaptureStartOfContentLengthUnparser(override val context: ElementRuntimeDa override val runtimeDependencies = Array() override def unparse(state: UState): Unit = { - val dos = state.getDataOutputStream - val elem = state.currentInfosetNode.asInstanceOf[DIElement] - if (dos.maybeAbsBitPos0b.isDefined) { - elem.contentLength.setAbsStartPos0bInBits(dos.maybeAbsBitPos0b.getULong) - } else { - elem.contentLength.setRelStartPos0bInBits(dos.relBitPos0b, dos) + // Write-only: this reaches write via a generic non-WriteUnparser + // fallback, which has no way to know build already (uselessly) ran + // it on build's fake DOS. Without this gate, build's call sets the + // position once (wrong), and write's later call trips the "must + // only be set once" invariant. + if (!state.isBuildOnly) { + val dos = state.getDataOutputStream + val elem = state.currentInfosetNode.asInstanceOf[DIElement] + if (dos.maybeAbsBitPos0b.isDefined) { + elem.contentLength.setAbsStartPos0bInBits(dos.maybeAbsBitPos0b.getULong) + } else { + elem.contentLength.setRelStartPos0bInBits(dos.relBitPos0b, dos) + } } } } @@ -246,27 +253,32 @@ class CaptureEndOfContentLengthUnparser( override val runtimeDependencies = Array() override def unparse(state: UState): Unit = { - val dos = state.getDataOutputStream - val elem = state.currentInfosetNode.asInstanceOf[DIElement] - - if ( - elem.contentLength.isStartAbsolute && dos.maybeAbsBitPos0b.isEmpty && maybeFixedLengthInBits.isDefined - ) { - // If this element has an absolute starting bit position, but the current - // DOS bit position is only known relatively, that means there was some - // suspension related to this element that had an unknown length. - // However, if this is a fixed length element, we can calculate the - // absolute position of this DOS based on its absolute starting position - // position, its length, and the relative position of the current DOS - val startAbsBitPos0b: ULong = elem.contentLength.maybeStartPos0bInBits.getULong - val currentAbsPos0b = startAbsBitPos0b + maybeFixedLengthInBits.getULong - dos.setAbsStartingBitPos0b(currentAbsPos0b - dos.relBitPos0b) - } + // Write-only, for the same reason as content-length capture: build + // already (uselessly) runs this on its own fake DOS, and a second, + // write-side call would trip the "must only be set once" invariant. + if (!state.isBuildOnly) { + val dos = state.getDataOutputStream + val elem = state.currentInfosetNode.asInstanceOf[DIElement] - if (dos.maybeAbsBitPos0b.isDefined) { - elem.contentLength.setAbsEndPos0bInBits(dos.maybeAbsBitPos0b.getULong) - } else { - elem.contentLength.setRelEndPos0bInBits(dos.relBitPos0b, dos) + if ( + elem.contentLength.isStartAbsolute && dos.maybeAbsBitPos0b.isEmpty && maybeFixedLengthInBits.isDefined + ) { + // If this element has an absolute starting bit position, but the current + // DOS bit position is only known relatively, that means there was some + // suspension related to this element that had an unknown length. + // However, if this is a fixed length element, we can calculate the + // absolute position of this DOS based on its absolute starting position + // position, its length, and the relative position of the current DOS + val startAbsBitPos0b: ULong = elem.contentLength.maybeStartPos0bInBits.getULong + val currentAbsPos0b = startAbsBitPos0b + maybeFixedLengthInBits.getULong + dos.setAbsStartingBitPos0b(currentAbsPos0b - dos.relBitPos0b) + } + + if (dos.maybeAbsBitPos0b.isDefined) { + elem.contentLength.setAbsEndPos0bInBits(dos.maybeAbsBitPos0b.getULong) + } else { + elem.contentLength.setRelEndPos0bInBits(dos.relBitPos0b, dos) + } } } } @@ -277,12 +289,17 @@ class CaptureStartOfValueLengthUnparser(override val context: ElementRuntimeData override val runtimeDependencies = Array() override def unparse(state: UState): Unit = { - val dos = state.getDataOutputStream - val elem = state.currentInfosetNode.asInstanceOf[DIElement] - if (dos.maybeAbsBitPos0b.isDefined) { - elem.valueLength.setAbsStartPos0bInBits(dos.maybeAbsBitPos0b.getULong) - } else { - elem.valueLength.setRelStartPos0bInBits(dos.relBitPos0b, dos) + // Write-only, for the same reason as content-length capture: build + // already (uselessly) runs this on its own fake DOS, and a second, + // write-side call would trip the "must only be set once" invariant. + if (!state.isBuildOnly) { + val dos = state.getDataOutputStream + val elem = state.currentInfosetNode.asInstanceOf[DIElement] + if (dos.maybeAbsBitPos0b.isDefined) { + elem.valueLength.setAbsStartPos0bInBits(dos.maybeAbsBitPos0b.getULong) + } else { + elem.valueLength.setRelStartPos0bInBits(dos.relBitPos0b, dos) + } } } } @@ -293,12 +310,17 @@ class CaptureEndOfValueLengthUnparser(override val context: ElementRuntimeData) override val runtimeDependencies = Array() override def unparse(state: UState): Unit = { - val dos = state.getDataOutputStream - val elem = state.currentInfosetNode.asInstanceOf[DIElement] - if (dos.maybeAbsBitPos0b.isDefined) { - elem.valueLength.setAbsEndPos0bInBits(dos.maybeAbsBitPos0b.getULong) - } else { - elem.valueLength.setRelEndPos0bInBits(dos.relBitPos0b, dos) + // Write-only, for the same reason as content-length capture: build + // already (uselessly) runs this on its own fake DOS, and a second, + // write-side call would trip the "must only be set once" invariant. + if (!state.isBuildOnly) { + val dos = state.getDataOutputStream + val elem = state.currentInfosetNode.asInstanceOf[DIElement] + if (dos.maybeAbsBitPos0b.isDefined) { + elem.valueLength.setAbsEndPos0bInBits(dos.maybeAbsBitPos0b.getULong) + } else { + elem.valueLength.setRelEndPos0bInBits(dos.relBitPos0b, dos) + } } } } @@ -349,7 +371,7 @@ class TargetLengthOperation( * Several sub-unparsers need to have the value length, and the target length * in order to compute their own length. */ -sealed trait NeedValueAndTargetLengthMixin { +sealed trait NeedValueAndTargetLengthMixin { self: SuspendableOperation => def targetLengthEv: Evaluatable[MaybeJULong] def maybeLengthEv: Maybe[LengthEv] diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala index 0f9b8c0eae..5820e7c5b5 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/SpecifiedLengthUnparsers.scala @@ -22,6 +22,7 @@ import org.apache.daffodil.lib.schema.annotation.props.gen.LengthUnits import org.apache.daffodil.lib.schema.annotation.props.gen.Representation import org.apache.daffodil.lib.util.Maybe.* import org.apache.daffodil.runtime1.infoset.DIElement +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.infoset.DISimple import org.apache.daffodil.runtime1.infoset.Infoset import org.apache.daffodil.runtime1.processors.CharsetEv @@ -35,7 +36,8 @@ final class SpecifiedLengthExplicitImplicitUnparser( eUnparser: Unparser, erd: ElementRuntimeData, targetLengthInBitsEv: UnparseTargetLengthInBitsEv -) extends CombinatorUnparser(erd) { +) extends CombinatorUnparser(erd) + with WriteUnparser { override val runtimeDependencies = Array() @@ -52,7 +54,7 @@ final class SpecifiedLengthExplicitImplicitUnparser( dcs } - override final def unparse(state: UState): Unit = { + private def checkVariableWidthComplexType(state: UState): Unit = { lazy val dcs = getCharset(state) if ( erd.impliedRepresentation == Representation.Text && @@ -66,10 +68,26 @@ final class SpecifiedLengthExplicitImplicitUnparser( lengthKind.toString, lengthUnits.toString ) - } else { - eUnparser.unparse1(state) } } + + override final def unparse(state: UState): Unit = { + checkVariableWidthComplexType(state) + eUnparser.unparse1(state) + } + + // Without this, a SeqCompUnparser wrapping this class would treat + // eUnparser as a synchronous call via its generic fallback, but + // eUnparser can itself be a resumable group unparser expecting live + // InfosetInputter events, desyncing build's event stream entirely. + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop( + containerNode, + eUnparser, + state, + setup = checkVariableWidthComplexType, + teardown = (_, _) => () + ) } /** @@ -139,13 +157,48 @@ class SpecifiedLengthPrefixedUnparser( override val lengthUnits: LengthUnits, override val prefixedLengthAdjustmentInUnits: Long ) extends CombinatorUnparser(erd) - with CalculatedPrefixedLengthUnparserMixin { + with CalculatedPrefixedLengthUnparserMixin + with WriteUnparser { override val runtimeDependencies = Array() override def childProcessors = Vector(prefixedLengthUnparser, eUnparser) override def unparse(state: UState): Unit = { + if (state.isBuildOnly) { + // Build's runContentUnparser unconditionally recurses into eUnparser + // to navigate/construct the tree. Without this early return, build + // would also create its own detached prefix-length element and + // suspension against its no-op-sink DOS, one that can never resolve, + // permanently deadlocking SuspensionTracker.requireFinal. + eUnparser.unparse1(state) + return + } + val plElem = pushDetachedPrefixLengthElement(state) + eUnparser.unparse1(state) + resolvePrefixLength(state, state.currentInfosetNode.asInstanceOf[DIElement], plElem) + } + + // Without this, WriteUnparser dispatch (a plain recursive-dispatch + // fallback for a group-wrapped eUnparser) would call eUnparser.unparse1 + // synchronously, but it can itself be a resumable group unparser + // expecting live InfosetInputter events. + override def writeContent(containerNode: DINode, state: UState): Unit = + writeWithPushPop( + containerNode, + eUnparser, + state, + setup = pushDetachedPrefixLengthElement, + teardown = { (state, plElem) => + // resolvePrefixLength (via assignPrefixLength/suspension.run) + // expects state.processor to already be set, normally done by + // Unparser.unparse1, which this recursive-dispatch path bypasses. + state.setProcessor(SpecifiedLengthPrefixedUnparser.this) + resolvePrefixLength(state, containerNode.asInstanceOf[DIElement], plElem) + } + ) + + private def pushDetachedPrefixLengthElement(state: UState): DISimple = { // Create a "detached" DIDocument with a single child element that the // prefix length will be parsed to. This creates a completely new // infoset and parses to that, so care is taken to ensure this infoset @@ -160,10 +213,10 @@ class SpecifiedLengthPrefixedUnparser( state.currentInfosetNodeStack.push(One(plElem)) prefixedLengthUnparser.unparse1(state) state.currentInfosetNodeStack.pop + plElem + } - val elem = state.currentInfosetNode.asInstanceOf[DIElement] - eUnparser.unparse1(state) - + private def resolvePrefixLength(state: UState, elem: DIElement, plElem: DISimple): Unit = { if (elem.contentLength.maybeLengthInBits().isDefined) { // If we were able to immediately calculate the length of the element, // then just set it as the value of the detached element created above so diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala index 8b4488a884..f180b2131c 100644 --- a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/UnseparatedSequenceUnparsers.scala @@ -18,6 +18,9 @@ package org.apache.daffodil.unparsers.runtime1 import org.apache.daffodil.lib.exceptions.Assert import org.apache.daffodil.lib.schema.annotation.props.gen.OccursCountKind +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.runtime1.infoset.DIComplex +import org.apache.daffodil.runtime1.infoset.DINode import org.apache.daffodil.runtime1.processors.ElementRuntimeData import org.apache.daffodil.runtime1.processors.SequenceRuntimeData import org.apache.daffodil.runtime1.processors.TermRuntimeData @@ -56,7 +59,8 @@ class RepOrderedUnseparatedSequenceChildUnparser( class OrderedUnseparatedSequenceUnparser( rd: SequenceRuntimeData, childUnparsers: Array[SequenceChildUnparser] -) extends OrderedSequenceUnparserBase(rd) { +) extends OrderedSequenceUnparserBase(rd) + with WriteUnparser { // Sequences of nothing (no initiator, no terminator, nothing at all) should // have been optimized away @@ -66,6 +70,141 @@ class OrderedUnseparatedSequenceUnparser( override def childProcessors = childUnparsers.toVector + /** + * Same shape as OrderedSeparatedSequenceUnparser's writeContent, minus + * separator writing. Without this, ElementUnparserBase.writeContent's + * WriteUnparser check would be false here, falling through to the + * production unparse1 path against an untouched inputter. + */ + override def writeContent(containerNode: DINode, state: UState): Unit = { + val sharedCtx = state.sharedContext.get + val complex = containerNode.asComplex + + var index = 0 + while (index < childUnparsers.length) { + childUnparsers(index) match { + case rep: RepeatingChildUnparser => + writeRepeatingTerm(rep, complex, sharedCtx, state) + case cu => + writeRequiredTerm(cu, complex, sharedCtx, state) + } + index += 1 + } + } + + /** + * Writes an array/optional term's full occurrence loop, if it produced + * any occurrences; nothing to do otherwise (no separator bookkeeping to + * account for either way, unlike OrderedSeparatedSequenceUnparser's own + * writeRepeatingTerm). + */ + private def writeRepeatingTerm( + rep: RepeatingChildUnparser, + complex: DIComplex, + sharedCtx: UnparseSharedContext, + state: UState + ): Unit = { + val idx = state.childIndexStack.top.toInt + if (sharedCtx.childExistsOrFinal(complex, idx)) { + val next = complex.child(idx) + if (next.erd eq rep.erd) { + next match { + case arrayNode: DIArray => + // Mirrors the separated sequence's identical push. See + // its own comment. + state.arrayIterationIndexStack.push(1L) + state.occursIndexStack.push(1L) + var arrayOcc = 0 + while (sharedCtx.childExistsOrFinal(arrayNode, arrayOcc)) { + val occNode = sharedCtx.awaitChild(arrayNode, arrayOcc) + rep.childUnparser + .asInstanceOf[ElementUnparserBase] + .writeContent(occNode, state) + arrayNode.freeChildIfNoLongerNeeded(arrayOcc, state.releaseUnneededInfoset) + arrayOcc += 1 + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + } + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + state.moveOverOneElementChildOnly() + state.arrayIterationIndexStack.pop() + state.occursIndexStack.pop() + case scalarOptional => + val readyChild = sharedCtx.awaitChild(complex, idx) + // Mirrors the separated sequence's identical pairing. See + // its own comment for why occursIndexStack must track the + // actual occurrence being written (dfdl:occursIndex() in + // the occurrence's own content). + state.arrayIterationIndexStack.push(1L) + state.occursIndexStack.push(1L) + rep.childUnparser + .asInstanceOf[ElementUnparserBase] + .writeContent(readyChild, state) + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + state.moveOverOneArrayIterationIndexOnly() + state.moveOverOneOccursIndexOnly() + state.arrayIterationIndexStack.pop() + state.occursIndexStack.pop() + } + } + // else: this term produced zero occurrences; a different term's + // child appeared in this tree position instead, and there's no + // separator bookkeeping to resolve for it here, unlike the + // separated sequence's own zeroOccurrences. + } + // else: no more children will ever come; this optional/array term is + // absent, again with nothing further to do about it here. + } + + /** + * Writes a required term: a simple or complex element, a nested bare + * group, or a statement-only term with no tree child of its own. + */ + private def writeRequiredTerm( + cu: SequenceChildUnparser, + complex: DIComplex, + sharedCtx: UnparseSharedContext, + state: UState + ): Unit = { + cu.childUnparser match { + // A plain scalar element term: await its one tree child, write + // its content, then advance the element-child position. + case elemUnp: ElementUnparserBase => + val idx = state.childIndexStack.top.toInt + val child = sharedCtx.awaitChild(complex, idx) + elemUnp.writeContent(child, state) + state.moveOverOneElementChildOnly() + complex.freeChildIfNoLongerNeeded(idx, state.releaseUnneededInfoset) + case wu: WriteUnparser => + val idx = state.childIndexStack.top.toInt + // This nested group (e.g. a choice) may have resolved to a branch + // with no infoset footprint at all; only wait for a child's + // readiness when childExistsOrFinal says one genuinely exists. + if (sharedCtx.childExistsOrFinal(complex, idx)) { + sharedCtx.awaitChild(complex, idx) + } + wu.writeContent(complex, state) + case nvi: NewVariableInstanceEndUnparser => + // Always runs, with no isBuildOnly gate: build's own recursion + // already popped BuildState's local NVI scope stack; write's call + // performs the polymorphically-different pop of the shared vTable, + // which only write's call ever does. + nvi.unparse1(state) + case align: AlignmentPrimUnparser => + // Padding has no non-idempotent side effect (unlike setVariable/assert + // below), so re-running it here is required, not forbidden: build's + // own recursion only ran it against build's no-op-sink DOS. + align.unparse1(state) + case _ => + // Build's own unconditional recursion already ran this term exactly + // once; running it again would re-evaluate a + // dfdl:setVariable/assert/discriminator expression (e.g. "cannot + // set variable twice"). + () + } + } + /** * Unparses one iteration of an array/optional element */ diff --git a/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala new file mode 100644 index 0000000000..661304f1a0 --- /dev/null +++ b/daffodil-core/src/main/scala/org/apache/daffodil/unparsers/runtime1/WriteUnparser.scala @@ -0,0 +1,80 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.unparsers.runtime1 + +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.processors.unparsers.* + +/** + * Write-side dispatch for the build/write-prefetch unparse path, by + * ordinary recursive calls, using `UnparseSharedContext.awaitChild` to + * block (park write's coroutine thread, see `BuildWriteCoroutines.scala`) + * wherever a needed child doesn't yet exist or isn't ready. + */ +trait WriteUnparser { + + // Writes containerNode's content via direct recursive calls on write's + // own coroutine thread - "where we are" is just the JVM call stack, not + // a return-value state machine. May block (via + // state.sharedContext.get.awaitChild) until a needed child exists and is ready, then resumes exactly where it left off. + def writeContent(containerNode: DINode, state: UState): Unit + + // Shared push-once/pop-once skeleton: setup runs before recursing into + // bodyUnparser (dispatched to writeContent if it's a WriteUnparser, else + // plain unparse1), teardown runs once that call returns with setup's + // result (e.g. a prefixed-length unparser threads its detached element from setup to teardown, where it's finally assigned a value). + protected def writeWithPushPop[A]( + containerNode: DINode, + bodyUnparser: Unparser, + state: UState, + setup: UState => A, + teardown: (UState, A) => Unit + ): Unit = { + val setupResult = setup(state) + // Unlike single-pass unparse() (which aborts entirely on exception), + // write's AwaitChildStalledException is caught higher up and followed + // by finishWriteSide's invariant checks against this same state, so + // teardown must still run here, or a stall would leave state + // permanently imbalanced and fail those checks for an unrelated reason. + try { + bodyUnparser match { + case wu: WriteUnparser => wu.writeContent(containerNode, state) + case _ => bodyUnparser.unparse1(state) + } + } finally { + teardown(state, setupResult) + } + } + + /** + * `writeWithPushPop` for a combinator with nothing to push or pop; just + * the dispatch-to-`writeContent`-or-`unparse1` part. + */ + protected def writeWithPushPop( + containerNode: DINode, + bodyUnparser: Unparser, + state: UState + ): Unit = + writeWithPushPop( + containerNode, + bodyUnparser, + state, + (_: UState) => (), + (_: UState, _: Unit) => () + ) +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/core/outputValueCalc/TestOutputValueCalcPendingSuspensionRetry.scala b/daffodil-core/src/test/scala/org/apache/daffodil/core/outputValueCalc/TestOutputValueCalcPendingSuspensionRetry.scala new file mode 100644 index 0000000000..acfb719729 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/core/outputValueCalc/TestOutputValueCalcPendingSuspensionRetry.scala @@ -0,0 +1,118 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.core.outputValueCalc + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils + +import org.junit.Test + +/** + * Regression guard: a dfdl:valueLength forward reference must survive + * many sweep cycles, including two fields sharing the same target. + */ +class TestOutputValueCalcPendingSuspensionRetry { + + private val example = XMLUtils.EXAMPLE_NAMESPACE + + private val N = 12 + + private def schema = { + val record = + + + + + + + + + SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + + + {record} + + + , + elementFormDefault = "unqualified" + ) + } + + private def infoset = { + val recordXml = (1 to N).map { i => + {f"T$i%03d"}{f"data$i%04d"} + } + +
+ {recordXml} + + } + + // Every len field is a fixed-length "data" element (8 bytes), so all + // four should compute to 8 regardless of which record they target. + private def expectedOutput = { + val header = " 8 8 8 8" + val records = (1 to N).map(i => f"T$i%03d" + f"data$i%04d").mkString + header + records + } + + @Test def testManyForceRetryCyclesResolveCorrectly(): Unit = { + TestUtils.testUnparsing( + schema, + infoset, + expectedOutput, + tunables = Map("unparseSuspensionWaitOld" -> "1", "unparseSuspensionWaitYoung" -> "1") + ) + } + + @Test def testManyForceRetryCyclesDefaultTunablesStillMatch(): Unit = { + // Same schema at default tunables: confirms the pass above isn't + // vacuous (e.g. a schema mistake unrelated to the tunable). + TestUtils.testUnparsing(schema, infoset, expectedOutput) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/core/outputValueCalc/TestOutputValueCalcReadsOutputValueCalc.scala b/daffodil-core/src/test/scala/org/apache/daffodil/core/outputValueCalc/TestOutputValueCalcReadsOutputValueCalc.scala new file mode 100644 index 0000000000..7b39755564 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/core/outputValueCalc/TestOutputValueCalcReadsOutputValueCalc.scala @@ -0,0 +1,119 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.core.outputValueCalc + +import java.nio.channels.Channels + +import org.apache.daffodil.core.compiler.Compiler +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter +import org.apache.daffodil.runtime1.processors.DataProcessor + +import org.junit.Assert.* +import org.junit.Test + +/** + * Regression guard: lenA's OVC reads lenB's own OVC forward-reference + * value, resolvable only via requireFinal's final drain (retrying lenA). + */ +class TestOutputValueCalcReadsOutputValueCalc { + + private val example = XMLUtils.EXAMPLE_NAMESPACE + + private val N = 3 + + private def schema = { + val record = + + + + + + + + + SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + {record} + + + , + elementFormDefault = "unqualified" + ) + } + + private def infoset = { + val recordXml = (1 to N).map { i => + {f"T$i%03d"}{f"data$i%04d"} + } + +
+ {recordXml} + + } + + private def compile(tunables: Map[String, String]): DataProcessor = { + val compiler = Compiler().withTunables(tunables) + val pf = compiler.compileNode(schema) + if (pf.isError) fail(pf.getDiagnostics.toString) + val dp = pf.onPath("/").asInstanceOf[DataProcessor] + if (dp.isError) fail(dp.getDiagnostics.toString) + dp + } + + @Test def testOvcReadsOvcResolvesViaFinalDrain(): Unit = { + val dp = compile(Map("unparseSuspensionWaitOld" -> "1000000")) + + val outputStream = new java.io.ByteArrayOutputStream() + val out = Channels.newChannel(outputStream) + val inputter = new ScalaXMLInfosetInputter(infoset) + val actual = dp.unparse(inputter, out) + out.close() + assertFalse(actual.getDiagnostics.toString, actual.isProcessingError) + + val unparsed = outputStream.toString + // lenA reads lenB's own value (not its length): both should equal + // record[N]/data's length, 8. + assertEquals(" 8 8", unparsed.substring(0, 8)) + + val recordsPart = unparsed.substring(8) + val expectedRecords = (1 to N).map(i => f"T$i%03d" + f"data$i%04d").mkString + assertEquals(expectedRecords, recordsPart) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/lib/util/TestMStack.scala b/daffodil-core/src/test/scala/org/apache/daffodil/lib/util/TestMStack.scala index b4cebcdbcb..5905e6deb3 100644 --- a/daffodil-core/src/test/scala/org/apache/daffodil/lib/util/TestMStack.scala +++ b/daffodil-core/src/test/scala/org/apache/daffodil/lib/util/TestMStack.scala @@ -17,6 +17,11 @@ package org.apache.daffodil.lib.util +import org.apache.daffodil.lib.util.Maybe.* + +import org.junit.Assert.* +import org.junit.Test + /** * Compare MStack performance to ArrayStack. It should be faster for primitives */ @@ -24,6 +29,107 @@ class TestMStack { var junk: Long = 0 + // Copying into a sized destination must match copying into a default one. + @Test def testMStackOfCopyFromSizedDestinationMatchesDefault(): Unit = { + val source = new MStackOf[String] + source.push("a") + source.push("b") + source.push("c") + + val sizedDest = new MStackOf[String](source.length) + sizedDest.copyFrom(source) + + val defaultDest = new MStackOf[String] + defaultDest.copyFrom(source) + + assertEquals(defaultDest.toList, sizedDest.toList) + assertEquals(3, sizedDest.length) + assertEquals("c", sizedDest.top) + } + + // A sized-to-exact-depth destination must still grow past capacity. + @Test def testMStackOfSizedDestinationStillGrowsCorrectly(): Unit = { + val source = new MStackOf[String] + source.push("a") + source.push("b") + + val sizedDest = new MStackOf[String](source.length) + sizedDest.copyFrom(source) + + sizedDest.push("c") + sizedDest.push("d") + sizedDest.push("e") + + assertEquals(5, sizedDest.length) + assertEquals("e", sizedDest.top) + assertEquals(List("e", "d", "c", "b", "a"), sizedDest.toList) + } + + /** Same sized-clone pattern, but for MStackOfMaybe. */ + @Test def testMStackOfMaybeCopyFromSizedDestinationMatchesDefault(): Unit = { + val source = new MStackOfMaybe[String] + source.push(One("x")) + source.push(Nope) + source.push(One("z")) + + val sizedDest = new MStackOfMaybe[String](source.length) + sizedDest.copyFrom(source) + + val defaultDest = new MStackOfMaybe[String] + defaultDest.copyFrom(source) + + assertEquals(defaultDest.toListMaybe, sizedDest.toListMaybe) + assertEquals(3, sizedDest.length) + assertEquals(One("z"), sizedDest.top) + assertEquals(One("z"), sizedDest.pop) + assertEquals(Nope, sizedDest.pop) + assertEquals(One("x"), sizedDest.pop) + assertTrue(sizedDest.isEmpty) + } + + /** Same sized-construction pattern, for the primitive-specialized variants. */ + @Test def testMStackOfBooleanSizedConstructionWorks(): Unit = { + val stk = MStackOfBoolean(3) + stk.push(true) + stk.push(false) + stk.push(true) + assertEquals(3, stk.length) + assertEquals(true, stk.pop()) + assertEquals(false, stk.pop()) + assertEquals(true, stk.pop()) + } + + @Test def testMStackOfIntSizedConstructionWorks(): Unit = { + val stk = MStackOfInt(2) + stk.push(1) + stk.push(2) + stk.push(3) + assertEquals(3, stk.length) + assertEquals(3, stk.pop()) + assertEquals(2, stk.pop()) + assertEquals(1, stk.pop()) + } + + @Test def testMStackOfLongSizedConstructionWorks(): Unit = { + val stk = MStackOfLong(1) + stk.push(10L) + stk.push(20L) + assertEquals(2, stk.length) + assertEquals(20L, stk.pop()) + assertEquals(10L, stk.pop()) + } + + // trackMaxSizeReached is a `final val`; confirms it's off by default + // and maxSizeReached stays 0 while it is. + @Test def testMaxSizeReachedDisabledByDefault(): Unit = { + assertEquals(false, MStack.trackMaxSizeReached) + val stk = new MStackOf[String] + stk.push("a") + stk.push("b") + stk.push("c") + assertEquals(0, stk.maxSizeReached) + } + /** * This test compares MStackOfLong to ArrayStack[Long]. * diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/BuildWritePrefetchDataProcessorTest.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/BuildWritePrefetchDataProcessorTest.scala new file mode 100644 index 0000000000..1deb44f29e --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/BuildWritePrefetchDataProcessorTest.scala @@ -0,0 +1,1789 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors + +import java.io.ByteArrayOutputStream +import scala.jdk.CollectionConverters.* + +import org.apache.daffodil.api +import org.apache.daffodil.core.compiler.Compiler +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.externalvars.ExternalVariablesLoader +import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter + +import org.junit.Assert.* +import org.junit.Test + +/** + * Exercises DataProcessor.unparse with useBuildWritePrefetch enabled, + * confirming byte-identical output vs. single-pass across many schema shapes. + */ +class BuildWritePrefetchDataProcessorTest { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + /** A hidden, constant-valued OVC probe wired in via `dfdl:hiddenGroupRef="ex:ovcProbe"`. + * Needed because `hasAnyPrefetchBeneficialOVC` is false for a schema with no resolvable + * OVC, which would silently fall back to single-pass regardless of the + * `useBuildWritePrefetch` tunable; contributes a leading "Z" byte to expected output. */ + val ovcProbeGroup: scala.xml.Elem = + + + + + + + private def unparseToBytes(dp: DataProcessor, infoset: scala.xml.Elem): Array[Byte] = { + val out = new ByteArrayOutputStream() + val res = dp.unparse(new ScalaXMLInfosetInputter(infoset), out) + assertFalse(res.getDiagnostics.toString, res.isError) + out.toByteArray + } + + @Test def testOVCSuspensionSchemaMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = 005 + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + // dfdl:textNumberPadCharacter="0"/textNumberJustification="right" produces + // actual, schema-configured zero-padding, not generic fillByte-based padding. + assertEquals("006005", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + @Test def testArrayChoiceSeparatorSchemaMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = +
H
abcX
+ + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("ZH,a,b,c,X", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A choice reached while withinHiddenNest: both sides pick + // defaultUnparser unconditionally, since the outcome is schema-determined. + @Test def testHiddenChoiceMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = 3 + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("[hello,3", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A postfix-separated array as the last term: the separator must + // still be written after the final occurrence, not dropped. + @Test def testTrailingArrayPostfixSeparatorMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + abc + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("Za\nb\nc\n", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A choice branch with zero tree footprint (absent optional element) + // must still write its own initiator, or write's dispatch can hang. + @Test def testChoiceBranchWithAbsentOptionalElementMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + , + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = 1 + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("Zfirst_defaultable1", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + /** Same schema as above, but with "opt" PRESENT (an actual match, not the empty-branch fallback). */ + @Test def testChoiceBranchWithPresentOptionalElementMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + , + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = 01 + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("Zfirst_defaultable01", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + @Test def testManyOccurrenceArrayWithSmallPrefetchLimitMatchesSinglePass(): Unit = { + val numItems = 40 + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val items = (0 until numItems).map(i => {s"i$i"}) + val infoset = {items} + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals( + "Z" + (0 until numItems).map(i => s"i$i").mkString(","), + new String(singlePassBytes, "ascii") + ) + + // A tiny lookahead window forces many build<->write handoffs across this array's 40 + // occurrences. BoundedPrefetchTest proves the lead counter stays bounded; this test + // just confirms the tunable threads through DataProcessor.unparse correctly and the + // output is still byte-perfect. + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .withTunable("unparsePrefetchWindowNodes", "3") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A bare nested xs:sequence has no tree node of its own; the enclosing + // writeContent must not advance/free a tree-child position recursing into it. + @Test def testNestedBareSequenceMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + BI1I2A + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + // The nested xs:sequence has no dfdl:separator of its own and doesn't inherit the + // outer's "," (a local property of the outer xs:sequence's annotation, not pushed + // down to nested groups), so it compiles as unseparated: no separator between + // inner1/inner2, but the outer's separator still appears before/after the nested group. + assertEquals("ZB,I1I2,A", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A fixed-length choice must pad unused space after a shorter branch, + // not just write the branch content. + @Test def testFixedLengthChoicePaddingMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + // typeB is 2 bytes; the choice's declared length is 5 bytes, so 3 + // bytes of ChoiceUnusedUnparser filler should follow it. Plus the + // leading 1-byte ovcProbe. + val infoset = XY + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals(6, singlePassBytes.length) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A hidden group's elements never get hasValue=true; write's dispatch + // must special-case that instead of blocking on it forever. + @Test def testHiddenGroupMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + + + + + , + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = V + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("HV", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Initiator/terminator with no separator: same delimiter-frame code + // path as the separator tests above, via a different property combination. + @Test def testInitiatorTerminatorMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = AB + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("Z[AB]", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Escape-scheme state is confined entirely to write-only unparsers; + // confirms that end to end with no frame changes needed. + @Test def testEscapeSchemeMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + + + + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = one, twothree + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("Zone#, two,three", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A dfdlx:layer-wrapped sequence: confirms the layer's DOS-splitting + // setup produces byte-identical output under prefetch. + @Test def testLayeredSequenceMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + // s1 is exactly 8 bytes (the layer's declared fixedLength), long enough that + // FixedLengthOutputStream.write's accumulate-then-auto-close-on-count-==fixedLength + // logic runs across several bytes, not just one or two; a short value could pass + // while still masking an off-by-one in the accumulation. + val infoset = ABCDEFGHQ + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("ZABCDEFGHQ", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // FixedLengthLayer's length-exceeded error must surface the same way + // under prefetch, not hang write's coroutine or leak a raw exception. + @Test def testLayeredSequenceLengthMismatchErrorMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + // s1 is 10 bytes, 2 more than the layer's declared fixedLength=8. + val infoset = ABCDEFGHIJQ + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassOut = new ByteArrayOutputStream() + val singlePassRes = + singlePassDp.unparse(new ScalaXMLInfosetInputter(infoset), singlePassOut) + assertTrue("expected a failed UnparseResult, not a successful one", singlePassRes.isError) + assertTrue( + singlePassRes.getDiagnostics + .get(0) + .getMessage + .contains("exceeded fixed layer length of 8") + ) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchOut = new ByteArrayOutputStream() + val prefetchRes = prefetchDp.unparse(new ScalaXMLInfosetInputter(infoset), prefetchOut) + assertTrue("expected a failed UnparseResult, not a successful one", prefetchRes.isError) + assertTrue( + prefetchRes.getDiagnostics + .get(0) + .getMessage + .contains("exceeded fixed layer length of 8") + ) + } + + // A chained suspension: computed1's OVC references computed2 (itself + // an OVC forward reference), two levels deep instead of one. + @Test def testNestedOVCSuspensionsMatchSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = 005 + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("007006005", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Binary integers (every other test here is textual), through an + // array to also exercise the repeating-child frame path. + @Test def testBinaryIntArrayMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + + + + + , + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + 123 + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertArrayEquals( + Array[Byte]('Z'.toByte, 0, 0, 0, 1, 0, 0, 0, 2, 0, 0, 0, 3), + singlePassBytes + ) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A variable-length dfdl:length expression (every other test here is + // constant), exercising computeTargetLength's non-constant branch. + @Test def testVariableLengthExpressionMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = 5hello + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("Z05hello", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A prefixed-length element: the prefix depends on the content's + // unparsed length, a forward dependency like OVC but length-driven. + @Test def testPrefixedLengthMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + , + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = hello + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("Z05hello", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A prefixed-length element with complex (not simple) content, so + // write's dispatch must recurse into the wrapped content unparser. + @Test def testPrefixedLengthComplexContentMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + , + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = hibye + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("Z06hi,bye", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // An OVC referencing fn:count of a preceding array, resolvable + // directly against the tree: the ordinary non-suspending OVC path. + @Test def testOVCCountOfPrecedingArrayMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = 12 + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("122", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A large array where build fully finishes before write's coroutine + // starts, so write's first signal is BuildFinished, not a live coroutine. + @Test def testOVCCountOfPrecedingArrayAfterBuildFullyFinishesMatchesSinglePass(): Unit = { + val numItems = 500 + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val items = (0 until numItems).map(i => {i % 10}) + val infoset = {items} + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A dynamic dfdl:terminator expression referencing a preceding + // array's count: navigation-only, resolvable directly against the tree. + @Test def testDynamicTerminatorReferencingArrayCountMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("Zy", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // An IVC field (no unparse effect) referencing fn:exists over a + // nested array: exercises ordinary navigation past a nested array. + @Test def testIVCExistsOverNestedArrayMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = + + 1234 + true + + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("Z1,2,3,4", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Aggressive throttling forces every "len" suspension to be retried + // repeatedly; guards against re-running an already-resolved one. + @Test def testManyValueLengthForwardReferencesWithAggressiveThrottlingMatchesSinglePass() + : Unit = { + val numRecords = 25 + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val records = (0 until numRecords).map(i => {s"value$i"}) + val infoset = {records} + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .withTunable("unparsePrefetchWindowNodes", "2") + .withTunable("unparseSuspensionWaitYoung", "1") + .withTunable("unparseSuspensionWaitOld", "1") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A repeating NVI scope with a valueLength OVC reading it, letting + // build race many iterations ahead: guards against a stale iteration's value. + @Test def testNVIScopedVariableWithValueLengthOVCMatchesSinglePass(): Unit = { + val numRecords = 25 + val sch = SchemaUtils.dfdlTestSchema( + , { + + + }, + + + + + + + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val records = + (0 until numRecords).map(i => + {f"$i%02d"}{s"value$i"} + ) + val infoset = {records} + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + + // Small window forces build to race many iterations ahead of write, + // each pushing its own runningVar instance, before write catches up. + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .withTunable("unparsePrefetchWindowNodes", "2") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Simpler variant of the repeating-NVI test above: a single + // (non-repeating) scope, as a distinct data point. + @Test def testSingleNVIScopedVariableWithValueLengthOVCMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , { + + + }, + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = 42hello + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Regression guard for NVI-scope setVariable resolution: no forward + // OVC or separator, the minimal shape that exposes the race. + @Test def testNVIScopedSetVariableWithNoForwardReferenceMatchesSinglePass(): Unit = { + val numRecords = 25 + val sch = SchemaUtils.dfdlTestSchema( + , { + + + }, + + + + + + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val records = (0 until numRecords).map(i => {f"$i%02d"}) + val infoset = {records} + + val extVars = + ExternalVariablesLoader.mapToBindings(Map(s"{$example}runningVar" -> "-1").asJava) + + val singlePassDp = Compiler() + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + .withExternalVariables(extVars) + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .withTunable("unparsePrefetchWindowNodes", "2") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + .withExternalVariables(extVars) + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A setup failure (inputter never produces StartDocument) must yield + // a failed UnparseResult, not an NPE from a null error-path state. + @Test def testMalformedInfosetInputterGetsCleanErrorNotNPE(): Unit = { + // Needs at least one prefetch-beneficial (value-only, not length-dependent) OVC, or + // DataProcessor.unparse's dispatch (ssrd.hasAnyPrefetchBeneficialOVC) falls back to + // single-pass regardless of the tunable, and unparseViaBuildThenWrite (the method + // under test) would never run. + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + + // hasNext() = false immediately means initialize()'s + // "!delegate.hasNext" check fires straight away, before any actual + // infoset event is produced; exactly the "never starts with + // StartDocument" failure this guards. + val neverStartsInputter = new api.infoset.InfosetInputter { + override def getEventType() = null + override def getLocalName() = null + override def getNamespaceURI() = null + override def getSimpleText( + primType: org.apache.daffodil.runtime1.dpath.NodeInfo.Kind, + runtimeProperties: java.util.Map[String, String] + ) = null + override def isNilled(): java.lang.Boolean = null + override def hasNext() = false + override def next(): Unit = () + override def fini(): Unit = () + } + + val out = new ByteArrayOutputStream() + val res = dp.unparse(neverStartsInputter, out) + + assertTrue("expected a failed UnparseResult, not a successful one", res.isError) + assertTrue( + res.getDiagnostics.get(0).getMessage.contains("does not start with StartDocument") + ) + } + + // A variable-length dfdl:length expression inside a separated + // sequence must also call computeTargetLength from write's dispatch. + @Test def testDelimitedVariableLengthExpressionMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = 5hello + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("Z05|hello", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // Same dispatch path as the test above, but for a complex element + // whose expression-based dfdl:length wraps a whole group's content chain. + @Test def testDelimitedComplexVariableLengthExpressionMatchesSinglePass(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + Seq( + ovcProbeGroup, + + + + + + + + + + + + + + + + + + ), + elementFormDefault = "unqualified" + ) + val infoset = 3xyz + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + assertEquals("Z03|xyz", new String(singlePassBytes, "ascii")) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A schema where every OVC is content-length-dependent (never + // resolvable early) must fall back to single-pass automatically. + @Test def testPurelyContentLengthOVCSchemaFallsBackAutomatically(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = xyz + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertFalse( + "schema has only a content-length-dependent OVC; should never report prefetch-beneficial", + prefetchDp.ssrd.hasAnyPrefetchBeneficialOVC + ) + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A schema mixing a resolvable OVC with an unrelated + // content-length-dependent one must still use the prefetch path. + @Test def testMixedSchemaWithResolvableAndContentLengthOVCStillUsesPrefetchPath(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val infoset = 005xyz + + val singlePassDp = Compiler().compileNode(sch).onPath("/").asInstanceOf[DataProcessor] + val singlePassBytes = unparseToBytes(singlePassDp, infoset) + + val prefetchDp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertTrue( + "schema has a resolvable-without-writing OVC alongside a content-length one; " + + "must still report prefetch-beneficial", + prefetchDp.ssrd.hasAnyPrefetchBeneficialOVC + ) + val prefetchBytes = unparseToBytes(prefetchDp, infoset) + assertArrayEquals(singlePassBytes, prefetchBytes) + } + + // A schema with no OVC at all: hasAnyPrefetchBeneficialOVC must not + // throw, and simply reports false. + @Test def testSchemaWithNoOVCAtAllReportsNotBeneficial(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertFalse(dp.ssrd.hasAnyPrefetchBeneficialOVC) + } + + // A schema where every OVC is resolvable-without-writing must report + // true: the common case prefetch already handles. + @Test def testSchemaWithOnlyResolvableOVCReportsBeneficial(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertTrue(dp.ssrd.hasAnyPrefetchBeneficialOVC) + } + + // hasAnyPrefetchBeneficialOVC is baked in at compile time; confirms + // the fallback still applies after a save/reload round trip. + @Test def testHasAnyPrefetchBeneficialOVCSurvivesSaveReload(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + val dp = Compiler() + .withTunable("useBuildWritePrefetch", "true") + .compileNode(sch) + .onPath("/") + .asInstanceOf[DataProcessor] + assertFalse(dp.ssrd.hasAnyPrefetchBeneficialOVC) + + val os = new ByteArrayOutputStream() + dp.save(java.nio.channels.Channels.newChannel(os)) + val reloadedDp = Compiler() + .reload(new java.io.ByteArrayInputStream(os.toByteArray)) + .asInstanceOf[DataProcessor] + assertFalse(reloadedDp.ssrd.hasAnyPrefetchBeneficialOVC) + + val infoset = xyz + assertArrayEquals(unparseToBytes(dp, infoset), unparseToBytes(reloadedDp, infoset)) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/TestSuspension.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/TestSuspension.scala new file mode 100644 index 0000000000..894190d815 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/TestSuspension.scala @@ -0,0 +1,149 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.Maybe.One +import org.apache.daffodil.lib.util.MaybeULong +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.processors.unparsers.BuildState +import org.apache.daffodil.runtime1.processors.unparsers.UState +import org.apache.daffodil.runtime1.processors.unparsers.UnparseSharedContext +import org.apache.daffodil.runtime1.processors.unparsers.UnparseSharedContextTestFixture + +import org.junit.Assert.* +import org.junit.Test + +/** + * A minimal Suspension whose doTask counts invocations and either + * blocks or succeeds, under the test's control. + */ +private class SpySuspension(val rd: RuntimeData) extends Suspension { + + override val isReadOnly = true + + // Avoids prepareToSuspend's splitDOS branch (requires an actual + // currentInfosetNode to be positioned); irrelevant here since this test + // drives the suspension directly, not through an actual unparse traversal. + override protected def maybeKnownLengthInBits(ustate: UState): MaybeULong = MaybeULong(0L) + + var doTaskCallCount: Int = 0 + private var shouldSucceed = false + + def markShouldSucceedNextTime(): Unit = shouldSucceed = true + + protected def doTask(ustate: UState): Unit = { + doTaskCallCount += 1 + if (shouldSucceed) { + setDone() + } else { + block(this, this, 0, this) + } + } +} + +/** + * Unit tests for suspendWithoutAttempting: confirms it tracks/suspends + * like a genuine blocked run(), but without ever calling doTask. + */ +class TestSuspension { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + /** A trivial schema/BuildState, just to get an actual UState to drive the + * spy against; suspensionWaitYoung/Old default to schema-compiled tunables, + * or pass explicit values (e.g. 1) for a deterministic sweep cadence. + */ + private def newBuildState( + suspensionWaitYoung: Int = -1, + suspensionWaitOld: Int = -1 + ): (DataProcessor, BuildState, UnparseSharedContext) = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + , + elementFormDefault = "unqualified" + ) + val infoset = 123 + + val dp = TestUtils.compileForUnparse(sch) + + val inputter = TestUtils.newInitializedInputter(infoset, dp) + + // Doubling only applies to the tunable-derived fallback, matching + // production's build/write-interleaving compensation (see + // UnparseSharedContextTestFixture); an explicit override (e.g. 1, to + // force every call to sweep unconditionally) is used as given. + val waitYoung = if (suspensionWaitYoung > 0) { + suspensionWaitYoung + } else { + dp.tunables.unparseSuspensionWaitYoung * 2 + } + val waitOld = if (suspensionWaitOld > 0) { + suspensionWaitOld + } else { + dp.tunables.unparseSuspensionWaitOld * 2 + } + val sharedCtx = UnparseSharedContextTestFixture.build( + inputter.documentElement, + dp, + prefetchLimit = 100 + )(waitYoung, waitOld) + val buildState = new BuildState(inputter, sharedCtx, Nil, false) + buildState.initializeVariables() + // cloneForSuspension reads currentInfosetNodeStack.top, normally + // pushed at the start of a real element unparse; since this test + // drives the suspension directly rather than through a real unparse + // traversal, push a minimal entry manually (the other two stacks + // auto-push already). + buildState.currentInfosetNodeStack.push(One(sharedCtx.rootDoc)) + // Likewise normally set at the start of a real element unparse; + // needed by cloneForSuspension's own processor-copying call. + buildState.setProcessor(dp.ssrd.unparser) + (dp, buildState, sharedCtx) + } + + @Test def testSuspendWithoutAttemptingNeverCallsDoTask(): Unit = { + val (dp, buildState, sharedCtx) = newBuildState() + val spy = new SpySuspension(dp.ssrd.elementRuntimeData) + + spy.suspendWithoutAttempting(buildState) + + assertEquals(0, spy.doTaskCallCount) + assertFalse(spy.isDone) + } + + @Test def testSuspendWithoutAttemptingIsTrackedAndLaterResolves(): Unit = { + val (dp, buildState, sharedCtx) = newBuildState() + val spy = new SpySuspension(dp.ssrd.elementRuntimeData) + + spy.suspendWithoutAttempting(buildState) + assertTrue(sharedCtx.suspensionTracker.suspensions.exists(_ eq spy)) + + // Now make a genuine attempt succeed, and confirm the tracker's unfiltered + // drain resolves it; suspendWithoutAttempting only skips the wasted + // first attempt, not the suspension's normal later resolution. + spy.markShouldSucceedNextTime() + sharedCtx.suspensionTracker.evalSuspensionsUnthrottled() + + assertEquals(1, spy.doTaskCallCount) + assertTrue(spy.isDone) + sharedCtx.suspensionTracker.requireFinal() + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BoundedPrefetchTest.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BoundedPrefetchTest.scala new file mode 100644 index 0000000000..234dfd80d4 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BoundedPrefetchTest.scala @@ -0,0 +1,124 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase + +import org.junit.Assert.* +import org.junit.Test + +/** + * Proves build pauses to let write catch up: with a small prefetchLimit, + * the lead counter stays bounded mid-recursion, not equal to the full count. + */ +class BoundedPrefetchTest { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testBuildLeadStaysBoundedDuringRecursion(): Unit = { + val numItems = 40 + val prefetchLimit = 3L + + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + , + elementFormDefault = "unqualified" + ) + + val items = (0 until numItems).map(i => {s"i$i"}) + val infoset = {items} + val expectedBytes = (0 until numItems).map(i => s"i$i").mkString(",") + + val dp = TestUtils.compileForUnparse(sch, Map("releaseUnneededInfoset" -> "false")) + + val buildInputter = TestUtils.newInitializedInputter(infoset, dp) + + val sharedCtx = + UnparseSharedContextTestFixture.build(buildInputter.documentElement, dp, prefetchLimit)() + + val walkerOut = new ByteArrayOutputStream() + val writeInputter = TestUtils.newInitializedInputter(infoset, dp) + val writeState = UState.createInitialUStateForSharedVariables( + walkerOut, + dp, + writeInputter, + sharedCtx.variableBox, + false + ) + writeState.setSharedContext(sharedCtx) + writeState.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + val rootUnparser = dp.ssrd.unparser.asInstanceOf[ElementUnparserBase] + UnparseSharedContextTestFixture.wireCoroutines( + sharedCtx, + buildInputter.documentElement, + rootUnparser, + writeState + ) + + val buildState = new BuildState(buildInputter, sharedCtx, Nil, false) + buildState.initializeVariables() + + dp.ssrd.unparser.unparse1(buildState) + + // See class doc above for why this proves interleaving; numItems + 1 + // (row + all items) is what currentLead would equal here if + // resumeWrite never fired mid-recursion. + assertTrue( + s"expected lead close to prefetchLimit=$prefetchLimit after build, but was ${sharedCtx.currentLead} " + + s"(numItems=$numItems); build did not actually pause for write", + sharedCtx.currentLead <= prefetchLimit + 1 + ) + assertTrue( + "expected build to have gotten ahead of write by at least one node", + sharedCtx.currentLead > 0 + ) + + // Drain the rest and confirm the output is byte-for-byte correct + // despite having been produced across many separate resumeWrite calls + // rather than a single one-shot write pass. + val finalSignal = sharedCtx.resumeWrite(BuildFinished) + finalSignal match { + case WriteDone(Some(t)) => throw t + case WriteDone(None) => // continue below + case other => fail(s"unexpected final signal: $other") + } + writeState.evalSuspensions(isFinal = true) + writeState.getDataOutputStream.setFinished(writeState) + + assertEquals(expectedBytes, new String(walkerOut.toByteArray, "ascii")) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildStateNVIVariableScopeTest.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildStateNVIVariableScopeTest.scala new file mode 100644 index 0000000000..7596b4c944 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildStateNVIVariableScopeTest.scala @@ -0,0 +1,127 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import scala.jdk.CollectionConverters.* + +import org.apache.daffodil.core.compiler.Compiler +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.externalvars.ExternalVariablesLoader +import org.apache.daffodil.runtime1.processors.DataProcessor +import org.apache.daffodil.runtime1.processors.SuspensionTracker +import org.apache.daffodil.runtime1.processors.VariableBox + +import org.junit.Assert.* +import org.junit.Test + +/** + * Regression guard: after build pops a local NVI scope, a read must + * still see the top-level value, not a stale leftover from that scope. + */ +class BuildStateNVIVariableScopeTest { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testVariableReadAfterNVIScopeCloses(): Unit = { + val numRecords = 25 + val sch = SchemaUtils.dfdlTestSchema( + , { + + + }, + + + + + + + + + + + + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + val records = (0 until numRecords).map(i => {f"$i%02d"}) + val infoset = {records} + + val extVars = + ExternalVariablesLoader.mapToBindings(Map(s"{$example}runningVar" -> "-1").asJava) + + val pf = Compiler().withTunable("releaseUnneededInfoset", "false").compileNode(sch) + assertFalse(pf.getDiagnostics.toString, pf.isError) + val dp = pf.onPath("/").asInstanceOf[DataProcessor].withExternalVariables(extVars) + assertFalse(dp.getDiagnostics.toString, dp.isError) + + val inputter = TestUtils.newInitializedInputter(infoset, dp) + + val sharedCtx = new UnparseSharedContext( + inputter.documentElement, + new VariableBox(dp.variableMap.copy()), + new SuspensionTracker( + dp.tunables.unparseSuspensionWaitYoung, + dp.tunables.unparseSuspensionWaitOld + ), + dp, + dp.tunables, + prefetchLimit = 100 + ) + val buildState = new BuildState(inputter, sharedCtx, Nil, false) + buildState.initializeVariables() + + // No write coroutine is constructed/wired into sharedCtx, so build + // never pauses to interleave with write; the whole point (see class + // doc): isolates the NVI stale-read bug from the separate, unrelated + // interleaved-setVariable bug. + val rootUnparser = dp.ssrd.unparser + rootUnparser.unparse1(buildState) + + val rootNode = inputter.documentElement.child(0).asComplex + assertEquals(2, rootNode.numChildren) // the `record` array term, and `summary` + assertEquals(numRecords, rootNode.child(0).asArray.numChildren) + val summaryNode = rootNode.child(1).asSimple + assertEquals("summary", summaryNode.erd.name) + assertEquals( + "expected summary to read runningVar's own top-level (externally-bound) " + + "value, not a stale leftover instance from whichever record build pushed last", + -1, + summaryNode.dataValue.getInt + ) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildStateSuspensionTest.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildStateSuspensionTest.scala new file mode 100644 index 0000000000..522e305051 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildStateSuspensionTest.scala @@ -0,0 +1,134 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils + +import org.junit.Assert.* +import org.junit.Test + +/** + * Validates SuspensionCapableUState end to end: a build-side OVC forward + * reference must create a tracked Suspension, not throw ClassCastException. + */ +class BuildStateSuspensionTest { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testBuildSideOVCSuspensionIsTrackedNotClassCastException(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + + // "computed" is OVC. It's optional in the infoset input, since its value + // is computed, not provided (see OVCStartEndStrategy). + val infoset = 005 + + // releaseUnneededInfoset disabled; otherwise the actual unparse + // machinery frees "computed" from the tree as soon as it's done with + // it, and we want to inspect its final value afterward. + val dp = TestUtils.compileForUnparse(sch, Map("releaseUnneededInfoset" -> "false")) + + val inputter = TestUtils.newInitializedInputter(infoset, dp) + + val sharedCtx = + UnparseSharedContextTestFixture.build(inputter.documentElement, dp, prefetchLimit = 100)() + val buildState = new BuildState(inputter, sharedCtx, Nil, false) + buildState.initializeVariables() + + // Drives all the way through: "computed"'s OVC forward-references + // "actual", which doesn't exist yet when build reaches "computed" + // (document order). This should suspend, not throw; certainly not + // ClassCastException against BuildState. + dp.ssrd.unparser.unparse1(buildState) + + // Not asserted here: suspensions.nonEmpty would be flaky, since + // "actual"'s own unparseEnd calls evalSuspensions and retries/resolves + // "computed"'s suspension before unparse1 returns; expected, reflecting + // suspend/resume working within a single pass. The real check is below: + val row = + sharedCtx.rootDoc.child(0).asInstanceOf[org.apache.daffodil.runtime1.infoset.DIComplex] + val computed = row.child(0).asInstanceOf[org.apache.daffodil.runtime1.infoset.DISimple] + assertEquals(6, computed.dataValue.getInt) + + sharedCtx.suspensionTracker.evalSuspensions() + // requireFinal() would throw SuspensionDeadlockException if anything + // were still stuck. This confirms nothing was left hanging. + sharedCtx.suspensionTracker.requireFinal() + } + + // Build's first attempt at a valueLength OVC must skip doTask + // (suspendWithoutAttempting, not run()): isBlocked stays false. + @Test def testBuildSideValueLengthOVCNeverCallsDoTask(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + , + elementFormDefault = "unqualified" + ) + + val infoset = hello + + val dp = TestUtils.compileForUnparse(sch, Map("releaseUnneededInfoset" -> "false")) + + val inputter = TestUtils.newInitializedInputter(infoset, dp) + + val sharedCtx = + UnparseSharedContextTestFixture.build(inputter.documentElement, dp, prefetchLimit = 100)() + val buildState = new BuildState(inputter, sharedCtx, Nil, false) + buildState.initializeVariables() + + dp.ssrd.unparser.unparse1(buildState) + + assertEquals(1, sharedCtx.suspensionTracker.suspensions.length) + val suspension = sharedCtx.suspensionTracker.suspensions.head + assertFalse( + "doTask should never have been called for a suspendWithoutAttempting-created suspension", + suspension.isBlocked + ) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildStateTest.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildStateTest.scala new file mode 100644 index 0000000000..7ce751bef8 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildStateTest.scala @@ -0,0 +1,80 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils + +import org.junit.Assert.* +import org.junit.Test + +/** + * Validates BuildState in isolation: confirms it surfaces the same + * event sequence a real UStateMain would, for a separated schema. + */ +class BuildStateTest { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testBuildStateSurfacesCorrectEventSequence(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + val infoset = + Alice30Boston + + val dp = TestUtils.compileForUnparse(sch, Map("releaseUnneededInfoset" -> "false")) + + val inputter = TestUtils.newInitializedInputter(infoset, dp) + // initialize() pushes the root TRD as its last step - the same + // setup a real unparse's invariant check relies on. + + val sharedCtx = + UnparseSharedContextTestFixture.build(inputter.documentElement, dp, prefetchLimit = 100)() + val buildState = new BuildState(inputter, sharedCtx, Nil, false) + buildState.initializeVariables() + + // Drives through the ACTUAL Unparser recursion, not hand-driven + // advance() calls, since next-element resolution depends on the same + // TRD push/pop dance ElementUnparserBase.unparse performs. This schema's + // separator makes BuildState skip delimiter writing (gated on isBuildOnly). + val rootUnparser = dp.ssrd.unparser + rootUnparser.unparse1(buildState) + + assertEquals(4L, sharedCtx.currentLead) // row, name, age, city + + val rootNode = inputter.documentElement.child(0).asComplex + assertEquals(3, rootNode.numChildren) + assertEquals("name", rootNode.child(0).erd.name) + assertEquals("age", rootNode.child(1).erd.name) + assertEquals("city", rootNode.child(2).erd.name) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildWriteArrayChoiceTest.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildWriteArrayChoiceTest.scala new file mode 100644 index 0000000000..75aac93292 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/BuildWriteArrayChoiceTest.scala @@ -0,0 +1,203 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.runtime1.infoset.DIDocument +import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase + +import org.junit.Assert.* +import org.junit.Test + +/** + * Validates write-side dispatch for array presence-checking and choice + * resolution, via a ground-truth tree and a standalone BuildState run. + */ +class BuildWriteArrayChoiceTest { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testArrayAndChoiceWriteContentMatchesGroundTruth(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + // Three "item" occurrences exercise the array loop; typeB (not typeA) + // exercises actual choice resolution. lengthKind=delimited (not + // explicit) avoids double-firing CaptureStartOfContentLengthUnparser's + // non-idempotent marker, since this tree is reused from a completed unparse. + val infoset = +
H
abcX
+ + val dp = TestUtils.compileForUnparse(sch, Map("releaseUnneededInfoset" -> "false")) + + // Ground truth: the actual, unmodified single-pass unparse. Also gives + // us the intact built tree afterward (releaseUnneededInfoset disabled). + val groundTruthOut = new ByteArrayOutputStream() + val groundTruthResult = dp.unparse(new ScalaXMLInfosetInputter(infoset), groundTruthOut) + assertFalse(groundTruthResult.getDiagnostics.toString, groundTruthResult.isError) + val groundTruthBytes = groundTruthOut.toByteArray + assertEquals("H,a,b,c,X", new String(groundTruthBytes, "ascii")) + + val ustate = groundTruthResult.resultState.asInstanceOf[UStateMain] + val builtTree: DIDocument = ustate.documentElement + + // Redirect ustate's DataOutputStream to a fresh sink, then write the + // ALREADY-BUILT tree directly instead of an actual unparse pass. + val walkerOut = new ByteArrayOutputStream() + val writeInputter = TestUtils.newInitializedInputter(infoset, dp) + val writeState = UState.createInitialUState(walkerOut, dp, writeInputter, false) + writeState.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + // writeContent requires an actual UnparseSharedContext; the tree is + // already fully built, so a minimal one suffices; observeBuildSignal + // marks it as such, since this test never runs an actual build coroutine. + val sharedCtx = UnparseSharedContextTestFixture.build(builtTree, dp, prefetchLimit = 1000)() + sharedCtx.observeBuildSignal(BuildFinished) + writeState.setSharedContext(sharedCtx) + + UnparseSharedContextTestFixture.primeLeadCounter(sharedCtx, builtTree.child(0)) + + val rootUnparser = dp.ssrd.unparser.asInstanceOf[ElementUnparserBase] + val rootNode = sharedCtx.awaitChild(builtTree, 0) + rootUnparser.writeContent(rootNode, writeState) + // The separator (default separatorSuppressionPolicy "anyEmpty") is + // written speculatively via a suspension deciding, once known, if the + // region it precedes is zero-length; this drains that chain before the + // DOS is finalized, mirroring DataProcessor's finishWriteSide. + writeState.evalSuspensions(isFinal = true) + writeState.getDataOutputStream.setFinished(writeState) + + assertArrayEquals(groundTruthBytes, walkerOut.toByteArray) + } + + // Standalone-build regression: drives BuildState directly, then feeds + // its tree into write's writeContent (end-to-end build-then-write). + @Test def testStandaloneBuildStateNavigatesArrayChoiceSeparator(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + val infoset = +
H
abcX
+ + val dp = TestUtils.compileForUnparse(sch, Map("releaseUnneededInfoset" -> "false")) + + // Build phase: standalone BuildState drives the actual Unparser recursion, + // navigating past the sequence's separator and through the + // array/choice content, purely to build the tree. + val buildInputter = TestUtils.newInitializedInputter(infoset, dp) + + val sharedCtx = + UnparseSharedContextTestFixture.build( + buildInputter.documentElement, + dp, + prefetchLimit = 100 + )() + val buildState = new BuildState(buildInputter, sharedCtx, Nil, false) + buildState.initializeVariables() + + dp.ssrd.unparser.unparse1(buildState) + + // row, header, item x3, typeB = 6 elements total. + assertEquals(6L, sharedCtx.currentLead) + + val rootNode = buildInputter.documentElement.child(0).asComplex + assertEquals(3, rootNode.numChildren) + assertEquals("header", rootNode.child(0).erd.name) + assertEquals("item", rootNode.child(1).erd.name) + assertEquals(3, rootNode.child(1).asInstanceOf[DIArray].numChildren) + assertEquals("typeB", rootNode.child(2).erd.name) + + // Build was driven directly (no coroutine handoff), so sharedCtx can't + // know build is done; tell it so write's awaitChild call takes the + // post-BuildFinished (suspension-retry) path instead of resuming a + // coroutine that was never set up. + sharedCtx.observeBuildSignal(BuildFinished) + + // Write phase: write the tree BuildState just constructed, confirming + // it's a usable, fully-built tree, not just a navigation exercise. + val walkerOut = new ByteArrayOutputStream() + val writeInputter = TestUtils.newInitializedInputter(infoset, dp) + val writeState = UState.createInitialUState(walkerOut, dp, writeInputter, false) + writeState.setSharedContext(sharedCtx) + writeState.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + val rootUnparser = dp.ssrd.unparser.asInstanceOf[ElementUnparserBase] + val rootWriteNode = sharedCtx.awaitChild(buildInputter.documentElement, 0) + rootUnparser.writeContent(rootWriteNode, writeState) + // The separator (default separatorSuppressionPolicy "anyEmpty") is + // written speculatively via a suspension deciding, once known, if the + // region it precedes is zero-length; this drains that chain before the + // DOS is finalized, mirroring DataProcessor's finishWriteSide. + writeState.evalSuspensions(isFinal = true) + writeState.getDataOutputStream.setFinished(writeState) + + assertEquals("H,a,b,c,X", new String(walkerOut.toByteArray, "ascii")) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/LeadCounterTest.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/LeadCounterTest.scala new file mode 100644 index 0000000000..dcafeb779b --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/LeadCounterTest.scala @@ -0,0 +1,110 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase + +import org.junit.Assert.* +import org.junit.Test + +/** + * Validates the shared build/write lead counter end to end: build + * increments and write decrements against the same shared instance. + */ +class LeadCounterTest { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testLeadCounterIncrementsOnBuildAndDecrementsOnWrite(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + // Fixed dfdl:length is safe here because this tree comes from + // BuildState, which never runs content-writing (including + // CaptureStartOfContentLengthUnparser). + val infoset = + Alice30Boston + + val dp = TestUtils.compileForUnparse(sch, Map("releaseUnneededInfoset" -> "false")) + + // Build phase: drive BuildState through the actual recursion, incrementing + // the shared lead counter via the actual unparseBegin hookup, as in + // BuildStateTest and BuildStateSuspensionTest. + val buildInputter = TestUtils.newInitializedInputter(infoset, dp) + + val sharedCtx = + UnparseSharedContextTestFixture.build( + buildInputter.documentElement, + dp, + prefetchLimit = 100 + )() + val buildState = new BuildState(buildInputter, sharedCtx, Nil, false) + buildState.initializeVariables() + + assertEquals(0L, sharedCtx.currentLead) + dp.ssrd.unparser.unparse1(buildState) + + // row itself, name, age, city = 4 elements total, each incrementing + // once via unparseBegin's actual hookup. + assertEquals(4L, sharedCtx.currentLead) + + // Build was driven directly (no coroutine handoff), so sharedCtx can't + // know build is done; tell it so write's awaitChild calls take the + // post-BuildFinished (suspension-retry) path instead of resuming a + // coroutine that was never set up. + sharedCtx.observeBuildSignal(BuildFinished) + + // Write phase: write the SAME already-built tree + // (buildInputter.documentElement) against the SAME sharedCtx, + // decrementing the lead counter as it goes. + val walkerOut = new ByteArrayOutputStream() + val writeInputter = TestUtils.newInitializedInputter(infoset, dp) + val writeState = UState.createInitialUState(walkerOut, dp, writeInputter, false) + writeState.setSharedContext(sharedCtx) + writeState.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + val rootUnparser = dp.ssrd.unparser.asInstanceOf[ElementUnparserBase] + val rootNode = sharedCtx.awaitChild(buildInputter.documentElement, 0) + rootUnparser.writeContent(rootNode, writeState) + writeState.getDataOutputStream.setFinished(writeState) + + assertEquals("Alice30Boston", new String(walkerOut.toByteArray, "ascii")) + + // Write decremented once per element too, so the counter is back to 0: + // build and write agree on how many nodes exist, coordinated through + // the shared UnparseSharedContext. + assertEquals(0L, sharedCtx.currentLead) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/PendingSuspensionTripLimitTest.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/PendingSuspensionTripLimitTest.scala new file mode 100644 index 0000000000..d2f2c7c799 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/PendingSuspensionTripLimitTest.scala @@ -0,0 +1,142 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import java.io.ByteArrayOutputStream + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase + +import org.junit.Assert.* +import org.junit.Test + +/** + * Proves unparsePendingSuspensionTripLimit is an independent bound from + * prefetchLimit: pendingCount alone must still trip an early write catch-up. + */ +class PendingSuspensionTripLimitTest { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + private def testSchema = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + @Test def testPendingCountTripsEarlyWriteCatchUp(): Unit = { + val numRecords = 40 + val pendingSuspensionTripLimit = 3L + val prefetchLimit = 10000L // unreachably large: isolates pendingSuspensionTripLimit + + val sch = testSchema + val records = (0 until numRecords).map(i => {s"value$i"}) + val infoset = {records} + + val singlePassDp = TestUtils.compileForUnparse(sch) + val singlePassOut = new ByteArrayOutputStream() + val singlePassRes = + singlePassDp.unparse(new ScalaXMLInfosetInputter(infoset), singlePassOut) + assertFalse(singlePassRes.getDiagnostics.toString, singlePassRes.isError) + val expectedBytes = singlePassOut.toByteArray + + val dp = TestUtils.compileForUnparse( + sch, + Map( + "releaseUnneededInfoset" -> "false", + "unparsePendingSuspensionTripLimit" -> pendingSuspensionTripLimit.toString + ) + ) + + val buildInputter = TestUtils.newInitializedInputter(infoset, dp) + + val sharedCtx = + UnparseSharedContextTestFixture.build(buildInputter.documentElement, dp, prefetchLimit)() + + val walkerOut = new ByteArrayOutputStream() + val writeInputter = TestUtils.newInitializedInputter(infoset, dp) + val writeState = UState.createInitialUStateForSharedVariables( + walkerOut, + dp, + writeInputter, + sharedCtx.variableBox, + false + ) + writeState.setSharedContext(sharedCtx) + writeState.getDataOutputStream.setPriorBitOrder(dp.ssrd.elementRuntimeData.defaultBitOrder) + + val rootUnparser = dp.ssrd.unparser.asInstanceOf[ElementUnparserBase] + UnparseSharedContextTestFixture.wireCoroutines( + sharedCtx, + buildInputter.documentElement, + rootUnparser, + writeState + ) + + val buildState = new BuildState(buildInputter, sharedCtx, Nil, false) + buildState.initializeVariables() + + dp.ssrd.unparser.unparse1(buildState) + + // numRecords*3 + 1 (root, record, len, data per record) is what + // currentLead would equal if pendingSuspensionTripLimit weren't wired in. + assertTrue( + s"expected lead bounded well below the full tree via pendingSuspensionTripLimit=" + + s"$pendingSuspensionTripLimit tripping an early write catch-up (prefetchLimit=$prefetchLimit " + + s"alone never would), but lead was ${sharedCtx.currentLead} (numRecords=$numRecords)", + sharedCtx.currentLead < numRecords + ) + assertTrue( + "expected build to have gotten ahead of write by at least one node", + sharedCtx.currentLead > 0 + ) + + val finalSignal = sharedCtx.resumeWrite(BuildFinished) + finalSignal match { + case WriteDone(Some(t)) => throw t + case WriteDone(None) => // continue below + case other => fail(s"unexpected final signal: $other") + } + writeState.evalSuspensions(isFinal = true) + writeState.getDataOutputStream.setFinished(writeState) + + assertArrayEquals(expectedBytes, walkerOut.toByteArray) + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContextTestFixture.scala b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContextTestFixture.scala new file mode 100644 index 0000000000..cf2410821f --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/runtime1/processors/unparsers/UnparseSharedContextTestFixture.scala @@ -0,0 +1,99 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.runtime1.processors.unparsers + +import org.apache.daffodil.runtime1.infoset.DIArray +import org.apache.daffodil.runtime1.infoset.DIComplex +import org.apache.daffodil.runtime1.infoset.DIDocument +import org.apache.daffodil.runtime1.infoset.DINode +import org.apache.daffodil.runtime1.processors.DataProcessor +import org.apache.daffodil.runtime1.processors.SuspensionTracker +import org.apache.daffodil.runtime1.processors.VariableBox +import org.apache.daffodil.unparsers.runtime1.ElementUnparserBase + +/** + * Shared BuildState construction for tests, mirroring production's `* 2` + * doubling of tunable-derived suspension-wait thresholds (default only). + */ +object UnparseSharedContextTestFixture { + def build(documentElement: DIDocument, dp: DataProcessor, prefetchLimit: Long)( + suspensionWaitYoung: Int = dp.tunables.unparseSuspensionWaitYoung * 2, + suspensionWaitOld: Int = dp.tunables.unparseSuspensionWaitOld * 2 + ): UnparseSharedContext = { + new UnparseSharedContext( + documentElement, + new VariableBox(dp.variableMap.copy()), + new SuspensionTracker(suspensionWaitYoung, suspensionWaitOld), + dp, + dp.tunables, + prefetchLimit + ) + } + + /** Wires a BuildCoroutine/WriteCoroutine pair onto sharedCtx, mirroring + * production's WriteCoroutine, so mid-recursion resumeWrite calls do + * something; caller must still perform the final handoff afterward. + */ + def wireCoroutines( + sharedCtx: UnparseSharedContext, + documentElement: DIDocument, + rootUnparser: ElementUnparserBase, + writeState: UState + ): Unit = { + val buildCoroutine = new BuildCoroutine() + val writeCoroutine = new WriteCoroutine({ (wc, firstSignal) => + try { + sharedCtx.observeBuildSignal(firstSignal) + try { + val rootNode = sharedCtx.awaitChild(documentElement, 0) + rootUnparser.writeContent(rootNode, writeState) + } catch { + case _: AwaitChildStalledException => + } + wc.resumeFinal(buildCoroutine, WriteDone(None)) + } catch { + case t: Throwable => + wc.resumeFinal(buildCoroutine, WriteDone(Some(t))) + } + }) + sharedCtx.setCoroutines(buildCoroutine, writeCoroutine) + } + + /** + * Pre-increments the lead counter once per element in `node`'s + * subtree, for a tree built outside BuildState (writeContent's + * decrementLead call requires the counter already be symmetric). + */ + def primeLeadCounter(sharedCtx: UnparseSharedContext, node: DINode): Unit = node match { + case complex: DIComplex => + sharedCtx.incrementLead() + var i = 0 + while (i < complex.numChildren) { + primeLeadCounter(sharedCtx, complex.child(i)) + i += 1 + } + case array: DIArray => + var i = 0 + while (i < array.numChildren) { + primeLeadCounter(sharedCtx, array.child(i)) + i += 1 + } + case _ => + sharedCtx.incrementLead() + } +} diff --git a/daffodil-core/src/test/scala/org/apache/daffodil/unparsers/runtime1/WriteContentWalkerTest.scala b/daffodil-core/src/test/scala/org/apache/daffodil/unparsers/runtime1/WriteContentWalkerTest.scala new file mode 100644 index 0000000000..95df07ab74 --- /dev/null +++ b/daffodil-core/src/test/scala/org/apache/daffodil/unparsers/runtime1/WriteContentWalkerTest.scala @@ -0,0 +1,142 @@ +/* + * Licensed to the Apache Software Foundation (ASF) under one or more + * contributor license agreements. See the NOTICE file distributed with + * this work for additional information regarding copyright ownership. + * The ASF licenses this file to You under the Apache License, Version 2.0 + * (the "License"); you may not use this file except in compliance with + * the License. You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +package org.apache.daffodil.unparsers.runtime1 + +import java.io.ByteArrayOutputStream + +import org.apache.daffodil.core.util.TestUtils +import org.apache.daffodil.io.DirectOrBufferedDataOutputStream +import org.apache.daffodil.lib.util.SchemaUtils +import org.apache.daffodil.lib.xml.XMLUtils +import org.apache.daffodil.runtime1.infoset.DIDocument +import org.apache.daffodil.runtime1.infoset.ScalaXMLInfosetInputter +import org.apache.daffodil.runtime1.processors.SuspensionTracker +import org.apache.daffodil.runtime1.processors.VariableBox +import org.apache.daffodil.runtime1.processors.unparsers.BuildFinished +import org.apache.daffodil.runtime1.processors.unparsers.UStateMain +import org.apache.daffodil.runtime1.processors.unparsers.UnparseSharedContext +import org.apache.daffodil.runtime1.processors.unparsers.UnparseSharedContextTestFixture + +import org.junit.Assert.* +import org.junit.Test + +/** + * Validates write's side of build/write-prefetch: writing an + * already-built tree matches a real single-pass unparse, byte for byte. + */ +class WriteContentWalkerTest { + + val example = XMLUtils.EXAMPLE_NAMESPACE + + @Test def testWriteWalkerMatchesActualUnparse(): Unit = { + val sch = SchemaUtils.dfdlTestSchema( + , + , + + + + + + + + + , + elementFormDefault = "unqualified" + ) + + val infoset = + Alice30Boston + + val dp = TestUtils.compileForUnparse(sch) + + // Ground truth: the actual, unmodified single-pass unparse, with default + // (production) tunables, including releaseUnneededInfoset, which frees + // nodes as unparse goes. + val groundTruthOut = new ByteArrayOutputStream() + val groundTruthResult = dp.unparse(new ScalaXMLInfosetInputter(infoset), groundTruthOut) + assertFalse(groundTruthResult.getDiagnostics.toString, groundTruthResult.isError) + val groundTruthBytes = groundTruthOut.toByteArray + + // Second unparse, releaseUnneededInfoset disabled, purely to obtain an + // intact DIDocument tree to walk independently (the default frees the + // tree by completion). The actual build/write-prefetch path never frees + // during build; only write does, after emitting a node's content. + val dpNoFree = + TestUtils.compileForUnparse(sch, Map("releaseUnneededInfoset" -> "false")) + val discardOut = new ByteArrayOutputStream() + val treeResult = dpNoFree.unparse(new ScalaXMLInfosetInputter(infoset), discardOut) + assertFalse(treeResult.getDiagnostics.toString, treeResult.isError) + val ustate = treeResult.resultState.asInstanceOf[UStateMain] + val builtTree: DIDocument = ustate.documentElement + + // Redirect ustate's DataOutputStream to a fresh sink, then write the + // ALREADY-BUILT tree directly instead of an actual unparse pass. + val walkerOut = new ByteArrayOutputStream() + val freshDos = DirectOrBufferedDataOutputStream( + walkerOut, + null, + false, + ustate.tunable.outputStreamChunkSizeInBytes, + ustate.tunable.maxByteArrayOutputStreamBufferSizeInBytes, + ustate.tunable.tempFilePath + ) + // Mirrors DataProcessor.unparse's setup. A freshly constructed DOS has + // no prior bit order until told one. + freshDos.setPriorBitOrder(dpNoFree.ssrd.elementRuntimeData.defaultBitOrder) + ustate.setDataOutputStream(freshDos) + + // writeContent requires an actual UnparseSharedContext (awaitChild, + // decrementLead, etc. are methods on it); the tree is already fully + // built, so a minimal one suffices; observeBuildSignal marks it as + // such, since this test never runs an actual build coroutine. + val sharedCtx = + new UnparseSharedContext( + builtTree, + new VariableBox(dpNoFree.variableMap.copy()), + new SuspensionTracker( + dpNoFree.tunables.unparseSuspensionWaitYoung, + dpNoFree.tunables.unparseSuspensionWaitOld + ), + dpNoFree, + dpNoFree.tunables, + prefetchLimit = 1000 + ) + sharedCtx.observeBuildSignal(BuildFinished) + ustate.setSharedContext(sharedCtx) + + UnparseSharedContextTestFixture.primeLeadCounter(sharedCtx, builtTree.child(0)) + + val rootUnparser = dpNoFree.ssrd.unparser.asInstanceOf[ElementUnparserBase] + val rootNode = sharedCtx.awaitChild(builtTree, 0) + rootUnparser.writeContent(rootNode, ustate) + // Mirrors unparseViaBuildThenWrite's ordering: this schema's default + // separatorSuppressionPolicy ("anyEmpty") writes separators via actual + // suspensions that must be drained before the DOS is finalized, or the + // speculative regions never resolve. + ustate.evalSuspensions(isFinal = true) + // Mirrors DataProcessor's writeState.getDataOutputStream.setFinished. + // Separator suspensions split the stream into a chain of DOS objects, + // so by the time writing is done, ustate's current DataOutputStream is + // no longer necessarily freshDos. + ustate.getDataOutputStream.setFinished(ustate) + + val walkerBytes = walkerOut.toByteArray + + assertArrayEquals(groundTruthBytes, walkerBytes) + } +} diff --git a/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd b/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd index db5568846a..5783a7530f 100644 --- a/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd +++ b/daffodil-propgen/src/main/resources/org/apache/daffodil/xsd/dafext.xsd @@ -534,6 +534,51 @@
+ + + + If true, unparsing uses a build/write-prefetch path instead of the + default single-pass unparse: a build pass runs ahead of a write pass + over the same infoset tree, resolving forward references against the + tree directly instead of via Suspensions where possible. How far + build may run ahead of write is bounded by + unparsePrefetchWindowNodes. Experimental; defaults to false + (today's single-pass behavior, unchanged). + + + + + + + Only consulted when useBuildWritePrefetch is true. The maximum + number of infoset nodes build may construct ahead of write before + build pauses to let write catch up. Bounds peak memory use to + roughly this many resident nodes rather than the whole infoset. + + + + + + + Only consulted when useBuildWritePrefetch is true. A second, + independent limit on how far build may run ahead of write, + alongside unparsePrefetchWindowNodes: the maximum number of + pending suspensions (forward references that cannot resolve + until write itself writes the referenced bytes, e.g. + dfdl:valueLength/dfdl:contentLength) build may accumulate + before it pauses to let write catch up, regardless of how + large unparsePrefetchWindowNodes is set. Without this, + unparsePrefetchWindowNodes alone does not bound this + backlog, since a node's lead-counter contribution is + released once write's structural pass reaches it, whether or + not that node's own suspension has resolved yet, so a large + window can otherwise let many resolved-but-still- + pending nodes accumulate unboundedly before anything + forces write to actually catch up to what they're + waiting on. + + + @@ -644,7 +689,11 @@ evaluating suspensions that are likely to fail. The unparseSuspensionWaitYoung and unparseSuspensionWaitOld values determine how many elements are unparsed before evaluating - young and old suspensions, respectively. + young and old suspensions, respectively. When useBuildWritePrefetch + is true, both values are used internally at double what is + configured here, since build and write independently advance this + same counter, so it would otherwise tick twice as fast as a single + traversal. diff --git a/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala b/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala index d6632308b8..3f04975a5b 100644 --- a/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala +++ b/daffodil-propgen/src/main/scala/org/apache/daffodil/propGen/TunableGenerator.scala @@ -79,7 +79,8 @@ class TunableGenerator(schemaRootConfig: scala.xml.Node, schemaRootExt: scala.xm | if (configOpt.isDefined) { | val loader = new DaffodilXMLLoader() | val node = loader.load(URISchemaSource(Paths.get(configPath).toFile, configOpt.get), Some(XMLUtils.dafextURI)) - | tunablesMap(node) + | val optTunablesNode = (node \ "tunables").headOption + | optTunablesNode.map(tunablesMap(_)).getOrElse(Map.empty) | } else { | Map.empty | }