Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
44 changes: 40 additions & 4 deletions src/main/java/com/amazon/ion/bytecode/BytecodeGenerator.kt
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@ package com.amazon.ion.bytecode
import com.amazon.ion.Decimal
import com.amazon.ion.Timestamp
import com.amazon.ion.bytecode.util.AppendableConstantPoolView
import com.amazon.ion.bytecode.util.ByteSlice
import com.amazon.ion.bytecode.util.BytecodeBuffer
import java.math.BigInteger

Expand Down Expand Up @@ -53,8 +54,7 @@ internal interface BytecodeGenerator {
* no more top-level values may be filled into the destination.
*
* If there is incomplete data, and no values can be returned, it should
* throw IncompleteDataException (if streaming source) or IonException (if
* fully-buffered source).
* throw IonException (if fully-buffered source).
*/
fun refill(
/** The BytecodeBuffer that is to be filled with the bytecode */
Expand All @@ -78,12 +78,48 @@ internal interface BytecodeGenerator {
symTab: Array<String?>,
)

/**
* Reads an Ion int value from this [BytecodeGenerator]'s source data.
* Arguments to this function should come from an [INT_REF] instruction.
*/
fun readBigIntegerReference(position: Int, length: Int): BigInteger

/**
* Reads an Ion decimal value from this [BytecodeGenerator]'s source data.
* Arguments to this function should come from a [DECIMAL_REF] instruction.
*/
fun readDecimalReference(position: Int, length: Int): Decimal

/**
* Reads an Ion timestamp value from this [BytecodeGenerator]'s source data.
* Arguments to this function should come from a [SHORT_TIMESTAMP_REF] instruction.
*/
fun readShortTimestampReference(position: Int, opcode: Int): Timestamp

/**
* Reads an Ion timestamp value from this [BytecodeGenerator]'s source data.
* Arguments to this function should come from a [TIMESTAMP_REF] instruction.
*/
fun readTimestampReference(position: Int, length: Int): Timestamp

/**
* Reads any UTF-8 text from this [BytecodeGenerator]'s source data.
* Arguments to this function should usually come from a [STRING_REF], [SYMBOL_REF], [ANNOTATION_REF], or
* [FIELD_NAME_REF] instruction.
*
* In the case of a text reader, this could hypothetically also be used to read text for a [META_COMMENT] instruction.
*/
fun readTextReference(position: Int, length: Int): String
fun readBytesReference(position: Int, length: Int): ByteArray

/**
* Reads a [ByteSlice] from this [BytecodeGenerator]'s source data.
* Arguments to this function should come from a [BLOB_REF] or [CLOB_REF] instruction.
*
* This function returns a [ByteSlice] rather than a [ByteArray] so that it is relatively cheap to call multiple
* times for reading a blob or clob in chunks rather than all at once.
* If [BytecodeGenerator] is ever exposed in the public API, we should revisit the return type of this function.
*/
fun readBytesReference(position: Int, length: Int): ByteSlice

/** The Ion Minor Version supported by this [BytecodeGenerator] */
fun ionMinorVersion(): Int
Expand All @@ -93,5 +129,5 @@ internal interface BytecodeGenerator {
* Implementations must return a BytecodeGenerator that supports the requested version and is positioned
* to continue compiling bytecode exactly where this BytecodeGenerator left off.
*/
fun getGeneratorForMinorVersion(minorVersion: Int): BytecodeGenerator = throw UnsupportedOperationException()
fun getGeneratorForMinorVersion(minorVersion: Int): BytecodeGenerator
}
14 changes: 11 additions & 3 deletions src/main/java/com/amazon/ion/bytecode/ir/Debugger.kt
Original file line number Diff line number Diff line change
Expand Up @@ -75,6 +75,11 @@ internal object Debugger {
var indent = ""
var i = start
while (i < end) {
if (bytecode[i] == 0) {
i++
continue
}

if (useNumbers) write(line(i))

val instruction = bytecode[i++]
Expand All @@ -100,8 +105,11 @@ internal object Debugger {
// Write the operation name, and any data carried in the instruction
write(indent)
write(instructionInfo.name)
write(" ")
write(instructionInfo.dataType.formatter(Instructions.getData(instruction)).toString())

if (instructionInfo.dataType != InstructionInfo.DataInfo.NO_DATA) {
write(" ")
write(instructionInfo.dataType.formatter(Instructions.getData(instruction)).toString())
}

// If we have symbol table or constant pool available, add in supplemental information
if (constantPool != null && instructionInfo.dataType == InstructionInfo.DataInfo.CP_INDEX) {
Expand Down Expand Up @@ -154,6 +162,7 @@ internal object Debugger {
InstructionInfo.DIRECTIVE_ADD_MACROS,
InstructionInfo.DIRECTIVE_USE,
InstructionInfo.DIRECTIVE_MODULE,
InstructionInfo.DIRECTIVE_IMPORT,
InstructionInfo.DIRECTIVE_ENCODING,
InstructionInfo.LIST_START,
InstructionInfo.SEXP_START,
Expand All @@ -162,7 +171,6 @@ internal object Debugger {
}

when (instructionInfo) {
InstructionInfo.REFILL,
InstructionInfo.END_OF_INPUT -> break
else -> continue
}
Expand Down
9 changes: 9 additions & 0 deletions src/main/java/com/amazon/ion/bytecode/ir/InstructionInfo.kt
Original file line number Diff line number Diff line change
Expand Up @@ -69,12 +69,21 @@ internal enum class InstructionInfo(
PLACEHOLDER_TAGLESS(Operation.OP_PLACEHOLDER_TAGLESS, DataInfo.OPCODE),
ARGUMENT_NONE(Operation.OP_ARGUMENT_NONE, DataInfo.NO_DATA),
IVM(Operation.OP_IVM, DataInfo.IVM),
/** Contents of this directive in bytecode should be strings or symbols. */
DIRECTIVE_SET_SYMBOLS(Operation.OP_DIRECTIVE_SET_SYMBOLS, DataInfo.NO_DATA),
/** Contents of this directive in bytecode should be strings or symbols. */
DIRECTIVE_ADD_SYMBOLS(Operation.OP_DIRECTIVE_ADD_SYMBOLS, DataInfo.NO_DATA),
/** Contents of this directive in bytecode should be s-expressions containing name-template pairs. */
DIRECTIVE_SET_MACROS(Operation.OP_DIRECTIVE_SET_MACROS, DataInfo.NO_DATA),
/** Contents of this directive in bytecode should be s-expressions containing name-template pairs. */
DIRECTIVE_ADD_MACROS(Operation.OP_DIRECTIVE_ADD_MACROS, DataInfo.NO_DATA),
/** Contents of this directive in bytecode should be triples of name (string) version (int), and maxId (null or int). */
DIRECTIVE_USE(Operation.OP_DIRECTIVE_USE, DataInfo.NO_DATA),
/** Contents of this directive in bytecode should follow the module definition grammar. */
DIRECTIVE_MODULE(Operation.OP_DIRECTIVE_MODULE, DataInfo.NO_DATA),
/** Contents of this directive in bytecode should be triples of bindingName (symbol), catalogName (string), and version (int). */
DIRECTIVE_IMPORT(Operation.OP_DIRECTIVE_IMPORT, DataInfo.NO_DATA),
/** Contents of this directive in bytecode should be symbols. */
DIRECTIVE_ENCODING(Operation.OP_DIRECTIVE_ENCODING, DataInfo.NO_DATA),
INVOKE(Operation.OP_INVOKE, DataInfo.MACRO_ID),
REFILL(Operation.OP_REFILL, DataInfo.NO_DATA),
Expand Down
8 changes: 8 additions & 0 deletions src/main/java/com/amazon/ion/bytecode/ir/Instructions.kt
Original file line number Diff line number Diff line change
Expand Up @@ -2,6 +2,8 @@
// SPDX-License-Identifier: Apache-2.0
package com.amazon.ion.bytecode.ir

import com.amazon.ion.bytecode.ir.Operation.OPERATION_KIND_OFFSET

/**
* Utility object for working with packed instruction formats and instruction constants.
*
Expand All @@ -22,6 +24,12 @@ internal object Instructions {
@JvmStatic
fun toOperation(instruction: Int) = instruction ushr OPERATION_OFFSET

/**
* Given an [OperationKind] value that represents an Ion type, returns the appropriate NULL variant operation.
*/
@JvmStatic
fun typedNullFromOperationKind(operationKind: Int): Int = (operationKind.shl(OPERATION_KIND_OFFSET) + Operation.NULL_VARIANT).shl(OPERATION_OFFSET)

/**
* Extracts the operand count bits from a packed instruction.
*
Expand Down
7 changes: 4 additions & 3 deletions src/main/java/com/amazon/ion/bytecode/ir/Operation.kt
Original file line number Diff line number Diff line change
Expand Up @@ -23,9 +23,9 @@ internal object Operation {
fun toOperationKind(operation: Int): Int = operation ushr OPERATION_KIND_OFFSET

/** Variant identifier used for null value operations */
private const val NULL_VARIANT = 7
const val NULL_VARIANT = 7
/** Bit offset for extracting instruction kind from operation codes */
private const val OPERATION_KIND_OFFSET = 3
const val OPERATION_KIND_OFFSET = 3

// Operation code constants
// Each constant combines an instruction kind with a variant identifier
Expand Down Expand Up @@ -104,7 +104,8 @@ internal object Operation {
const val OP_DIRECTIVE_ADD_MACROS = (OperationKind.DIRECTIVE shl OPERATION_KIND_OFFSET) + 3
const val OP_DIRECTIVE_USE = (OperationKind.DIRECTIVE shl OPERATION_KIND_OFFSET) + 4
const val OP_DIRECTIVE_MODULE = (OperationKind.DIRECTIVE shl OPERATION_KIND_OFFSET) + 5
const val OP_DIRECTIVE_ENCODING = (OperationKind.DIRECTIVE shl OPERATION_KIND_OFFSET) + 6
const val OP_DIRECTIVE_IMPORT = (OperationKind.DIRECTIVE shl OPERATION_KIND_OFFSET) + 6
const val OP_DIRECTIVE_ENCODING = (OperationKind.DIRECTIVE shl OPERATION_KIND_OFFSET) + 7

const val OP_INVOKE = (OperationKind.INVOKE_TEMPLATE shl OPERATION_KIND_OFFSET)

Expand Down
66 changes: 64 additions & 2 deletions src/main/java/com/amazon/ion/bytecode/ir/instruction_reference.md
Original file line number Diff line number Diff line change
Expand Up @@ -96,23 +96,85 @@ This instruction indicates that there is a String value, and the text of the str
| DIRECTIVE_ADD_MACROS | `0x8B` | `10001` | `011` | `00` | - | - | Must have END_CONTAINER instruction to delimit end of directive |
| DIRECTIVE_USE | `0x8C` | `10001` | `100` | `00` | - | - | Must have END_CONTAINER instruction to delimit end of directive |
| DIRECTIVE_MODULE | `0x8D` | `10001` | `101` | `00` | - | - | Must have END_CONTAINER instruction to delimit end of directive |
| DIRECTIVE_ENCODING | `0x8E` | `10001` | `110` | `00` | - | - | Must have END_CONTAINER instruction to delimit end of directive |
| DIRECTIVE_IMPORT | `0x8E` | `10001` | `110` | `00` | - | - | Must have END_CONTAINER instruction to delimit end of directive |
| DIRECTIVE_ENCODING | `0x8F` | `10001` | `111` | `00` | - | - | Must have END_CONTAINER instruction to delimit end of directive |
| PLACEHOLDER_TAGGED | `0x90` | `10010` | `000` | `11` | bytecode_length (u22) | - | Optional tagged macro parameter.[^0x91] |
| PLACEHOLDER_TAGLESS | `0x91` | `10010` | `001` | `00` | opcode (u8) | - | Tagless macro parameter |
| ARGUMENT_NONE | `0x98` | `10011` | `000` | `00` | - | - | Represents an argument that is absent. |
| INVOKE | `0xA0` | `10100` | `000` | `00` | macro_id (u22) | - | Only used when bypassing macro evaluation.[^0xA0] |
| REFILL | `0xA8` | `10101` | `000` | `00` | - | - | End of bytecode, reader must request refill of bytecode buffer |
| END_TEMPLATE | `0xB0` | `10110` | `000` | `00` | - | - | End of template, return to caller.[^0xB0] |
| END_OF_INPUT | `0xB1` | `10110` | `001` | `00` | - | - | Only applicable for fixed-sized input streams. |
| END_OF_INPUT | `0xB1` | `10110` | `001` | `00` | - | - | No more values; insufficient source data in the generator. |
| END_CONTAINER | `0xB2` | `10110` | `010` | `00` | - | - | Delimits the end of a directive, list, sexp, or struct. |
| META_OFFSET | `0xB8` | `10111` | `000` | `01` | | offset (u32) | To support input >4GB, pack 22 high-order bits in "data". |
| META_ROWCOL | `0xB9` | `10111` | `001` | `01` | column (u22) | row (u32) | 0-based, row/col position |
| META_COMMENT | `0xBA` | `10111` | `010` | `01` | ref_length (u22) | offset (u32) | Hypothetical.[^0xBA] |

Possible TODOs:
* Could we consolidate `END_TEMPLATE` and `END_OF_INPUT` into a single `RETURN` instruction?
* OR rename `END_OF_INPUT` to `NEEDS_DATA`?
* Consider renaming `REFILL` to `GENERATE_MORE_BYTECODE`
* When we implement the BytecodeIonReader, we might find that we need to separate `END_CONTAINER` from the other end
types into its own operation kind since they have different conditions before you can resume them. You can theoretically
call `next()` after an `END_OF_INPUT`, and more data might have arrived. However, you can never get more values after
a `END_CONTAINER` without first calling `stepOut()`.

[^0x91]: There is no delimited end marker for the default value. If there is no default value, then bytecode_length=0.
[^0xA0]: Must be followed by value instructions for all parameters in macro signature. Absent arguments are represented
with `ARGUMENT_NONE`.
[^0xB0]: Only used in the macro table—for marking the end of a template body
[^0xBA]: Potential inclusion to make it possible to expose comments from Ion text, or it could reference arbitrary data
from inside a lengthy NOP. Comments that are longer than u22 max value could be encoded using multiple comment
instructions. The span should include the comment-delimiting characters.

## Directive Content

### `SET_SYMBOLS`, `ADD_SYMBOLS`

Child values are any `STRING_*` or `SYMBOL_*` instructions, each one representing a single symbol definition.
Unknown symbol text is denoted using the `SYMBOL_SID 0` instruction.

### `SET_MACROS`, `ADD_MACROS`

Child values are pairs of:
* a name (`SYMBOL_*`/`NULL_NULL`)
* any instructions representing a single value.

Example:
```text
DIRECTIVE_ADD_MACROS
SYMBOL_SID $23
FLOAT_F32
3.1415
SYMBOL_SID $24
INT_I16 42
Comment on lines +146 to +150

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Should these be contained in s-expressions?

Copy link
Copy Markdown
Contributor Author

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

The S-Expressions are for human readability more than anything else. If you think that we should keep the S-Expressions in the bytecode, we can, but it's not strictly necessary to make things work.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Got it, let's keep the bytecode as lean as possible.

END_CONTAINER
```

### `USE`

Content is triples of `name` (`STRING_*`), `version` (`INT_*`), `maxId` (`INT_*`/`NULL_NULL`).
For Ion 1.1 bytecode generators, `maxId` is always `NULL_NULL`.

### `MODULE`

Content is bytecode that is a 1:1 equivalent of the values and expressions used to define the module.

Example:
```text
DIRECTIVE_MODULE
SEXP_START
SYMBOL_CP 4 // "macros"
SEXP_START
SYMBOL_CP 5 // "macro"
SYMBOL_CP 6 // "foo"
// ...
```

### `IMPORT`

Content is triples of `bindingName` (`SYMBOL_*`), `catalogName` (`STRING_*`), and version (`INT_*`).

### `ENCODING`

Content is `SYMBOL_*` instructions, each one representing the name of a module.
Original file line number Diff line number Diff line change
Expand Up @@ -10,4 +10,6 @@ interface AppendableConstantPoolView {
fun add(value: Any?): Int
/** Retrieves a value from the constant pool. */
fun get(i: Int): Any?

val size: Int
}
9 changes: 7 additions & 2 deletions src/main/java/com/amazon/ion/bytecode/util/ByteSlice.kt
Original file line number Diff line number Diff line change
Expand Up @@ -9,8 +9,13 @@ import edu.umd.cs.findbugs.annotations.SuppressFBWarnings
*
* Positions are relative to `bytes`, not the underlying data stream.
*
* This is not intended to be exposed publicly, but it might end up needing to be exposed in order to expose a
* template-building API. If it does get exposed, we need to revisit the location of this class.
* TODO: Consider adding helper functions that allow
* * copying some or all of the bytes into an existing byte array
* * creating a new byte array that is a copy of just a subsection of this [ByteSlice].
*
* This is not intended to be exposed publicly, but it might end up needing to be exposed in order to expose the
* [BytecodeGenerator][com.amazon.ion.bytecode.BytecodeGenerator] for use by e.g. an object mapper. If it does get
* exposed, we need to revisit the location of this class and make [bytes], [startInclusive], and [endExclusive] private.
*/
@SuppressFBWarnings(
value = ["EI_EXPOSE_REP"],
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -177,7 +177,7 @@ internal class BytecodeBuffer private constructor(
* @param value the new bytecode instruction value
* @throws IndexOutOfBoundsException if the index is out of range (index < 0 || index >= size())
*/
fun set(index: Int, value: Int) {
operator fun set(index: Int, value: Int) {
if (index < 0 || index >= numberOfValues) {
throw java.lang.IndexOutOfBoundsException()
}
Expand Down
2 changes: 1 addition & 1 deletion src/main/java/com/amazon/ion/bytecode/util/ConstantPool.kt
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,7 @@ internal class ConstantPool private constructor(

constructor(initialCapacity: Int) : this(data = arrayOfNulls(initialCapacity), numberOfValues = 0)

val size: Int
override val size: Int
get() = numberOfValues

fun isEmpty(): Boolean = numberOfValues == 0
Expand Down
50 changes: 50 additions & 0 deletions src/test/java/com/amazon/ion/TextToBinaryUtils.kt
Original file line number Diff line number Diff line change
@@ -0,0 +1,50 @@
// Copyright Amazon.com, Inc. or its affiliates. All Rights Reserved.
// SPDX-License-Identifier: Apache-2.0
package com.amazon.ion

import java.util.Locale
import java.util.stream.Collectors
import java.util.stream.Stream

object TextToBinaryUtils {
/**
* Converts a string of octets in the given radix to a byte array. Octets must be separated by a space.
* @param octetString the string of space-separated octets.
* @param radix the radix of the octets in the string.
* @return a new byte array.
*/
@JvmStatic
private fun octetStringToByteArray(octetString: String, radix: Int): ByteArray {
if (octetString.isEmpty()) return ByteArray(0)
val bytesAsStrings = octetString.split(" ".toRegex()).dropLastWhile { it.isEmpty() }.toTypedArray()
val bytesAsBytes = ByteArray(bytesAsStrings.size)
for (i in bytesAsBytes.indices) {
bytesAsBytes[i] = (bytesAsStrings[i].toInt(radix) and 0xFF).toByte()
}
return bytesAsBytes
}

/**
* Converts a string of hex octets, such as "BE EF", to a byte array.
*/
@JvmStatic
fun String.hexStringToByteArray(): ByteArray {
return octetStringToByteArray(this, 16)
}

/**
* @param hexBytes a string containing white-space delimited pairs of hex digits representing the expected output.
* The string may contain multiple lines. Anything after a `|` character on a line is ignored, so
* you can use `|` to add comments.
*/
@JvmStatic
fun String.cleanCommentedHexBytes(): String {
return Stream.of(*this.split("\n".toRegex()).dropLastWhile { it.isEmpty() }.toTypedArray())
.map { it.replace("\\|.*$".toRegex(), "").trim() }
.filter { it.trim().isNotEmpty() }
.collect(Collectors.joining(" "))
.replace("\\s+".toRegex(), " ")
.uppercase(Locale.getDefault())
.trim()
}
}
Loading
Loading