# from_stream.py -*- Python -*-
#
# Copyright (C) 2025-2026 Advanced Micro Devices, Inc.
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#
"""``from_stream`` example — Iron API design with ``@iron.jit``.

A 24-element int32 vector is forwarded shim -> memtile -> core -> memtile
-> shim.  The memtile->core ObjectFifo's
``from_stream=TensorAccessPattern.full((8, 3)).T`` reshapes the linear stream into the equivalent of a (3, 8) ->
(8, 3) transpose by the time the core sees it, so the host output is
the transposed view of the input ``arange(24)``.
"""

import argparse

import aie.iron as iron
import numpy as np
from aie.helpers.taplib import TensorAccessPattern
from aie.iron import In, ObjectFifo, Out, Program, Runtime, Worker
from aie.iron.controlflow import range_
from aie.utils.hostruntime.argparse import (
    add_compile_args,
    device_from_args,
)
from aie.utils.hostruntime.cli import run_design_cli
from aie.utils.verify import assert_pass

data_ty = np.ndarray[(24,), np.dtype[np.int32]]


@iron.jit
def from_stream(a_in: In, c_out: Out):
    of_in0 = ObjectFifo(data_ty, name="in0")
    of_in1 = of_in0.cons().forward(
        name="in1",
        obj_type=data_ty,
        # Write the incoming (3, 8) stream into the object as its (8, 3)
        # transpose: sizes [3, 8], strides [1, 3].
        from_stream=TensorAccessPattern.full((8, 3)).T,
    )

    of_out1 = ObjectFifo(data_ty, name="out1")
    of_out0 = of_out1.cons().forward(name="out0", obj_type=data_ty)

    def core_fn(of_in, of_out):
        elem_in = of_in.acquire(1)
        elem_out = of_out.acquire(1)
        for i in range_(24):
            elem_out[i] = elem_in[i]
        of_in.release(1)
        of_out.release(1)

    my_worker = Worker(core_fn, [of_in1.cons(), of_out1.prod()])

    def sequence(a, c, in_h, out_h):
        in_h.fill(a)
        out_h.drain(c, wait=True)

    rt = Runtime(sequence, [data_ty, data_ty, of_in0.prod(), of_out0.cons()])

    return Program(iron.get_current_device(), rt, workers=[my_worker]).resolve_program()


def _run_and_verify(opts):
    a_in = iron.arange(24, dtype=np.int32, device="npu")
    c_out = iron.zeros(24, dtype=np.int32, device="npu")
    from_stream(a_in, c_out)
    assert_pass(
        c_out.numpy(),
        np.arange(24, dtype=np.int32).reshape(3, 8).T.reshape(-1),
        fail_msg="from_stream mismatch",
    )


def main():
    p = argparse.ArgumentParser(prog="from_stream example")
    add_compile_args(p, with_emit_mlir=True)
    opts = p.parse_args()
    run_design_cli(
        from_stream,
        opts,
        compile_kwargs={},
        run_and_verify=_run_and_verify,
        device=lambda o: device_from_args(o, n_cols=1),
    )


if __name__ == "__main__":
    main()
