# from rwkvstic.load import RWKV
# from rwkvstic.agnostic.backends import TORCH
# import torch
# import os

# os.environ["RWKV_JIT_ON"] = '1'
# os.environ["RWKV_CUDA_ON"] = '1' # if '1' then use CUDA kernel for seq mode (much faster)
# # this is the dtype used for trivial operations, such as vector->vector operations and is the dtype that will determine the accuracy of the model
# runtimedtype = torch.float32 # torch.float64, torch.bfloat16

# # this is the dtype used for matrix-vector operations, and is the dtype that will determine the performance and memory usage of the model
# dtype = torch.bfloat16 # torch.float32, torch.float64, torch.bfloat16

# useGPU = False # False

# model = RWKV("RWKV-4-Pile-3B-Instruct-test2-20230209.pth", mode=TORCH, useGPU=useGPU, runtimedtype=runtimedtype, dtype=dtype)

# model.loadContext(newctx=f"Q: How many hours are there in a day?\n\nA:")
# output = model.forward(number=100)["output"]

# print(output)