Lab 01 - Program Sources
Course: Computer Architectures
Results: Lab 01 - gem5 and ASE Studio
Source snapshots from ASE Studio, 2026-10-03. These listings are the versions used for the documented measurements. The calculate_pi snapshot includes the Linux exit compatibility fix; its arithmetic is unchanged.
program_1
# Data section
.section .data
# Place here your program data. In this example
# two vector of floats, a vector of ints anda single int are defined
T0: .word 0x0BADC0DE
# The text section contains the instructions that the CPU runs.
.section .text
# Make _start visible as the point where the program begins.
.globl _start
_start:
Main:
# Initialize Fibonacci variables
li x1, 0 # x1 = a = first Fibonacci number (0)
li x2, 1 # x2 = b = second Fibonacci number (1)
li x3, 21 # x3 = count = number of terms to generate
li x4, 0 # x4 = i = loop counter
# Loop to generate and print remaining 20 numbers
addi x4, x4, 1 # i = 1 (start from second iteration)
fib_loop:
beq x4, x3, End # if i == count, exit loop
# Calculate next Fibonacci number
add x5, x1, x2 # x5 = next = a + b
# Update variables for next iteration
mv x1, x2 # a = b (previous second becomes first)
mv x2, x5 # b = next (calculated next becomes second)
# Increment counter and continue loop
addi x4, x4, 1 # i++
j fib_loop # Jump back to loop start
# The End block stops the program and returns control to the simulator.
End:
li a0, 0
li a7, 93
ecallprogram_2
# Lab 01, Exercise 2
# Preserve v1 order; store each matching v1 element once, even if
# that value occurs multiple times in v2. Repeated v1 elements remain.
.section .data
v1: .byte 2, 6, -3, 11, 9, 18, -13, 16, 5, 1
v2: .byte 4, 2, -13, 3, 9, 9, 7, 16, 4, 7
v3: .space 10
v3_len: .byte 0
flag1: .byte 0
flag2: .byte 0
flag3: .byte 0
# The text section contains the instructions that the CPU runs.
.section .text
# Make _start visible as the point where the program begins.
.globl _start
_start:
la x5, v1
la x6, v2
la x7, v3
li x8, 10 # Remaining v1 elements
li x9, 0 # Length of v3
outer_loop:
beq x8, x0, matching_done
lb x10, 0(x5) # Signed 8-bit v1 value
mv x11, x6
li x12, 10
inner_loop:
beq x12, x0, next_v1
lb x13, 0(x11)
beq x10, x13, match_found
addi x11, x11, 1
addi x12, x12, -1
j inner_loop
match_found:
sb x10, 0(x7)
addi x7, x7, 1
addi x9, x9, 1
# Stop scanning v2 after the first match.
next_v1:
addi x5, x5, 1
addi x8, x8, -1
j outer_loop
matching_done:
la x14, v3_len
sb x9, 0(x14)
la x15, flag1
la x16, flag2
la x17, flag3
sb x0, 0(x15)
sb x0, 0(x16)
sb x0, 0(x17)
bne x9, x0, nonempty
li x18, 1
sb x18, 0(x15) # Empty: flags = 1, 0, 0
j End
nonempty:
li x18, 1
sb x18, 0(x16) # Assume increasing until disproved
sb x18, 0(x17) # Assume decreasing until disproved
addi x19, x9, -1 # Number of adjacent pairs
beq x19, x0, End # Singleton: both order flags are 1
la x20, v3
lb x21, 0(x20)
check_order:
lb x22, 1(x20)
blt x21, x22, increasing_ok
sb x0, 0(x16) # Equal or falling: not strictly increasing
increasing_ok:
blt x22, x21, decreasing_ok
sb x0, 0(x17) # Equal or rising: not strictly decreasing
decreasing_ok:
mv x21, x22
addi x20, x20, 1
addi x19, x19, -1
bne x19, x0, check_order
# The End block stops the program and returns control to the simulator.
End:
li a0, 0 # exit(0)
li a7, 93
ecallcalculate_pi
.data
pi_result: .float 0.0
sixteen: .float 16.0
four: .float 4.0
two: .float 2.0
one: .float 1.0
eight: .float 8.0
# The text section contains the instructions that the CPU runs.
.text
# Make _start visible as the point where the program begins.
.globl _start
_start:
# BBP formula: π = Σ[k=0 to ∞] (1/16^k) * (4/(8k+1) - 2/(8k+4) - 1/(8k+5) - 1/(8k+6))
flw f0, one, t0 # f0 = 1.0
flw f1, sixteen, t0 # f1 = 16.0
flw f2, four, t0 # f2 = 4.0
flw f3, two, t0 # f3 = 2.0
flw f4, eight, t0 # f4 = 8.0
fmv.s f10, f0 # f10 = running sum = 0
fmv.s f11, f0 # f11 = 16^(-k), start with 1
li t1, 50 # Number of terms (50 gives good precision)
li t2, 0 # k counter
bbp_loop:
beq t2, t1, bbp_finish
# Convert k to float
fcvt.s.w f12, t2 # f12 = k (float)
# Calculate 8k
fmul.s f13, f4, f12 # f13 = 8k
# Calculate terms: 4/(8k+1), 2/(8k+4), 1/(8k+5), 1/(8k+6)
fadd.s f14, f13, f0 # f14 = 8k+1
fdiv.s f15, f2, f14 # f15 = 4/(8k+1)
fadd.s f14, f13, f2 # f14 = 8k+4
fdiv.s f16, f3, f14 # f16 = 2/(8k+4)
fsub.s f15, f15, f16 # f15 = 4/(8k+1) - 2/(8k+4)
fadd.s f14, f13, f2 # f14 = 8k+5
fadd.s f14, f14, f0 # f14 = 8k+5
fdiv.s f16, f0, f14 # f16 = 1/(8k+5)
fsub.s f15, f15, f16 # Subtract 1/(8k+5)
fadd.s f14, f13, f2 # f14 = 8k+6
fadd.s f14, f14, f3 # f14 = 8k+6
fdiv.s f16, f0, f14 # f16 = 1/(8k+6)
fsub.s f15, f15, f16 # Final bracket term
# Multiply by (1/16^k)
fmul.s f15, f15, f11 # Multiply by 16^(-k)
# Add to sum
fadd.s f10, f10, f15
# Update 16^(-k) for next iteration
fdiv.s f11, f11, f1 # f11 = f11/16
addi t2, t2, 1 # k++
j bbp_loop
bbp_finish:
# Store result
fsw f10, pi_result, t0
# Linux exit for gem5
# The End block stops the program and returns control to the simulator.
End:
li a0, 0
li a7, 93
ecallinsertion_sort
.data
array: .word 0x9342DC99,0x4F696186,0x1C85D9EE,0x7A0B6710,0x37CA911D,0x5EF87A2B,0xB7E01DD9,0x47EF58E8
.word 0x20CB53C9,0x79A74EAB,0x06F8E483,0x6D26753D,0x126C0B47,0x70F313AF,0x035F1EF5,0x745232A4
.word 0xE8D095FD,0x46036BDD,0x9B5A43EA,0x4DE3F1D8,0x102EABBA,0x5279659D,0x49E29089,0x4496CDA9
.word 0x09B7D04A,0x6D594E20,0x55905DE5,0x4CE57C0D,0x78A1848B,0x4115A0AC,0x48B3BDA6,0x5051DAA6
.word 0xE11593C0,0x71C3730C,0x68370FC5,0x425A9FAE,0x85354426,0x6B265F84,0x9C713D01,0x4E935A84
.word 0x8E5170D7,0x77311058,0x3803A780,0x5B133F18,0x525C362C,0x49A52D37,0x49D8A123,0x4A0C150C
.word 0xA41FB066,0x7962EC77,0xF3417D04,0x5D3A087A,0x31DC3B2E,0x7076F960,0x05D02FDD,0x404EC3D1
.word 0x189A7A8B,0x5484F578,0x19037E03,0x65EA86F8,0xAE35B27A,0x4367E6F2,0x9394DB43,0x63C1CF86
.word 0x269E583C,0x59421109,0x9C8598EF,0x6B9F1B52,0x129AF1BD,0x4C877DCC,0xF56D884F,0x58401EDB
.word 0xE3F8BFCF,0x754C5475,0x11111111,0x11111111,0xF3FAE203,0x786213BF,0x23F8D4B5,0x53F6C772
.word 0x4815DFBF,0x5304A0C7,0xB7E84DED,0x701BFCF2,0x1BA476AD,0x72C3DEDE,0x1C0A436C,0x557C0537
.word 0xBAEBBBB3,0x741CECCD,0xE5C72202,0x577156E9,0xF6E59822,0x641D1FEF,0x45E6AFC6,0x623B6D2C
.word 0x37A754F0,0x6976994C,0x6963A020,0x4CE48C6E,0xCF3CD3AC,0x4EDDBCD1,0xC1AE08E4,0x706AAA8F
.word 0x8E4ACB59,0x674DE62D,0x83AF7749,0x791423B5,0x08F70D0A,0x45890096,0x3430F238,0x55159D9A
.word 0xE3048518,0x70BD250B,0x8C603831,0x6D1B6012,0x0E29CEE8,0x5397AB7F,0x374A9A97,0x58EF0102
.word 0x94D1E2D1,0x625D9DBD,0x7165FDF6,0x5E843943,0x37353266,0x4F621F3A,0x1149F170,0x426B3ACC
.word 0x7FA3F476,0x59D789FA,0xD30D6D4B,0x4C4353E0,0xA02F0B1C,0x492F120F,0xA97CFF59,0x720DFD78
.word 0x14551D39,0x5BC2140E,0x9D4656B9,0x68718C03,0xFFFFFFFF,0x7FFFFFFF,0xCBC9A739,0x48F63330
.word 0xFD5F8C20,0x6E47955A,0xD10F9D2A,0x44972B6A,0xCA1151A1,0x46578121,0x7672B320,0x46281A1E
.word 0x3E05BD98,0x4094CC80,0x7812A363,0x5FF5B63C,0x7F7612C5,0x6AF41E21,0xB1E208AC,0x4B7B4452
.word 0xFA5E72E4,0x750F8A67,0x9B5E8AD1,0x51C8ECF2,0x3D81B486,0x58055035,0xF3970ABF,0x668CD4C5
.word 0xA16715AD,0x480BEE00,0x9EE02467,0x4888D5AC,0x62669040,0x77C3DDBA,0x7F706867,0x48D55CDF
.word 0x1FE6E445,0x72067034,0x191C2CC9,0x6CAE4383,0xD0270344,0x4F9E28BA,0x8A8A3979,0x46DAD432
.word 0x98729716,0x55B7AEB5,0xC5FF97C5,0x76D0F139,0x9C2DC380,0x4B876EB3,0xD91E6FDF,0x781ADC2A
.word 0xF4AA0625,0x53BDEAF8,0xB9A73772,0x624D7EA5,0xA787850D,0x75A02137,0xC33A32E6,0x4259BDE1
len: .word 50
# The text section contains the instructions that the CPU runs.
.text
# Make _start visible as the point where the program begins.
.global _start
_start:
# Set i = 4 (byte offset for second element when using 4-byte words)
li x5,4 # x5 = i
la x30,len # load address of len
lw x6,0(x30) # x6 = len (count of elements)
slli x6,x6,2 # x6 = len*4 (bytes)
for_loop:
slt x7,x5,x6 # x7 = (i < len*4)
beqz x7,out # if !(i < len*4) goto out
mv x28,x5 # x28 = j = i
la x31,array
add x31,x31,x5 # address of array[i]
lw x29,0(x31) # x29 = B = a[i]
loop:
slt x7,x0,x28 # x7 = (j > 0)
beqz x7,over # if j==0 goto over
addi x30,x28,-4 # x30 = j-4 (j-1 element offset)
la x31,array
add x31,x31,x30 # address of array[j-1]
lw x31,0(x31) # x31 = a[j-1]
slt x7,x31,x29 # x7 = (a[j-1] < B)
beqz x7,over # if !(a[j-1] < B) goto over
# a[j] = a[j-1]
la x7,array
add x7,x7,x28 # address of array[j]
sw x31,0(x7)
mv x28,x30 # j = j-4
j loop
over:
la x31,array
add x31,x31,x28 # address of array[j]
sw x29,0(x31) # a[j] = B
addi x5,x5,4 # i += 4
j for_loop
out:
# Exit via ecall (Linux convention): x17=93 (exit), x10=0
li x10,0
li x17,93
ecall