Compare commits

98 Commits
Author SHA1 Message Date
glm94 c95d739e16 Updated the parser to handle treating variables as expressions. Updated the symbols table to handle computing the values of variables. 2023-09-07 21:29:58 -05:00
glm94 21c2bf046e Wrote the very first part of the tree walking code, more just to make sure the logic is sound. 2023-09-01 10:04:21 -05:00
glm94 b810d7803c Updated the parser to create an expression tree for symbols declared via .db directives. Scanner now see some math operators as 'punctuation'. 2023-08-31 23:27:33 -05:00
glm94 9e178ca3da Build a quick draft of how the symbol value should be computed. 2023-08-25 01:27:22 -05:00
glm94 82b2136a69 Started to work on setting up Symbol value resolution which will be based on an abstract syntax tree that can handle undefined values if unknown symbols are used in an expression. 2023-08-25 00:21:19 -05:00
glm94 4bb90da6df Make the compiler options a bit more strict about C compliance (in other words I just learned about these flags and they seem to be ideal for this project). 2023-08-25 00:15:18 -05:00
glm94 4af2527806 Updated the structure of symbols and tokens to hopefully help make computing symbol values easier. 2023-08-07 01:18:01 -05:00
glm94 65383cded2 Updated the way symbols are represented and how they are marked as resolved. 2023-07-09 22:04:55 -05:00
glm94 9c8fe17c8f Minor code cleanup and added an extra flag to the compiler. If a symbol is a label it will now hold a reference to the opcode and by extension the offset into the file it points to. 2023-07-06 19:39:03 -05:00
glm94 3031fa8b9b Code cleanup. 2023-06-29 20:53:39 -05:00
glm94 2918f2964d Fixed a bug where words weren't being written to 'memory' correctly. 2023-06-24 18:00:13 -05:00
glm94 980b0449fc Fixed a bug with the parse discarding identifier tokens when parsing a COPY instruction. Added the first batch of code to encode the final token array to a binary form. 2023-06-22 20:22:20 -05:00
glm94 6741dde7ce Hooked up the program counter in the parser to set the memory offset of symbols, well labels to be more acurrate. 2023-06-18 14:51:37 -05:00
glm94 c6e6e2c5dc Fixed a bug in the parser not setting an identifier when one is used with a CMP and fixed a bug in the example assembly program. 2023-06-18 12:11:30 -05:00
glm94 0eaa5527ea Updated the parser to be able to handle the changes made to the ISA. 2023-06-18 00:50:18 -05:00
glm94 403439def4 Updated the opcode listing and 'fixed' the code to at least compile, but obviously I'll need to rework how the new copy instructions are to be parsed. Fixed some errors in the documentation where I skipped some hex numbers and added a copy immediate op. 2023-06-17 21:15:08 -05:00
glm94 0fd6413d81 Updated the instruction set and mapped out the opcodes as well as wrote a sample program to see if the ISA was complete enough to build a trivial program. 2023-06-17 16:56:42 -05:00
glm94 a9e57132be Fixed issue #1 Scanner Bug where the start of comment would become part of an identifier or throw off number parsing. 2023-04-14 13:28:44 -05:00
glm94 aa3888b578 Fixed up the RemoveCurrentToken function which was causing all sorts of bugs, and added the ability to delcare a data byte, '.db', to have a variable's address as a value. This will be needed for building up an interrupt table at least with the version I have in mind. 2023-04-04 22:14:17 -05:00
glm94 c3d12da54a Added some ideas to the encoding scheme to be used by the processor. 2023-03-30 13:21:25 -05:00
glm94 288ba9f9cb Updated the docs to include new jump operands and setup the assembler to handle parsing all the jump instructions. 2023-03-29 22:30:12 -05:00
glm94 a9a50055a4 Updated the parser to validate most of the opcodes. NOP falls through via the default case since it takes no parameters but the jump instructions need some thought before proceeding. 2023-03-29 20:57:04 -05:00
glm94 2e888913be Added support for the STORE directive. 2023-03-29 12:57:07 -05:00
glm94 ee60d14570 Refactored the code and added expect functions to make the code flow a bit nicer. 2023-03-29 12:41:53 -05:00
glm94 fa12a04bb3 Moved some of the grunt work to their own functions to clean the code up a bit. 2023-03-28 22:14:31 -05:00
glm94 2d7aad1617 Fixed a bug with labels not advancing the parser after being processed resulting in a redefinition error. 2023-03-28 12:41:56 -05:00
glm94 ae03991fa5 Fixed some parser bugs processing the syntax for some LOAD instruction types. 2023-03-27 22:22:01 -05:00
glm94 95ed2143ea Updated the parser to handle processing a simple load byte instruction. Added the LODB opcode to the opcode enum listing and updated the Scanner to look for 'byte' and mark it as a directive. 2023-03-24 23:24:24 -05:00
glm94 f96e1c6380 Added grammar checking for DB declarations. 2023-03-08 21:38:33 -06:00
glm94 4f9f4202b2 Wrote the first batch of directive handling code, so far just handling creating variables. 2023-03-08 13:25:46 -06:00
glm94 4230708f15 Added a function to remove an item from the generic list, which is mainly used for the tokens. 2023-03-07 21:31:14 -06:00
glm94 5e9e1bdec4 Massive refactoring to simplify this whole setup. The Token List will be modified in place and reused in the Parser to normalize the list and generate a symbols table. Normalizing the token list will make sure the syntax is valid and the symbols table will have the relative offsets and length of symbol values. At least this is all the plan but one step at a time. 2023-03-06 21:51:36 -06:00
glm94 1b246e90bc Started rethinking how this machine should behave. Updated and refactoring some things with a few new ideas. 2023-02-23 22:22:07 -06:00
glm94 9a602824b6 Added all the instructions supported to a structure to help make validation of parameters easier. 2022-10-18 21:14:50 -05:00
glm94 6d6bf1cbaf Wrote the first step for the binary output. Doesn't handle instruction pointer symbols yet since those emit addresses that still need to be computed. 2022-10-10 18:36:38 +00:00
glm94 bca2ad67f1 Fixed a bug with single parameter instructions that land at the end of a file expecting a comma. 2022-10-08 22:02:51 -05:00
glm94 fd44e03b1b Updated the Parser to try and give 'AddressParameter's more meaning depending on the SymbolType. 2022-10-08 18:21:11 -05:00
glm94 48a77950be Updated the way string symbols are handled, allowing the programmer to terminate a string with any byte or none at all. 2022-10-08 16:35:05 -05:00
glm94 8b6edc39be Updated the Parser to handle the variable declaration directive. 2022-10-08 16:15:10 -05:00
glm94 c142d2c6d0 Cleaned up the Symbol struct and updated code accordingly. 2022-10-08 14:43:00 -05:00
glm94 4b08707e67 Cleaned up the list implementation so as to NOT copy items into a new buffer. 2022-10-07 20:46:55 +00:00
glm94 bdb4d2b241 Updated the way Instructions keep track of symbols. Only their name is important but we need the Symbols to be able to track which Instruction they point to (if they do that is, i.e. a label symbol). 2022-10-06 21:11:22 -05:00
glm94 d53833c25f Added some debug print out for the instruction list and a helper function to get a string for any Registers enum. 2022-10-06 21:05:57 +00:00
glm94 f38da9cc74 A bit of cleanup. 2022-10-04 21:05:15 +00:00
glm94 33309a2759 Updated the Instruction's structure and updated the Parser accordingly. 2022-10-04 20:37:47 +00:00
glm94 1f7989533b Fixed a bug in the SymbolsTable where it wouldn't update its size. 2022-10-03 20:11:04 -05:00
glm94 cd1258a724 Fixed various Parser bugs. 2022-10-03 19:58:44 -05:00
glm94 be91e3c112 Updated the parser to better handle parameters, sort of and including a string length assembly program to use as a test for the whole assembler. 2022-10-03 21:26:28 +00:00
glm94 706f480e31 The Scanner will now check for empty lines (is the previous token and the current token a new line?) and simply not emit a NewLine Token. 2022-10-03 19:40:00 +00:00
glm94 a949007c75 Updated the ISA so the program will correctly see things like 'load', 'inc' and 'dec'. 2022-10-03 16:56:39 +00:00
glm94 a0d5d62a34 Fixed a bug in the Scanner when parsing a hex number, still not super robust but it'll work. 2022-10-03 15:20:31 +00:00
glm94 f7cd87f13f Started reworking the Parser to simplify how instructions will be represented. It will act like a 'first pass' that will do grammar checks but not verify the parameters of the opcodes. 2022-10-02 22:54:18 -05:00
glm94 55f4c32764 Missed this for the Scanner fix. 2022-10-02 22:12:16 -05:00
glm94 097eca383b Fixed a bug with the Scanner adding a blank line via a single NewLine Token when a file starts with a block of comments. 2022-10-01 20:43:55 -05:00
glm94 984683bfc5 Minor adjustment to Figure 1.1 2022-09-29 22:57:12 -05:00
glm94 0f42ad2997 Updated the ISA and added links in the table of contents. 2022-09-29 22:11:19 -05:00
glm94 63438b551b Got the scanner running again. Seems to be picking up punctuation, labels, identifiers and strings as expected. 2022-09-29 21:32:08 -05:00
glm94 e94e486e18 More refinements to the loading data instructions. I am truly bad at this whole thing... 2022-09-29 20:42:37 +00:00
glm94 2b71ac05a0 Added a note about a proposal regarding assembly language design. 2022-09-27 17:46:50 +00:00
glm94 02cf72902e Updated the load instructions. 2022-09-26 22:13:48 -05:00
glm94 afa84076c8 Added a command to make generating an opcode table a one liner. Updated the ISA. 2022-09-23 19:36:27 +00:00
glm94 925f7703d6 Updated the ISA, hopefully I can get this all put together in a thought out way. 2022-09-22 20:49:20 +00:00
glm94 a82a4c23ad Updated the docs. 2022-09-21 21:44:38 -05:00
glm94 f5c82d1c36 Added a LaTex document to put in writing how the machine should behave. 2022-09-21 16:23:11 +00:00
glm94 0b58e0124f Some refactoring to update everything to use the new structures and some new considerations for how Symbols and Instructions shoudl be represented. 2022-09-15 22:14:38 -05:00
glm94 2ae60a607c Incomplete, but I want to make sure the ideas I have here don't get wiped. Sadly this commit won't compile. 2022-09-15 21:29:27 +00:00
glm94 6d17dffcdf Added a SymbolTable object (untested at the moment) and redefined the Symbol object. I think when a Symbol is made only the size of it should matter to the code that will assemble the final binary. 2022-09-08 20:56:02 +00:00
glm94 2518214585 Added a length attribute to the symbol. 2022-09-06 14:15:45 +00:00
glm94 7004fff659 This feels like a trainwreck but eh. Changed the way the opcodes are managed. Hopefully this is the right direction when I add support for multiple ASM files. 2022-09-01 18:46:59 +00:00
glm94 13f8ff194b Smoothbrain indeed... 2022-08-30 22:48:49 -05:00
glm94 dac01ffd0c Changed the TokenClass from opcode to nomic, a less smoothbrain name IMO and also frees up Opcode for a better use later on. 2022-08-30 22:06:50 -05:00
glm94 3c7056c5e6 The main function will now write out the output of the parser. 2022-08-29 21:23:02 +00:00
glm94 6c7b8d5356 Fixed a bug where CMP REG, REG wouldn't get encoded and a disassembler bug related to said CMP bug. 2022-08-29 19:29:34 +00:00
glm94 5f437b6869 Both the assembler and the disassembler should now support all the instructions I've layed out thus far. 2022-08-29 18:51:48 +00:00
glm94 e3b4293314 Fixed a bug in the parser where the size of the binary was too big. 2022-08-29 18:02:11 +00:00
glm94 ae1e73e00f Fixed a bug to correct the location of symbols in the heap. 2022-08-29 16:00:02 +00:00
glm94 7385a807a5 Created a basic binary format to help guide the disassembler. 2022-08-29 15:57:14 +00:00
glm94 0d3f8a460e Added more instructions for the parser to emit and refactored the code in doing so. 2022-08-28 15:37:56 -05:00
glm94 c2111e2b88 Added support for the ADD and SUB opcodes to the disassembler. 2022-08-28 13:08:18 -05:00
glm94 8e9d1aefe8 Fixed some disassembler issues. At least some instuctions are being disassembled correctly. 2022-08-26 21:43:03 -05:00
glm94 0785f21805 Added a disassembler which is still very buggy. Also fixed some bugs with the opcode mask getting function. 2022-08-26 21:41:02 +00:00
glm94 e4a191a17e Refined the encoding some more. 2022-08-26 19:00:37 +00:00
glm94 eb5e2c506e Got the start of encoding opcodes. 2022-08-26 18:27:41 +00:00
glm94 0d060248cb Forgot about the bloody IN instruction... 2022-08-25 21:01:35 -05:00
glm94 80bea6f1a8 I'm dumb, fixed the CMP opcode overlap.. 2022-08-25 20:57:33 -05:00
glm94 a96f485da6 I believe I have the bit patterns for my opcodes worked out. 2022-08-25 20:45:46 -05:00
glm94 88d550f344 Added unconditional jump opcode to my comment section. 2022-08-25 20:27:25 -05:00
glm94 0e53f9798f Added a distinction between constant numbers and what are effectively memory addresses (i.e. strings and labels). 2022-08-25 19:47:16 -05:00
glm94 03d5660677 Seems I fixed the memory offset bugs, so far anyway... 2022-08-25 19:54:47 +00:00
glm94 4f6d5f652e Made use of the Symbols Table, and started the proccess of determining the memory location of opcodes and variables. 2022-08-25 19:06:05 +00:00
glm94 64326937b1 Added a LineEnd token type. 2022-08-25 18:37:25 +00:00
glm94 89ab951c6a Refactored the parsing code just a touch. 2022-08-24 21:23:14 +00:00
glm94 947de1d14c Added support for declaring a variable with the '.db' directive. Address calculations still need to be done, but at the very least the variable's name is showing up in the symbols table. 2022-05-16 15:17:19 +00:00
glm94 3c4cce0883 Added assembler directives as a class of token. 2022-05-16 14:27:49 +00:00
glm94 5265666178 Got the Parser to, so far, correctly (minus memory leaks) parse out and report errors with opcodes and their parameters. 2022-05-12 23:16:02 -05:00
glm94 2de3538d80 Parser is now mostly setup for the 'peek and advance' pattern. Still having infinite loop issues somewhere though. 2022-05-12 21:30:02 +00:00
glm94 72525959f6 Reworked the tokens to include a broad class to make parameter matching a bit easier, and removed the hexadecimal distinction. 2022-05-12 20:26:14 +00:00
glm94 72abf6c4ab Started work on rewriting the Parser. 2022-05-10 21:03:09 +00:00
20 changed files with 1867 additions and 340 deletions
+3
View File
@@ -1,3 +1,6 @@
assm assm
obj/ obj/
bin/ bin/
*.bin
docs/*
!docs/*.tex
+3 -3
View File
@@ -1,5 +1,5 @@
CC = gcc CC = gcc
CFLAGS=-g -Wall -DDEBUG -Wpedantic CFLAGS=-g -Wall -DDEBUG -Wpedantic -Wextra -Wunused-result -std=c99 -pedantic-errors
SRCDIR=src SRCDIR=src
OBJDIR=obj OBJDIR=obj
SRCS=$(wildcard $(SRCDIR)/*.c) SRCS=$(wildcard $(SRCDIR)/*.c)
@@ -11,7 +11,7 @@ BIN=$(BINDIR)/assm
all: $(BIN) all: $(BIN)
release: CFLAGS=-Wall -Wpedantic -O2 release: CFLAGS=-Wall -Wpedantic -std=c99 -pedantic-errors -O2
release: clean release: clean
release: $(BIN) release: $(BIN)
@@ -35,7 +35,7 @@ clean:
rm -rf $(BINDIR)/* $(OBJDIR)/* rm -rf $(BINDIR)/* $(OBJDIR)/*
test: test:
$(BIN) misc/test.asm $(BIN) misc/another_test.asm
disass: disass:
objdump -S --disassemble $(OBJDIR)/$(FILE).o > $(OBJDIR)/$(FILE).s objdump -S --disassemble $(OBJDIR)/$(FILE).o > $(OBJDIR)/$(FILE).s
+187
View File
@@ -0,0 +1,187 @@
\documentclass[a4paper,12pt]{book}
\usepackage{tikz}
\usepackage{hyperref}
\hypersetup{
linktoc=all
}
\title{Unnamed Machine}
\author{A Very Terrible 16-bit Machine}
\newcommand{\OpcodeTable}[5] {
\begin{tabular}{ c c c c c }
\hline
Opcode & Bit Pattern & Mnemonic & Operand 1 & Operand 2 \\
\hline\hline
#1 & #2 & #3 & #4 & #5 \\
\hline
\end{tabular}
}
\begin{document}
\maketitle
\tableofcontents
\chapter{Overview}
\section{Introduction}
This is a very poorly thought out 16-bit machine, but you've got to start somewhere. Currently debating between a CPU status register, 80x86 style, or just placing arithmetic results into a predetermined register. This is more of a loadstore architecture to try and keep the instruction set simple. The machine will be big-endian.
\section{Registers}
The following are the general purpose registers that can be used.
\begin{itemize}
\item[] R1
\item[] ...
\item[] R8
\end{itemize}
Additionally, there will be a special status register that will be a signed 16 bit register that the compare and jump instructions will read or write to when determining what action, if any, they'll take.
\section{Memory Model}
Memory will be implicitly mapped I/O. The bottom 3201 bytes of memory will be reserved for the keyboard input and graphics.
A single byte is reserved for the keyboard's input. The current key will be stored in byte 0xF37E, with the most significant bit being a flag indicating that the keyboard
is ready to be read from. This means that the character encoding is actually 7 bits.
Video memory starts at 0xF37F (62335 decimal), and every byte represents an ASCII character in monochrome.
\begin{figure}[!htb]
\centering
\begin{tikzpicture}
\fill[gray!5] (0,0)rectangle(5,10);
%\draw (0,10) .. controls (-2,6) and (-2,4) .. (0,1);
%\draw (0,10) arc (0:180:3cm);
\draw (0,10) -- (5,10);
\draw (0,1) -- node[above] {Video 0xF37F} (5,1);
\draw (0,0) -- (5,0);
\node[label=right:Top 0x0000] at (5,10) {};
\node[label=right:Bottom 0xFFFF] at (5,0) {};
\end{tikzpicture}
\caption{Memory Layout}
\end{figure}
\chapter{Instruction Set Architecture}
\section{Instruction Encoding}
The instruction encoding is fixed width to exactly 8 bites wide.
The encoding for the instructions will differ depending on if the opcode requires a register as its first operand.
For instructions that require, as their first argument, a register the encoding format will be as shown in Figure \ref{fig:WithRegister}.
The lower 3 bits may be any bit pattern as long as at least one bit is set in the upper 5 bits.
%https://tex.stackexchange.com/questions/32598/force-latex-image-to-appear-in-the-section-in-which-its-declared
\begin{figure}[!htb]
\centering
\begin{tabular}{ c c }
Instruction & Register \\
\hline
XXXX X & 000 \\
\hline
\end{tabular}
\caption{Encoding Layout, Register Required}
\label{fig:WithRegister}
\end{figure}
Figure \ref{fig:NoRegisterEncoding} shows how encoding will look for instructions that lack arguments or do not require a register.
In contrast with the previous encoding scheme the upper 5 bits MUST be zero, allowing 7 possible instructions to use this format.
All zeroes is not considered a legal instruction.
\begin{figure}[!htb]
\centering
\begin{tabular}{ c c }
Must Be Zero & Instruction \\
\hline
0000 0 & XXX \\
\hline
\end{tabular}
\caption{Encoding Layout, No Register Required}
\label{fig:NoRegisterEncoding}
\end{figure}
\section{Notes}
For the opcodes that load or store data at the assembly language level we could have the mnemonics
"store" and "load" and have the assembler pick the opcode based on the inclusion of the word "byte"
or "word" for two bytes. That would make the assembly easier to read but put a bit more work on the
assembler. The Stack Pointer will start 2 bytes above the video memory start.
\section{COPY (Copy Word from Address)}
\OpcodeTable{0x08}{0000 1000}{copya}{Register}{Address}\\[6pt]
Copy a machine word from Operand 2 into Operand 1.
\section{COPY BYTE (Copy Byte from Address)}
\OpcodeTable{0x10}{0001 0000}{copyab}{Register}{Address}\\[6pt]
Copy a byte (8 bits) from Operand 2 into Operand 1, clearing the setting the most significant bits to zero.
\section{COPY (Copy Word Indirect Address)}
\OpcodeTable{0x18}{0001 1000}{copyra}{Register}{[Register]}\\[6pt]
Copy a machine word from the address stored in Operand 2 into Operand 1.
\section{COPY BYTE (Copy Byte Indirect Address)}
\OpcodeTable{0x20}{0010 0000}{copyrab}{Register}{[Register]}\\[6pt]
Copy a byte (8 bits) from the address stored in Operand 2 into Operand 1.
\section{COPY (Copy)}
\OpcodeTable{0x28}{0010 1000}{copyrara}{[Register]}{[Register]}\\[6pt]
Copy a machine word from the address stored in Operand 2 into the address stored in Operand 1.
\section{COPY BYTE (Copy Byte)}
\OpcodeTable{0x30}{0011 0000}{copyrarab}{[Register]}{[Register]}\\[6pt]
Copy a byte (8 bits) from the address stored in Operand 2 into the address stored in Operand 1.
\section{COPY (Copy)}
\OpcodeTable{0x38}{0011 1000}{copy}{Register}{Register}\\[6pt]
Copy a machine word from Operand 2 into Operand 1.
\section{COPY (Copy Immediate)}
\OpcodeTable{0xC0}{1100 0000}{copyi}{Register}{Constant}\\[6pt]
Copy a machine word from Operand 2 into Operand 1.
\section{COPY BYTE (Copy Byte)}
\OpcodeTable{0x40}{0100 0000}{copyb}{Register}{Constant}\\[6pt]
Copy a byte (8 bits) from Operand 2 into Operand 1.
\section{CMP (Compare)}
\OpcodeTable{0x48}{0100 1000}{cmp}{Register}{Register}\\[6pt]
Compares two registers and somewhere sets a result in the status register.
\section{CMPI (Compare Immediate)}
\OpcodeTable{0x50}{0101 0000}{cmpi}{Register}{Constant}\\[6pt]
Compares an immediate 2 byte value to the contents of a register setting the status register accordingly.
\section{ADD (Add)}
\OpcodeTable{0x58}{0101 1000}{add}{Register}{Register}\\[6pt]
Performs addition on a register with a value from another (or the same) register.
\section{SUB (Subtract)}
\OpcodeTable{0x60}{0110 0000}{sub}{Register}{Register}\\[6pt]
Performs subtraction on a register with a value from another (or the same) register.
\section{AND (Logical AND)}
\OpcodeTable{0x68}{0110 1000}{and}{Register}{Register}\\[6pt]
Logical ANDs the two registers together storing the result in operand 1.
\section{XOR (Logical Exclusive OR)}
\OpcodeTable{0x70}{0111 0000}{xor}{Register}{Register}\\[6pt]
Logical XORs the two registers together storing the result in operand 1.
\section{OR (Logical OR)}
\OpcodeTable{0x78}{0111 1000}{or}{Register}{Register}\\[6pt]
Logical ORs the two registers together storing the result in operand 1.
\section{NOT (Logical Negation)}
\OpcodeTable{0x80}{1000 0000}{not}{Register}{None}\\[6pt]
Inverts the bits of the target register.
\section{SHR (Shift Right)}
\OpcodeTable{0x88}{1000 1000}{shr}{Register}{Constant}\\[6pt]
Bit-wise shifts the contents of the register right Constant number of times.
\section{SHL (Shift Left)}
\OpcodeTable{0x90}{1001 0000}{shl}{Register}{Constant}\\[6pt]
Bit-wise shifts the contents of the register left Constant number of times.
\section{INC (Increment)}
\OpcodeTable{0x98}{1001 1000}{inc}{Register}{None}\\[6pt]
Increments the contents of the register by one. Over-flows will not be reported.
\section{DEC (Decrement)}
\OpcodeTable{0xA0}{1010 0000}{dec}{Register}{None}\\[6pt]
Decrements the contents of the register by one. Under-flows will not be reported.
\section{Push}
\OpcodeTable{0xA8}{1010 1000}{push}{Register}{None}
Pushes the value of Register onto the stack, decrementing the Stack Pointer by 2.
\section{Pop}
\OpcodeTable{0xB0}{1011 0000}{pop}{Register}{None}
Pops the top of the stack into Register, incrementing the Stack Pointer by 2.
\section{JMPI (Jump Indirect)}
\OpcodeTable{0xB8}{1011 1000}{jmpi}{Register}{None}\\[6pt]
Jumps unconditionally to a memory address stored in Operand 1. Sets the Program Counter to Operand 1.
\section{JMP (Jump)}
\OpcodeTable{0x02}{0000 0010}{jmp}{Address}{None}\\[6pt]
Jumps unconditionally to a memory address. Sets the Program Counter to Address.
\section{JZ (Jump if Zero)}
\OpcodeTable{0x03}{0000 0011}{jz}{Address}{None}\\[6pt]
Jumps to a memory address if the status flag is zero. Sets the Program Counter to Address.
\section{JG (Jump if Greater Than)}
\OpcodeTable{0x04}{0000 0100}{jg}{Address}{None}\\[6pt]
Jump to the Address if the status flag is greater than zero. Sets the Program Counter to Address.
\section{JL (Jump if Less Than)}
\OpcodeTable{0x05}{0000 0101}{jl}{Address}{None}\\[6pt]
Jumps to the address if the status flag is less than zero. Sets the Program Counter to Address.
\section{NOP (No Operation)}
\OpcodeTable{0x01}{0000 0001}{nop}{None}{None}\\[6pt]
Skips a clock cycle, incrementing the program counter.
\section{CALL (Call Subroutine)}
\OpcodeTable{0x06}{0000 0110}{call}{Address}{None}\\[6pt]
Pushes the base address to the stack and sets the Program Counter to Address.
\section{RET (Return from Subroutine)}
\OpcodeTable{0x07}{0000 0111}{ret}{None}{None}\\[6pt]
Pops the stack and sets the Program Counter to that value.
\end{document}
+9
View File
@@ -0,0 +1,9 @@
#ifndef DIASSEMBLER_H
#define DIASSEMBLER_H
#include "parser.h"
//void Disassemble(unsigned char image[HIGHMEMORY]);
#endif
-21
View File
@@ -1,21 +0,0 @@
#ifndef LIST_H
#define LIST_H
#include <stddef.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#define LISTDEFAULTSIZE 4
typedef struct {
void** content;
int size;
int capacity;
} List;
List* CreateList(void);
int AddListItem(const void *, size_t, List *);
void DestroyList(List*);
#endif
+14 -25
View File
@@ -2,34 +2,23 @@
#define OPCODES_H #define OPCODES_H
#include <string.h> #include <string.h>
#include "token.h" #include <errno.h>
#include <stdio.h>
#define OPCODECOUNT 10
#define REGISTERCOUNT 8
typedef enum { typedef enum {
None = 0, R1 = 0, R2, R3, R4, R5, R6, R7, R8 = 7
Reg, } Registers;
Imm8,
Imm16
} ParameterType;
typedef struct { typedef enum {
char* lexeme; COPYA = 0x08, COPYAB = 0x10, COPYRA = 0x18, COPYRAB = 0x20, COPYRARA = 0x28, COPYRARAB = 0x30, COPY = 0x38, COPYB = 0x40, COPYI = 0xC0,
TokenType op; CMP = 0x48, CMPI = 0x50, ADD = 0x58, SUB = 0x60, AND = 0x68, XOR = 0x70, OR = 0x78, NOT = 0x80, SHR = 0x88, SHL = 0x90,
ParameterType parameter_one; INC = 0x98, DEC = 0xA0, PUSH = 0xA8, POP = 0xB0,
ParameterType parameter_two; JMPI = 0xB8, JMP = 0x02, JZ = 0x03, JG = 0x04, JL = 0x05, NOP = 0x01, CALL = 0x06, RET = 0x07
} OpCode; } Mnemonic;
typedef struct { int IsOpcode(const char*, Mnemonic*);
char* lexeme; int IsRegister(const char*, Registers*);
TokenType type; void GetMnemonicText(Mnemonic mnemonic, char buffer[12]);
} Register; void GetRegisterText(Registers reg, char buffer[3]);
extern OpCode opcodes[OPCODECOUNT];
extern Register registers[REGISTERCOUNT];
int IsOpcode(const char*, TokenType*);
int IsRegister(const char*, TokenType*);
#endif #endif
+6 -2
View File
@@ -1,11 +1,15 @@
#ifndef PARSER_H #ifndef PARSER_H
#define PARSER_H #define PARSER_H
#include <stdint.h>
#include <stdio.h> #include <stdio.h>
#include <stdlib.h> #include <stdlib.h>
#include "list.h" #include <string.h>
#include <stdarg.h>
#include "token.h" #include "token.h"
#include "opcodes.h"
#include "symbols_table.h"
void ParseTokens(List*); void ParseTokens(TokenList* tokens, SymbolTable** symbolsTable);
#endif #endif
+1 -2
View File
@@ -7,9 +7,8 @@
#include <errno.h> #include <errno.h>
#include "stdlib.h" #include "stdlib.h"
#include "token.h" #include "token.h"
#include "list.h"
#include "opcodes.h" #include "opcodes.h"
List* GenerateTokenList(const char*); TokenList* GenerateTokenList(const char*);
#endif #endif
+44
View File
@@ -0,0 +1,44 @@
#ifndef SYMBOLSTABLE_H
#define SYMBOLSTABLE_H
#include <stdint.h>
#include <stdlib.h>
#include <errno.h>
#include <stdio.h>
#include <string.h>
#include <sys/types.h>
#include "opcodes.h"
#include "token.h"
#define SYMBOLSTABLE_DEFAULT_CAPACITY 128
typedef struct __token_node {
const Token* Token;
struct __token_node* Left;
struct __token_node* Right;
} TokenNode;
typedef struct _symbol {
const char* Name;
int Address;
int Length;
const Token* Token;
TokenNode* ValueExpression;
} Symbol;
typedef struct {
Symbol** Symbols;
int Size;
int Capacity;
} SymbolTable;
SymbolTable* CreateSymbolTable(void);
TokenNode* CreateTokenNode(const Token* token);
int TryGetSymbol(const char* name, const SymbolTable* table, Symbol** outSymbol);
Symbol* AddSymbolToTable(const char* name, int address, SymbolTable* table);
void FreeSymbolTable(SymbolTable* table);
void FreeSymbol(Symbol* symbol);
int SymbolResolved(const Symbol* symbol, const SymbolTable* table);
int TryGetSymbolValue(const Symbol* symbol, const SymbolTable* table, unsigned short* value);
#endif
+58 -30
View File
@@ -5,40 +5,68 @@
#include <string.h> #include <string.h>
#include <errno.h> #include <errno.h>
#include <stdio.h> #include <stdio.h>
#include "opcodes.h"
typedef enum { typedef enum {
PLUS, Plus = '+',
MINUS, Minus = '-',
STAR, Star = '*',
SLASH, Slash = '/',
POWER, Power = '^',
LPARAM, LParan = '(',
RPARAM, RParan = ')',
LBRACKET, LBracket = '[',
RBracket, RBracket = ']',
COMMA, Comma = ',',
STRING, NewLine = '\n'
IDENTIFIER, } TokenPunctuation;
LABEL,
NUMBER,
HEX,
//Keywords
DB, ORG,
//Opcodes
COPY, ADD, SUB, JZ, INT, YLD, RET,
CMP, IN, OUT,
//Registers
R1, R2, R3, R4, R5, R6, R7, R8
} TokenType;
typedef struct { typedef enum {
TokenType type; DB,
char* lexeme; Include,
void* value; Byte
int line; } Directive;
typedef enum {
RegisterClass = 0,
NumberClass = 1,
CharacterClass = 2,
MnemonicClass = 4,
DirectiveClass = 8,
PunctuationClass = 16,
IdentifierClass = 32,
LabelClass = 64,
AddressClass = LabelClass | IdentifierClass
} TokenClass;
typedef struct __token {
char* Lemexe;
TokenClass Class;
int LineNumber;
int EndOfFile;
union {
TokenPunctuation Punctuation;
Registers Register;
Mnemonic Mnemonic;
Directive Directive;
unsigned short Number;
} Value;
struct __token* Prev;
struct __token* Next;
} Token; } Token;
Token* CreateToken(char*, void*, int, TokenType); typedef struct {
void FreeToken(Token*); Token** content;
int size;
int capacity;
} TokenList;
Token* CreateToken(int lineNumber, TokenClass tokenClass);
TokenList* CreateTokenList(void);
int AddToken(Token* token, TokenList* list);
void RemoveToken(int index, TokenList* list);
void FreeToken(Token* token);
#endif #endif
+44
View File
@@ -0,0 +1,44 @@
.db MAX_MEM 0xFFFF
.db VIDEO_MEM 0xF37F
.db MAX_LENGTH MAX_MEM; - MAX_LENGTH
.db MSG "Hello, World!", 0
__start:
copy r1, MSG ; Pointer into r1
copy r8, VIDEO_MEM
call strlen
copy r1, MSG
cmp r2, 0
jz _end
cmp r2, MAX_LENGTH
jg _end
draw_loop:
copy byte [r8], [r1]
inc r1
inc r8
dec r2
cmp r2, 0
jz _end
jmp draw_loop
_end:
jmp _end
.db NewMsg "My message", 0
; Returns the length of a NULL terminated string
; Arguments: R1 - Pointer to the string
; Returns: R2 - Contains the length of the string
strlen:
copy r2, 0 ; length
copy byte r3, [r1]
cmp r3, 0
jz end ;The string is zero length
loop:
inc r1
copy byte r3, [r1]
cmp r3, 0
jz end
inc r2
jmp loop
end:
ret
+26 -44
View File
@@ -1,45 +1,27 @@
.org 0x100 ;.include "./another_test.asm"
.db labelsz reference
.db video_start 0xF37F
.db msg "Hello, world!", 0 .db msg "Hello, world!", 0
.db more_stuff "And yet another string!", 0 .db NOTERM "No terminating byte here"
.db null_byte 0
copy r1, msg jmp [r4]
copy r2, 0 ;load r2, label
loop: something_insance:
cmp r1, 0 load r1, unknown_symbol
jz loop load r1, 5
add r1, 1 .db fun_alright 0x70;does this break?
mov r1, 5 ; move the immediate value 5 into r1 load r3, r4
add r1,5 ; add 5 into r1 store byte r3, r5
int 21 ; maybe that will call some string drawing BIOS-like routine store r5, r7
ret unknown_symbol:;Does this work?
; form YYYY YYXX - Y opcode, X modifier cmp r1, 5;or one after the register?
; nop: 0000 0000 cmp r1, r2;This might work
; copy: 0000 01XX and r1, r2
; add: 0000 10XX - | Perhaps running these will clear any overflow add r4, r7
; addc: 0000 11XX - | or under flow flags if no errors occur. inc r1 ;increment r1
; sub: 0001 00XX - | jz should clear the zero flag. xor r1, r1 ;clear self
; subb: 0001 01XX - | pop r3
; call: 0001 1100 - Always call [imm16] jmp unknown_symbol
; jmp: 0010 0000 - Always jmp [imm16] (No short jumps) jmp [r4]
; jz: 0010 0100 - Always jz [imm16] (no short jumps) jz [r6]
; cmp: 0010 11XX ;And a comment at the end
; push: 0011 11XX -| pop / push imm16/imm8 | reg
; pop: 0100 00XX -|
;
; imm8 imm16 reg
; mov reg, imm16|reg
; add reg, imm16|reg
; int imm16
; db [label] 'String data here' (Null byte is added implicatly by the assemblier)
;
; enum Type {
; Op,
; Reg,
; Imm16,
; Imm8
; }
; struct Token {
; enum Type type;
; char *value;
; }
+161
View File
@@ -0,0 +1,161 @@
#include "../includes/disass.h"
#include <stdlib.h>
const unsigned char* Image;
const unsigned char COPYMASKREG = 0x20;
const unsigned char COPYMASKADD = 0xA0;
const unsigned char ADDMASK = 0x40;
const unsigned char SUBMASK = 0x60;
const unsigned char CMPMASK = 0x80;
const unsigned char REGMASK = 0x00;//0x18; //0001 1000
const unsigned char CONSTMASK = 0x08; //0000 1000
const unsigned char ADDRESSMASK = 0x10; //0001 0000
int IsRegisterPattern(unsigned char pattern, char* lexeme);
void GetParameter(unsigned char instruction, char* text);
/*
void Disassemble(unsigned char image[HIGHMEMORY]) {
Image = image;
int position = image[0] + image[1];
int end = image[2] + image[3];
char* parameter1 = calloc(32, sizeof(char));
char* parameter2 = calloc(32, sizeof(char));
unsigned char instruction = image[position];
while(end > position) {
unsigned char masked = instruction & 0xE0;
if (masked) {
IsRegisterPattern(instruction & 0x07, parameter1);
unsigned char parameterType = instruction & 0x18;
if (parameterType == REGMASK) {
IsRegisterPattern(Image[position + 1], parameter2);
position += 2;
} else if (parameterType == CONSTMASK) {
snprintf(parameter2, sizeof(char) * 31, "%d", Image[position + 1] + Image[position + 2]);
position += 3;
}
else if (parameterType == ADDRESSMASK) {
snprintf(parameter2, sizeof(char) * 31, "[%#04X]", Image[position + 1] + Image[position + 2]);
position += 3;
}
//If the top bits are set
if (masked == COPYMASKREG) {
printf("COPY %s, %s\n", parameter1, parameter2);
}
else if (masked == COPYMASKADD) {
printf("COPY %s, %s\n", parameter2, parameter1);
}
else if (masked == ADDMASK) {
printf("ADD %s, %s\n", parameter1, parameter2);
}
else if (masked == SUBMASK) {
printf("SUB %s, %s\n", parameter1, parameter2);
}
else if (masked == CMPMASK) {
printf("CMP %s, %s\n", parameter1, parameter2);
}
else {
fprintf(stderr, "[Error] Unknown opcode %#02X\n", masked);
exit(1);
}
}
else {
int parameter = Image[position + 1] + Image[position + 2];
switch(instruction) {
case 0:
printf("NOP\n");
position++;
break;
case 1:
printf("JZ [%#04X]\n", parameter);
position += 3;
break;
case 2:
printf("INT %d\n", parameter);
position += 3;
break;
case 3:
printf("YLD\n");
position++;
break;
case 4:
printf("RET\n");
position++;
break;
case 5:
printf("CALL [%#04X]\n", parameter);
position += 3;
break;
case 6:
printf("JMP [%#04X]\n", parameter);
position += 3;
break;
case 7:
printf("IN %d\n", parameter);
position += 3;
break;
case 8:
printf("OUT %d\n", parameter);
position += 3;
break;
default:
position++;
break;
}
}
instruction = Image[position];
memset(parameter1, '\0', sizeof(char) * 32);
memset(parameter2, '\0', sizeof(char) * 32);
}
}
*/
int IsRegisterPattern(unsigned char pattern, char* lexeme) {
if (pattern > 8) {
lexeme[0] = '\0';
return 0;
}
//The registers' bit patterns are zero indexed, so 'r1' is '000'
//but 'r8' is '111' and so forth.
pattern++;
lexeme[0] = 'r';
lexeme[1] = pattern | 0x30;
lexeme[2] = '\0';
return 1;
}
/*
//REG XXX0 0XXX
//Constant XXX0 1XXX
//Address XXX1 0XXX
NOP - 0000 0000 NOP
JZ - 0000 0001 JZ Address
INT - 0000 0010 INT Constant
YLD - 0000 0011 YLD
RET - 0000 0100 RET
CALL - 0000 0101 CALL Address
JMP - 0000 0110 JMP Address
IN - 0000 0111 IN Constant
OUT - 0000 1000 OUT Constant
COPY - 0010 0XXX COPY REG, REG
- 0010 1XXX COPY REG, Constant
- 0011 0XXX COPY REG, Address
- 1010 0XXX COPY Address, REG
- 1010 1XXX COPY Address, Constant
- 1011 0XXX COPY Address, Address
ADD - 0100 0XXX ADD REG, REG
- 0100 1XXX ADD REG, Constant
SUB - 0110 0XXX SUB REG, REG
- 0110 1XXX SUB REG, Constant
CMP - 1000 0XXX CMP REG, REG
- 1000 1XXX CMP REG, Constant
- 1001 0XXX CMP REG, Address */
-66
View File
@@ -1,66 +0,0 @@
#include "../includes/list.h"
List* CreateList() {
List *new = malloc(sizeof(List));
if (!new) {
fprintf(stderr, "Failed to malloc() for new new List.\n");
return NULL;
}
new->content = malloc(sizeof(void*) * LISTDEFAULTSIZE);
if (!new->content) {
fprintf(stderr, "Failed to malloc() memory for List contents.\n");
free(new);
return NULL;
}
new->size = 0;
new->capacity = LISTDEFAULTSIZE;
return new;
}
int AddListItem(const void *value, size_t size, List* list) {
if (!list) return -1;
if (!value) return -1;
if (size == 0) return -1;
if (list->capacity < list->size + 1) {
void* ptr = realloc(list->content, sizeof(void*) * list->capacity * 2);
//Note: realloc will free list->root if it succeeds.
if (!ptr) {
fprintf(stderr, "Failed to resize array with realloc() (%d bytes).\n", list->capacity * 2);
return -1;
}
list->content = ptr;
list->capacity = list->capacity * 2;
}
void* item = calloc(1, size);
if (!item) {
fprintf(stderr, "Failed to calloc() new memory (%lu bytes).\n", size);
return -1;
}
memcpy(item, value, size);
list->content[list->size] = item;
list->size++;
return 0;
}
void DestroyList(List* list) {
if (!list) return;
for(int i = 0; i < list->size; i++) {
free(list->content[i]);
}
free(list->content);
free(list);
}
+226 -12
View File
@@ -1,11 +1,20 @@
#include <bits/types/FILE.h>
#include <stdlib.h> #include <stdlib.h>
#include <stdio.h> #include <stdio.h>
#include <string.h> #include <string.h>
#include <sysexits.h> #include <sysexits.h>
#include "../includes/list.h" #include "../includes/token.h"
#include "../includes/parser.h"
#include "../includes/futil.h" #include "../includes/futil.h"
#include "../includes/scanner.h" #include "../includes/scanner.h"
#include "../includes/parser.h"
const char* MagicStartName = "__start";
TokenList* LIST;
SymbolTable* Symbols = NULL;
unsigned char mem[128] = {0};
void print(void);
void assemble(void);
int main(int argc, char* args[]) { int main(int argc, char* args[]) {
if (argc == 1) { if (argc == 1) {
@@ -13,22 +22,227 @@ int main(int argc, char* args[]) {
return EX_USAGE; return EX_USAGE;
} }
atexit(print);
char* source_code; char* source_code;
size_t bytes_read; size_t bytes_read;
if (!ReadAllString(args[1], &source_code, &bytes_read)) return EX_IOERR; if (!ReadAllString(args[1], &source_code, &bytes_read)) return EX_IOERR;
List* list = GenerateTokenList(source_code); LIST = GenerateTokenList(source_code);
// for(int i = 0; i < list->size; i++) { ParseTokens(LIST, &Symbols);
// Token* t = (Token*) list->content[i];
// printf("[%i] '%s'", t->type, t->lexeme);
// if (t->type == LABEL) printf("*");
// printf("\n");
// }
ParseTokens(list);
free(source_code); free(source_code);
assemble();
printf("\nBytes:\n");
for(unsigned long i = 0; i < sizeof(mem); i++) {
if (i != 0 && i % 8 == 0) printf("\n");
printf("%02X ", mem[i] & 0xFF);
}
printf("\n");
} }
void print(void) {
char mnemonic[12];
printf("printing tokens...\n");
for(int i = 0; i < LIST->size; i++) {
Token* t = (Token*) LIST->content[i];
if (t->EndOfFile) {
printf("EOF\n");
break;
}
if (t->Class == PunctuationClass){
if(t->Value.Punctuation == NewLine) {
printf("<%d>\n", t->LineNumber);
continue;
}
printf("[P]%c", t->Value.Punctuation);
continue;
}
if (t->Class == LabelClass) {
printf("[L]%s*", t->Lemexe);
continue;
}
if (t->Class == IdentifierClass) {
printf("[I]%s ", t->Lemexe);
continue;
}
if (t->Class == RegisterClass) {
printf("[R]%d", t->Value.Register);
continue;
}
if (t->Class == NumberClass) {
printf("[N]%d ", t->Value.Number);
continue;
}
if (t->Class == MnemonicClass) {
GetMnemonicText(t->Value.Mnemonic, mnemonic);
printf("[M]%s ", mnemonic);
continue;
}
if (t->Class == DirectiveClass) {
printf("[D]%d ", t->Value.Directive);
}
if (t->Class == CharacterClass) {
printf("[C]'%s'", t->Lemexe);
}
}
}
int CurrentIndex = 0;
int PC2 = 0;
void WriteByte(unsigned char byte) {
mem[PC2] = byte;
PC2++;
}
void WriteWord(unsigned short word) {
mem[PC2] = (word >> 8) & 0xFF;
mem[PC2 + 1] = word & 0xFF;
PC2 += 2;
}
unsigned short GetValueFromToken(Token* token) {
if (token->Class & IdentifierClass) {
Symbol* symbol;
if (TryGetSymbol(token->Lemexe, Symbols, &symbol)) {
return symbol->Address;
}
else {
fprintf(stderr, "Failed to find symbol %s while assembling.\n", token->Lemexe);
exit(1);
}
}
else if (token->Class == NumberClass) return token->Value.Number;
fprintf(stderr, "Failed to find token '%d' while assmbling.\n", token->Class);
exit(1);
}
int IsAtEnd(void){
return CurrentIndex >= LIST->size;
}
void assemble() {
while(!IsAtEnd()) {
Token* current = LIST->content[CurrentIndex];
switch(current->Value.Mnemonic) {
case COPYA:
case COPYAB:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case COPYRA:
case COPYRAB:
case COPYRARA:
case COPYRARAB:
case COPY:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteByte(((Token*) LIST->content[CurrentIndex])->Value.Register);
break;
case COPYI: //TODO: should explicitly get number from value I feel
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case COPYB:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteByte((unsigned char)GetValueFromToken(LIST->content[CurrentIndex]));
break;
case CMP:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteByte(((Token*) LIST->content[CurrentIndex])->Value.Register);
break;
case CMPI:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case ADD:
case SUB:
case AND:
case OR:
case XOR:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteByte(((Token*) LIST->content[CurrentIndex])->Value.Register);
break;
case SHL:
case SHR:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case INC:
case DEC:
case PUSH:
case POP:
case JMPI:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
break;
case JMP:
case JZ:
case JG:
case JL:
WriteByte(current->Value.Mnemonic);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case NOP:
case RET:
WriteByte(current->Value.Mnemonic);
break;
case CALL:
WriteByte(current->Value.Mnemonic);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
default:
break;
}
CurrentIndex++;
}
}
+75 -31
View File
@@ -1,53 +1,97 @@
#include "../includes/opcodes.h" #include "../includes/opcodes.h"
#include <ctype.h>
#include <stdlib.h> #include <stdlib.h>
#include <string.h> #include <string.h>
OpCode opcodes[OPCODECOUNT] = { #define OPCODECOUNT 34
{ "copy", COPY, Reg, Reg | Imm8 | Imm16 },
{ "add", ADD, Reg, Reg | Imm8 | Imm16 }, struct _instruction {
{ "sub", SUB, Reg, Reg | Imm8 | Imm16 }, char* Name;
{ "jz", JZ, Imm8 | Imm16, None }, Mnemonic Mnemonic;
{ "int", INT, Imm8, None },
{ "yld", YLD, None, None },
{ "ret", RET, None, None },
{ "cmp", CMP, Reg | Imm8 | Imm16, Reg | Imm8 | Imm16 },
{ "in", IN, None, None},
{ "out", OUT, None, None}
}; };
Register registers[REGISTERCOUNT] = { struct _instruction instructions[31] = {
{ "r1", R1 }, { "copya", COPYA },
{ "r2", R2 }, { "copyab", COPYAB },
{ "r3", R3 }, { "copyra", COPYRA },
{ "r4", R4 }, { "copyrab", COPYRAB },
{ "r5", R5 }, { "copyrara", COPYRARA },
{ "r6", R6 }, { "copyrarab", COPYRARAB },
{ "r7", R7 }, { "copy", COPY },
{ "r8", R8 } { "copyi", COPYI },
{ "copyb", COPYB },
{ "cmp", CMP },
{ "cmpi", CMPI },
{ "add", ADD },
{ "sub", SUB },
{ "and", AND },
{ "xor", XOR },
{ "or", OR },
{ "not", NOT },
{ "shr", SHR },
{ "shl", SHL },
{ "inc", INC },
{ "dec", DEC },
{ "push", PUSH },
{ "pop", POP },
{ "jmpi", JMPI },
{ "jmp", JMP },
{ "jz", JZ },
{ "jg", JG },
{ "jl", JL },
{ "nop", NOP },
{ "call", CALL },
{ "ret", RET }
//yld
}; };
int IsOpcode(const char* text, TokenType* opcode) { void GetMnemonicText(Mnemonic mnemonic, char buffer[12]) {
if (!text) return 0; memset(buffer, '\0', 12);
for(int i = 0; i < OPCODECOUNT; i++) { for(int i = 0; i < OPCODECOUNT; i++) {
if (strcmp(opcodes[i].lexeme, text) == 0) { if (instructions[i].Mnemonic == mnemonic) {
*opcode = opcodes[i].op; strncpy(buffer, instructions[i].Name, 11);
return 1; break;
} }
} }
return 0;
} }
int IsRegister(const char* text, TokenType* reg) { void GetRegisterText(Registers reg, char buffer[3]) {
memset(buffer, '\0', 3);
if (reg < R1 || reg > R8) return;
buffer[0] = 'r';
buffer[1] = reg + 49;
}
int IsOpcode(const char* text, Mnemonic* opcode) {
if (!text) return 0; if (!text) return 0;
for(int i = 0; i < REGISTERCOUNT; i++) { for(unsigned long i = 0; i < sizeof(instructions) / sizeof(struct _instruction); i++) {
if (strcmp(registers[i].lexeme, text) == 0) { if (strcmp(instructions[i].Name, text) == 0) {
*reg = registers[i].type; if (opcode) *opcode = instructions[i].Mnemonic;
return 1; return 1;
} }
} }
return 0; return 0;
} }
int IsRegister(const char* text, Registers* reg) {
if (!text) return 0;
int length = strlen(text);
Registers r = R8;
if (length != 2) return 0;
if (text[0] != 'r') return 0;
if (!isdigit(text[1])) return 0;
r = text[1] - 0x31;
if (reg) *reg = r;
return 1;
}
+570 -51
View File
@@ -3,87 +3,606 @@
#include <stdlib.h> #include <stdlib.h>
#include <string.h> #include <string.h>
typedef struct { typedef enum {
Token* token; NoOptions = 0, ForwardParser = 1, RemoveExpected = 2
int address; } ExpectOptions;
} Symbol;
TokenList* TokensList;
int CurrentToken = 0;
Symbol* CreateSymbol(Token*, int);
void AddSymbol(Token*);
void PrintSymbols(void); void PrintSymbols(void);
Token* GetSymbol(char*); void RemoveCurrentToken(void);
List* SymbolsTable; void HandleAssemblerDirective(void);
unsigned int ProgramCounter = 0; Token* ExpectMnemonic(void);
void AdvanceParser(void);
void IgnoreParserLine(void);
Token* PeekToken(void);
int ParserAtEnd(void);
void ParseTokens(List* tokens) { void ExpectPuncuation(TokenPunctuation punctuation, ExpectOptions options);
SymbolsTable = CreateList(); void ExpectRegister(void);
Symbol* ExpectIdentifier(ExpectOptions options, int expectUnresolved);
void ExpectCharacterClass(Symbol* symbol);
Token* ExpectLineEndOrFileEnd(ExpectOptions options);
Token* ExpectMathOperator(ExpectOptions options);
Token* ExpectMathOperand(ExpectOptions options);
for(int i = 0; i < tokens->size; i++){ SymbolTable* SymbolsTable;
Token* t = tokens->content[i]; TokenNode* ParseSymbolExpression(void);
switch(t->type) { //int HeapSize = 0;
case IDENTIFIER: int PC = 0;
case LABEL:
AddSymbol(t); void ParseTokens(TokenList* tokens, SymbolTable** symbols) {
if (!tokens) return;
if (!symbols) return;
TokensList = tokens;
if (!*symbols) *symbols = CreateSymbolTable();
SymbolsTable = *symbols;
while(!ParserAtEnd()) {
Token* t = PeekToken();
if (!t || t->EndOfFile) break;
switch(t->Class) {
case DirectiveClass:
HandleAssemblerDirective();
break; break;
case STRING: case MnemonicClass:
ProgramCounter += strlen(t->lexeme); ExpectMnemonic();
break; break;
case NUMBER: case LabelClass:
case HEX: {
if ((long) t->value < 256) ProgramCounter += 1; Symbol* symbol;
else ProgramCounter += 2; int found = TryGetSymbol(t->Lemexe, SymbolsTable, &symbol);
if (found && SymbolResolved(symbol, SymbolsTable)) {
fprintf(stderr, "[Error] Line %d: Redefinition of symbol '%s'\n", t->LineNumber, t->Lemexe);
exit(1);
}
if (!found) symbol = AddSymbolToTable(t->Lemexe, PC, SymbolsTable);
symbol->Address = PC;
RemoveCurrentToken(); //label
ExpectLineEndOrFileEnd(RemoveExpected);
symbol->Token = ExpectMnemonic();
}
break; break;
default: default:
if (t->type > ORG) ProgramCounter += 1; fprintf(stderr, "[Warning] Line %d: Syntax error, expected start of expression, got '%c' [%d].\n", t->LineNumber, t->Value.Punctuation, t->Class);
exit(1);
break; break;
} }
} }
PrintSymbols(); PrintSymbols();
DestroyList(SymbolsTable);
printf("Program Counter: %d\n", ProgramCounter);
} }
void PrintSymbols(void) { void HandleAssemblerDirective() {
printf("-----SYMBOLS-----\n"); Token* directive = PeekToken();
for(int i = 0; i < SymbolsTable->size; i++) {
Symbol* symbol = SymbolsTable->content[i]; switch(directive->Value.Directive){
printf("[%#08X] %s\n", symbol->address, symbol->token->lexeme); case DB:
{
RemoveCurrentToken();
Symbol* symbol = ExpectIdentifier(RemoveExpected, 1);
if (PeekToken()->Class == CharacterClass) {
ExpectCharacterClass(symbol);
}
else if (PeekToken()->Class == NumberClass || PeekToken()->Class == IdentifierClass) {
symbol->Length = 2;
symbol->ValueExpression = ParseSymbolExpression();
}
}
break;
default:
fprintf(stderr, "Synatx error on line %d, %s.\n", directive->LineNumber, directive->Lemexe);
exit(1);
} }
printf("-----SYMBOLS-----\n");
} }
void AddSymbol(Token* token) { Token* ExpectMnemonic(void) {
if (!token) return; Token* opcode = PeekToken();
if (token->type != IDENTIFIER && token->type != LABEL) return;
for(int i = 0; i < SymbolsTable->size; i++) { if (opcode->Class != MnemonicClass) {
Symbol* s = SymbolsTable->content[i]; fprintf(stderr, "Syntax error on line %d: expected mnemonic.\n", opcode->LineNumber);
if (strcmp(s->token->lexeme, token->lexeme) == 0) return; exit(1);
} }
Symbol* symbol = CreateSymbol(token, ProgramCounter); AdvanceParser();
AddListItem(symbol, sizeof(Symbol), SymbolsTable); switch(opcode->Value.Mnemonic) {
} case COPY:
{
Token* arg = PeekToken();
int isByte = 0;
Token* GetSymbol(char* name) { if (arg->Class == DirectiveClass) {
if (!name) return NULL; if (arg->Value.Directive != Byte) {
fprintf(stderr, "Syntax error on line %d: expected keyword 'byte'.\n", opcode->LineNumber);
return NULL; exit(1);
} }
RemoveCurrentToken(); //byte
Symbol* CreateSymbol(Token* token, int offset) { isByte = 1;
Symbol* symbol = calloc(1, sizeof(Symbol));
if (!symbol) { arg = PeekToken();
return NULL;
} }
symbol->token = token; PC++;
symbol->address = offset;
if (arg->Class == RegisterClass) {
ExpectRegister();
ExpectPuncuation(Comma, RemoveExpected);
arg = PeekToken();
if (arg->Class == NumberClass) {
opcode->Value.Mnemonic = isByte ? COPYB : COPYI;
opcode->Lemexe = isByte ? "copyb" : "copyi";
isByte ? PC++ : (PC += 2);
AdvanceParser();
}
else if (arg->Class == RegisterClass) {
if (isByte) {
fprintf(stderr, "Syntax error on line %d: unexpected modifier 'Byte' for Register to Register copy.\n", opcode->LineNumber);
exit(1);
}
ExpectRegister();
opcode->Value.Mnemonic = COPY;
opcode->Lemexe = "copy";
PC++;
}
else if (arg->Class & IdentifierClass) {
opcode->Value.Mnemonic = isByte ? COPYAB : COPYA;
opcode->Lemexe = isByte ? "copyab" : "copya";
PC += 2;
ExpectIdentifier(ForwardParser, 0);
}
else {
ExpectPuncuation(LBracket, RemoveExpected);
ExpectRegister();
ExpectPuncuation(RBracket, RemoveExpected);
opcode->Value.Mnemonic = isByte ? COPYRAB : COPYRA;
opcode->Lemexe = isByte ? "copyrab" : "copyra";
PC++;
}
}
else {
ExpectPuncuation(LBracket, RemoveExpected);
ExpectRegister();
ExpectPuncuation(RBracket, RemoveExpected);
ExpectPuncuation(Comma, RemoveExpected);
ExpectPuncuation(LBracket, RemoveExpected);
ExpectRegister();
ExpectPuncuation(RBracket, RemoveExpected);
opcode->Value.Mnemonic = isByte ? COPYRARAB : COPYRARA;
opcode->Lemexe = isByte ? "copyrarab" : "copyrara";
PC++;
}
}
break;
case CMP:
{
ExpectRegister();
ExpectPuncuation(Comma, RemoveExpected);
TokenClass class = PeekToken()->Class;
PC++;
if (class == NumberClass) {
opcode->Value.Mnemonic = CMPI;
opcode->Lemexe = "CMPI";
AdvanceParser();
PC += 2;
}
else if (class & IdentifierClass) {
opcode->Value.Mnemonic = CMPI;
opcode->Lemexe = "CMPI";
ExpectIdentifier(ForwardParser, 0);
PC += 2;
}
else {
ExpectRegister();
PC++;
}
}
break;
case ADD:
case SUB:
case AND:
case OR:
case XOR:
PC++;
ExpectRegister();
ExpectPuncuation(Comma, RemoveExpected);
ExpectRegister();
PC++;
break;
case SHL:
case SHR:
PC++;
ExpectRegister();
ExpectPuncuation(Comma, RemoveExpected);
if (PeekToken()->Class != NumberClass) {
fprintf(stderr, "[Line %d] Syntax error, expected numeric literal.\n", PeekToken()->LineNumber);
exit(12);
}
AdvanceParser();
PC += 2;
break;
case INC:
case DEC:
case PUSH:
case POP:
PC++;
ExpectRegister();
break;
case JMP:
{
PC++;
Token* arg = PeekToken();
if (arg->Class & IdentifierClass) {
ExpectIdentifier(ForwardParser, 0);
opcode->Value.Mnemonic = JMP;
opcode->Lemexe = "jmp";
PC += 2;
}
else
{
ExpectPuncuation(LBracket, RemoveExpected);
ExpectRegister();
ExpectPuncuation(RBracket, RemoveExpected);
opcode->Value.Mnemonic = JMPI;
opcode->Lemexe = "jmpi";
}
}
break;
case JZ:
{
PC++;
ExpectIdentifier(ForwardParser, 0);
PC += 2;
opcode->Value.Mnemonic = JZ;
opcode->Lemexe = "jz";
}
break;
case JG:
{
PC++;
ExpectIdentifier(ForwardParser, 0);
PC += 2;
opcode->Value.Mnemonic = JG;
opcode->Lemexe = "jg";
}
break;
case JL:
{
PC++;
ExpectIdentifier(ForwardParser, 0);
PC += 2;
opcode->Value.Mnemonic = JL;
opcode->Lemexe = "jl";
}
break;
case CALL:
{
PC++;
Token* arg = PeekToken();
if ((arg->Class & IdentifierClass) == 0) {
fprintf(stderr, "Syntax error on line %d: expected subroutine call target.\n", opcode->LineNumber);
exit(1);
}
ExpectIdentifier(ForwardParser, 0);
PC += 2;
}
break;
default:
break;
}
ExpectLineEndOrFileEnd(ForwardParser);
return opcode;
}
void ExpectPuncuation(TokenPunctuation punctuation, ExpectOptions options) {
Token* token = PeekToken();
if (token->Class != PunctuationClass || token->Value.Punctuation != punctuation) {
fprintf(stderr, "[Line %d] Syntac error, expected %c.\n", token->LineNumber, punctuation);
exit(1);
}
if (options & RemoveExpected) RemoveCurrentToken();
if (options & ForwardParser) AdvanceParser();
}
void ExpectRegister() {
Token* token = PeekToken();
if (token->Class != RegisterClass) {
fprintf(stderr, "[Line %d] Syntax error, expected Register.\n", token->LineNumber);
exit(2);
}
AdvanceParser();
}
Symbol* ExpectIdentifier(ExpectOptions options, int expectUnresolved) {
Token* token = PeekToken();
if (token->Class != IdentifierClass) {
fprintf(stderr, "[Line %d] Expected Identifier.\n", token->LineNumber);
exit(3);
}
Symbol* symbol;
int found = TryGetSymbol(token->Lemexe, SymbolsTable, &symbol);
if (found && TryGetSymbolValue(symbol, SymbolsTable, NULL) && expectUnresolved) {
fprintf(stderr, "[Line %d] Redefinition of symbol '%s'.\n", token->LineNumber, token->Lemexe);
exit(4);
}
if (!found) symbol = AddSymbolToTable(token->Lemexe, PC, SymbolsTable);
if (expectUnresolved) symbol->Token = token;
if (options & RemoveExpected) RemoveCurrentToken();
if (options & ForwardParser) AdvanceParser();
return symbol; return symbol;
} }
void ExpectCharacterClass(Symbol* symbol) {
Token* token = PeekToken();
if (token->Class != CharacterClass) {
fprintf(stderr, "[Line %d] Syntax error, expected character string.\n", token->LineNumber);
exit(5);
}
symbol->Length = strlen(token->Lemexe);
symbol->Token = token;
RemoveCurrentToken(); //Remove the string declared by this DB command.
token = PeekToken();
if (token->EndOfFile) return;
if (token->Class == PunctuationClass && token->Value.Punctuation == NewLine) {
RemoveCurrentToken();
return;
}
ExpectPuncuation(Comma, RemoveExpected);
token = PeekToken();
if (token->Class != NumberClass) {
fprintf(stderr, "[Line %d] Syntax error, expected terminating byte.\n", token->LineNumber);
exit(7);
}
RemoveCurrentToken(); //Remove terminating byte.
//TODO: add the raw value of the byte to the end of the string, but for now just pretend all numbers are zero.
symbol->Length++;
ExpectLineEndOrFileEnd(RemoveExpected);
}
Token* ExpectLineEndOrFileEnd(ExpectOptions options) {
Token* token = PeekToken();
if (token->EndOfFile) return token;
if (token->Class != PunctuationClass || token->Value.Punctuation != NewLine) {
fprintf(stderr, "[Line %d] Syntax error, expected line break.\n", token->LineNumber);
exit(9);
}
if (options & RemoveExpected) RemoveCurrentToken();
if (options & ForwardParser) AdvanceParser();
return token;
}
void PrintSymbols(void) {
char mn[12];
printf("-----SYMBOLS-----\n");
for(int i = 0; i < SymbolsTable->Size; i++) {
Symbol* symbol = SymbolsTable->Symbols[i];
unsigned short value = 0;
int resolved = SymbolResolved(symbol, SymbolsTable);
if (resolved) TryGetSymbolValue(symbol, SymbolsTable, &value);
printf("[%s] %s [Value: %d]", resolved == 0 ? "Unresolved" : "Resolved", symbol->Name, value);
if (symbol->Token && symbol->Token->Class == MnemonicClass)
{
GetMnemonicText(symbol->Token->Value.Mnemonic, mn);
printf(" -> [%s]", mn);
}
if (symbol->Token) printf(" Line %d", symbol->Token->LineNumber);
printf("\n");
}
printf("-----SYMBOLS-----\n");
}
void AdvanceParser(void) {
if (ParserAtEnd()) return;
CurrentToken++;
}
int ParserAtEnd(void) {
if (CurrentToken >= TokensList->size) return 1;
return 0;
}
Token* PeekToken(void) {
return TokensList->content[CurrentToken];
}
void IgnoreParserLine(void) {
while(!ParserAtEnd()) {
if (PeekToken()->Class == PunctuationClass && PeekToken()->Value.Punctuation == NewLine) {
AdvanceParser();
break;
}
if (PeekToken()->EndOfFile) break;
AdvanceParser();
}
}
void RemoveCurrentToken(void) {
RemoveToken(CurrentToken, TokensList);
}
TokenNode* ParseSymbolExpression(void) {
TokenNode* root = CreateTokenNode(ExpectMathOperand(RemoveExpected));
while(!ParserAtEnd()) {
if (PeekToken()->Class == PunctuationClass && PeekToken()->Value.Punctuation == NewLine) {
ExpectPuncuation(NewLine, RemoveExpected);
break;
}
TokenNode* value = CreateTokenNode(ExpectMathOperand(RemoveExpected));
TokenNode* operation = CreateTokenNode(ExpectMathOperator(RemoveExpected));
operation->Left = root;
operation->Right = value;
root = operation;
}
return root;
}
Token* ExpectMathOperator(ExpectOptions options) {
if (ParserAtEnd()) {
fprintf(stderr, "Syntax error, expected math operator but found unexpected end of file.\n");
exit (1);
}
Token* current = PeekToken();
if (current->Class != PunctuationClass) {
fprintf(stderr, "[Line %d] Syntax error, expected math operator.\n", current->LineNumber);
exit (1);
}
switch (current->Value.Punctuation) {
case Plus:
case Minus:
case Star:
case Slash:
case Power:
break;
default:
fprintf(stderr, "[Line %d] Syntax error, expected math operator 2.\n", current->LineNumber);
exit(1);
}
if (options & RemoveExpected) RemoveCurrentToken();
if (options & ForwardParser) AdvanceParser();
return current;
}
Token* ExpectMathOperand(ExpectOptions options) {
if (ParserAtEnd()) {
fprintf(stderr, "Syntax error, expected math operand but found unexpected end of file.\n");
exit (1);
}
Token* token = PeekToken();
if (token->Class != NumberClass && token->Class != IdentifierClass) {
fprintf(stderr, "[Line %d] Syntax error, expected a number or identifier.\n", token->LineNumber);
exit (1);
}
if (options & RemoveExpected) RemoveCurrentToken();
if (options & ForwardParser) AdvanceParser();
return token;
}
+165 -41
View File
@@ -1,13 +1,16 @@
#include "../includes/scanner.h" #include "../includes/scanner.h"
#include <ctype.h>
#include <stdlib.h> #include <stdlib.h>
#include <string.h> #include <string.h>
#include <limits.h>
const char* SourceCode; const char* SourceCode;
int Line = 0; int Line = 1;
int Position = 0; int Position = 0;
int SourceLength = 0; int SourceLength = 0;
int ScannerAtEnd(void); int ScannerAtEnd(void);
int IsPunctuation(char); int IsPunctuation(char);
int IsWhiteSpace(char);
char PeekScanner(void); char PeekScanner(void);
char PeekAheadScanner(void); char PeekAheadScanner(void);
void AdvanceScanner(void); void AdvanceScanner(void);
@@ -18,8 +21,8 @@ Token* ParseNumber(void);
Token* ParsePunctuation(char); Token* ParsePunctuation(char);
void IgnoreLine(void); void IgnoreLine(void);
List* GenerateTokenList(const char* source) { TokenList* GenerateTokenList(const char* source) {
List* tokens = CreateList(); TokenList* tokens = CreateTokenList();
Token* token = NULL; Token* token = NULL;
if (!tokens) return NULL; if (!tokens) return NULL;
@@ -34,39 +37,60 @@ List* GenerateTokenList(const char* source) {
case ' ': case ' ':
case '\r': case '\r':
case '\t': case '\t':
case '\v':
case '\f':
AdvanceScanner(); AdvanceScanner();
break; //Ignore whitespace break; //Ignore whitespace
case '\n': case '\n':
if (tokens->size > 0) {
token = tokens->content[tokens->size - 1];
if (token->Class == PunctuationClass && token->Value.Punctuation == NewLine) {
AdvanceScanner();
Line++;
continue;
}
}
token = CreateToken(Line, PunctuationClass);
token->Value.Punctuation = NewLine;
AddToken(token, tokens);
AdvanceScanner(); AdvanceScanner();
Line++; Line++;
break; break;
case ';': case ';':
IgnoreLine(); IgnoreLine();
if (tokens->size == 0) {
//There's nothing here so that means this is some comments block at the start of the file.
AdvanceScanner(); //Consume the actual new line char.
Line++;
}
break; break;
case '.': //directive like ".org" or ".db" case '.': //directive like ".include" or ".db"
token = ParseDirective(); token = ParseDirective();
if (token) AddListItem(token, sizeof(Token), tokens); if (token) AddToken(token, tokens);
break; break;
case '"': case '"':
token = ParseString(); token = ParseString();
if (token) AddListItem(token, sizeof(Token), tokens); if (token) AddToken(token, tokens);
break; break;
default: default:
if (isdigit(c)) { if (isdigit(c)) {
AddListItem(ParseNumber(), sizeof(Token), tokens); AddToken(ParseNumber(), tokens);
break; break;
} }
if (IsPunctuation(c)) { if (IsPunctuation(c)) {
AdvanceScanner(); AdvanceScanner();
AddListItem(ParsePunctuation(c), sizeof(Token), tokens); AddToken(ParsePunctuation(c), tokens);
break; break;
} }
token = ParseIdentifier(); token = ParseIdentifier();
if (token) AddListItem(token, sizeof(Token), tokens); if (token) AddToken(token, tokens);
break; break;
} }
@@ -74,20 +98,45 @@ List* GenerateTokenList(const char* source) {
token = NULL; token = NULL;
} }
if (tokens->size > 0) {
token = tokens->content[tokens->size - 1];
if (token->Class == PunctuationClass && token->Value.Punctuation == NewLine) {
//If the last token is a line break, remove it as its not too meaningful.
tokens->size--;
}
}
token = CreateToken(Line, PunctuationClass);
token->EndOfFile = 1;
AddToken(token, tokens);
return tokens; return tokens;
} }
Token* ParseNumber(void) { Token* ParseNumber(void) {
int start = Position; int start = Position;
int base = 10;
while(!ScannerAtEnd() && isdigit(PeekScanner())) while(!ScannerAtEnd() && isdigit(PeekScanner()))
AdvanceScanner(); AdvanceScanner();
if (PeekScanner() == 'x' && isdigit(PeekAheadScanner())) { if (tolower(PeekScanner()) == 'x') {
base = 16; char ahead = tolower(PeekAheadScanner());
if ((ahead >= 'a' && ahead <= 'f') || isdigit(ahead)) {
AdvanceScanner(); //Consume the 'x' AdvanceScanner(); //Consume the 'x'
while(!ScannerAtEnd() && isdigit(PeekScanner())) AdvanceScanner();
while(!ScannerAtEnd() && PeekScanner() != '\n' && !IsWhiteSpace(PeekScanner()) && PeekScanner() != ';') {
if (isdigit(PeekScanner())) {
AdvanceScanner();
continue;
}
if (tolower(PeekScanner()) >= 'a' && tolower(PeekScanner()) <= 'f') AdvanceScanner();
}
}
} }
int length = Position - start; int length = Position - start;
@@ -95,26 +144,26 @@ Token* ParseNumber(void) {
if (length == 0) return NULL; if (length == 0) return NULL;
char* lexeme = calloc(sizeof(char), length + 1); char* lexeme = calloc(sizeof(char), length + 1);
long* value = calloc(1, sizeof(long));
if (!lexeme) { if (!lexeme) {
fprintf(stderr, "Failed to calloc space for number. %s.\n", strerror(errno)); fprintf(stderr, "Failed to calloc space for number. %s.\n", strerror(errno));
return NULL; return NULL;
} }
if (!value) {
free(lexeme);
fprintf(stderr, "Failed to calloc space for the raw numeric value of a token. %s.\n", strerror(errno));
return NULL;
}
memcpy(lexeme, &SourceCode[start], length); memcpy(lexeme, &SourceCode[start], length);
*value = strtol(lexeme, NULL, base); Token* token = CreateToken(Line, NumberClass);
if (base == 10) return CreateToken(lexeme, value, Line, NUMBER); token->Value.Number = strtol(lexeme, NULL, 0);
return CreateToken(lexeme, value, Line, HEX); if (errno != 0) {
fprintf(stderr, "[Error] Line %d: Invalid number detected. %s.\n", Line, strerror(errno));
exit(1);
}
token->Lemexe = lexeme;
return token;
} }
Token* ParseDirective(void) { Token* ParseDirective(void) {
@@ -135,19 +184,23 @@ Token* ParseDirective(void) {
memcpy(directive, &SourceCode[start], length); memcpy(directive, &SourceCode[start], length);
Token* token = CreateToken(Line, DirectiveClass);
if (strcmp(directive, ".db") == 0) { if (strcmp(directive, ".db") == 0) {
return CreateToken(directive, directive, Line, DB); token->Value.Directive = DB;
return token;
} }
else if (strcmp(directive, ".org") == 0) { else if (strcmp(directive, ".include") == 0) {
fprintf(stderr, "[Warning] Org is not a supported directive.\n"); token->Value.Directive = Include;
free(directive);
IgnoreLine(); return token;
return NULL;
} }
fprintf(stderr, "[Error] Unknown assembler directive '%s'\n", directive); fprintf(stderr, "[Error] Unknown assembler directive '%s'\n", directive);
free(directive); free(directive);
free(token);
IgnoreLine(); IgnoreLine();
@@ -180,7 +233,9 @@ Token* ParseString(void) {
AdvanceScanner(); //Consume the trailing double quote. AdvanceScanner(); //Consume the trailing double quote.
Token* token = CreateToken(lexeme, lexeme, Line, STRING); Token* token = CreateToken(Line, CharacterClass);
token->Lemexe = lexeme;
return token; return token;
} }
@@ -188,7 +243,7 @@ Token* ParseString(void) {
Token* ParseIdentifier(void) { Token* ParseIdentifier(void) {
int start = Position; int start = Position;
while(!ScannerAtEnd() && !IsPunctuation(PeekScanner()) && PeekScanner() != ' ' && PeekScanner() != '\n') { while(!ScannerAtEnd() && !IsPunctuation(PeekScanner()) && PeekScanner() != ' ' && PeekScanner() != '\n' && PeekScanner() != ';') {
AdvanceScanner(); AdvanceScanner();
} }
@@ -196,7 +251,8 @@ Token* ParseIdentifier(void) {
if (length == 0) return NULL; if (length == 0) return NULL;
TokenType type; Mnemonic mnemonics;
Registers reg;
char* lexeme = calloc(sizeof(char), length + 1); char* lexeme = calloc(sizeof(char), length + 1);
if (!lexeme) { if (!lexeme) {
@@ -206,14 +262,41 @@ Token* ParseIdentifier(void) {
memcpy(lexeme, &SourceCode[start], length); memcpy(lexeme, &SourceCode[start], length);
if (IsOpcode(lexeme, &type)) return CreateToken(lexeme, lexeme, Line, type); if (IsOpcode(lexeme, &mnemonics)) {
if (IsRegister(lexeme, &type)) return CreateToken(lexeme, lexeme, Line, type); Token* token = CreateToken(Line, MnemonicClass);
token->Value.Mnemonic = mnemonics;
return token;
}
if (IsRegister(lexeme, &reg)) {
Token* token = CreateToken(Line, RegisterClass);
token->Value.Register = reg;
return token;
}
if (lexeme[length - 1] == ':') { if (lexeme[length - 1] == ':') {
lexeme[length - 1] = '\0'; //Bit hacky, but this makes the parser's job a bit easier. lexeme[length - 1] = '\0'; //Bit hacky, but this makes the parser's job a bit easier.
return CreateToken(lexeme, lexeme, Line, LABEL); Token* token = CreateToken(Line, LabelClass);
token->Lemexe = lexeme;
return token;
}
if (strcmp(lexeme, "byte") == 0) {
Token* token = CreateToken(Line, DirectiveClass);
token->Lemexe = lexeme;
token->Value.Directive = Byte;
return token;
} }
return CreateToken(lexeme, lexeme, Line, IDENTIFIER); Token* token = CreateToken(Line, IdentifierClass);
token->Lemexe = lexeme;
return token;
} }
char PeekScanner(void) { char PeekScanner(void) {
@@ -245,6 +328,23 @@ int IsPunctuation(char c) {
case '(': case '(':
case ')': case ')':
case ',': case ',':
case '-':
case '+':
case '*':
case '/':
return 1;
default:
return 0;
}
}
int IsWhiteSpace(char c) {
switch(c) {
case ' ':
case '\t':
case '\v':
case '\f':
case '\r':
return 1; return 1;
default: default:
return 0; return 0;
@@ -252,18 +352,42 @@ int IsPunctuation(char c) {
} }
Token* ParsePunctuation(char c) { Token* ParsePunctuation(char c) {
TokenPunctuation punctuation;
switch(c) { switch(c) {
case '[': case '[':
return CreateToken("[", NULL, Line, LBRACKET); punctuation = LBracket;
break;
case ']': case ']':
return CreateToken("]", NULL, Line, RBracket); punctuation = RBracket;
break;
case '(': case '(':
return CreateToken("(", NULL, Line, LPARAM); punctuation = LParan;
break;
case ')': case ')':
return CreateToken(")", NULL, Line, RPARAM); punctuation = RParan;
break;
case ',': case ',':
return CreateToken(",", NULL, Line, COMMA); punctuation = Comma;
break;
case '-':
punctuation = Minus;
break;
case '+':
punctuation = Plus;
break;
case '*':
punctuation = Star;
break;
case '/':
punctuation = Slash;
break;
default: default:
return NULL; return NULL;
} }
Token* token = CreateToken(Line, PunctuationClass);
token->Value.Punctuation = punctuation;
return token;
} }
+202
View File
@@ -0,0 +1,202 @@
#include "../includes/symbols_table.h"
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
int SymbolResolved(const Symbol* symbol, const SymbolTable* table) {
if (!symbol) return 0;
const Token* token = symbol->Token;
if (!token) return 0;
if (token->Class == LabelClass || token->Class == CharacterClass || token->Class == MnemonicClass) return 1;
return TryGetSymbolValue(symbol, table, NULL);
}
SymbolTable* CreateSymbolTable(void){
SymbolTable* table = calloc(1, sizeof(SymbolTable));
if (!table) {
fprintf(stderr, "Failed to calloc memory for a SymbolTable. %s.\n", strerror(errno));
return NULL;
}
table->Symbols = calloc(SYMBOLSTABLE_DEFAULT_CAPACITY, sizeof(Symbol*));
if (!table->Symbols) {
fprintf(stderr, "Failed to calloc Symbol list. %s.\n", strerror(errno));
free(table);
return NULL;
}
table->Capacity = SYMBOLSTABLE_DEFAULT_CAPACITY;
table->Size = 0;
return table;
}
Symbol* CreateSymbol(const char* name, int address) {
Symbol* symbol = calloc(1, sizeof(Symbol));
if (!symbol) {
fprintf(stderr, "Failed to create Symbol '%s'. %s.\n", name, strerror(errno));
return NULL;
}
symbol->Address = address;
symbol->Name = name;
//symbol->Type = RefUnknown;
return symbol;
}
int TryGetSymbol(const char* name, const SymbolTable* table, Symbol** outSymbol) {
*outSymbol = NULL;
if (!name || !table) return 0;
for(int i = 0; i < table->Size; i++) {
if (strcmp(table->Symbols[i]->Name, name) == 0) {
*outSymbol = table->Symbols[i];
return 1;
}
}
return 0;
}
Symbol* AddSymbolToTable(const char* name, int address, SymbolTable* table) {
//if (!name || !value || !table || length == 0) return NULL;
for(int i = 0; i < table->Size; i++) {
if (strcmp(table->Symbols[i]->Name, name) == 0) {
//TODO: Do we update or throw some kind of an error?
return table->Symbols[i];
}
}
if (table->Capacity < table->Size + 1) {
Symbol** newBlock = realloc(table->Symbols, sizeof(Symbol*) * table->Capacity * 2);
if (!newBlock) {
fprintf(stderr, "Failed to realloc space for a new symbol '%s'. %s.\n", name, strerror(errno));
return NULL;
}
table->Capacity *= 2;
table->Symbols = newBlock;
}
Symbol* symbol = CreateSymbol(name, address);
table->Symbols[table->Size] = symbol;
table->Size++;
return symbol;
}
void FreeSymbolTable(SymbolTable* table) {
if (!table) return;
for(int i = 0; i < table->Size; i++) FreeSymbol(table->Symbols[i]);
free(table);
}
void FreeSymbol(Symbol* symbol) {
if (!symbol) return;
free(symbol);
}
int TryGetTokenNodeValue(const TokenNode* node, const SymbolTable* table, unsigned short* value);
unsigned short DoOp(unsigned short left, unsigned short right, TokenPunctuation op);
int TryGetSymbolValue(const Symbol* symbol, const SymbolTable* table, unsigned short* value) {
if (!symbol || !symbol->Token || !table) return 0;
TokenNode* root = symbol->ValueExpression;
Symbol* s = NULL;
if (value) *value = 0;
if (!root || !root->Token) return 0;
switch (root->Token->Class) {
case NumberClass:
if (value) *value = root->Token->Value.Number;
return 1;
case IdentifierClass:
if (!TryGetSymbol(root->Token->Lemexe, table, &s)) return 0;
return TryGetTokenNodeValue(s->ValueExpression, table, value);
case PunctuationClass:
return TryGetTokenNodeValue(root, table, value);
default:
return 0;
}
}
int TryGetTokenNodeValue(const TokenNode* node, const SymbolTable* table, unsigned short* value) {
const Token* token = node->Token;
Symbol* symbol = { 0 };
if (token->Class == NumberClass) {
if (value) *value = token->Value.Number;
return 1;
}
if (token->Class == IdentifierClass) {
if (!TryGetSymbol(token->Lemexe, table, &symbol)) return 0;
return TryGetTokenNodeValue(symbol->ValueExpression, table, value);
}
if (token->Class != PunctuationClass) return 0;
unsigned short left = 0;
unsigned short right = 0;
if (!TryGetTokenNodeValue(node->Left, table, &left)) return 0;
if (!TryGetTokenNodeValue(node->Right, table, &right)) return 0;
if (value) *value = DoOp(left, right, token->Value.Punctuation);
return 1;
}
TokenNode* CreateTokenNode(const Token* token){
TokenNode* node = calloc(1, sizeof(TokenNode));
if (!node) {
fprintf(stderr, "Failed to calloc memory for a TokenNode. %s.\n", strerror(errno));
return NULL;
}
node->Token = token;
return node;
}
unsigned short DoOp(unsigned short left, unsigned short right, TokenPunctuation op) {
switch (op) {
case Plus:
return left + right;
case Minus:
return left - right;
case Star:
return left * right;
case Slash:
return left / right;
default:
break;
}
return 0;
}
+68 -7
View File
@@ -1,6 +1,8 @@
#include "../includes/token.h" #include "../includes/token.h"
Token* CreateToken(char* lexeme, void* value, int lineNumber, TokenType type) { #define LISTDEFAULTSIZE 32
Token* CreateToken(int lineNumber, TokenClass tokenClass) {
Token* token = calloc(1, sizeof(Token)); Token* token = calloc(1, sizeof(Token));
if (!token) { if (!token) {
@@ -8,10 +10,8 @@ Token* CreateToken(char* lexeme, void* value, int lineNumber, TokenType type) {
return NULL; return NULL;
} }
token->type = type; token->Class = tokenClass;
token->line = lineNumber; token->LineNumber = lineNumber;
token->lexeme = lexeme;
token->value = value;
return token; return token;
} }
@@ -19,7 +19,68 @@ Token* CreateToken(char* lexeme, void* value, int lineNumber, TokenType type) {
void FreeToken(Token* token) { void FreeToken(Token* token) {
if (!token) return; if (!token) return;
if (token->value && token->type >= STRING) free(token->value);
free(token); free(token);
} }
TokenList* CreateTokenList(void) {
TokenList *new = malloc(sizeof(TokenList));
if (!new) {
fprintf(stderr, "Failed to malloc() for new new List.\n");
return NULL;
}
new->content = calloc(LISTDEFAULTSIZE, sizeof(void*));
if (!new->content) {
fprintf(stderr, "Failed to malloc() memory for List contents.\n");
free(new);
return NULL;
}
new->size = 0;
new->capacity = LISTDEFAULTSIZE;
return new;
}
int AddToken(Token* token, TokenList* list) {
if (!list || !token) return 0;
if (list->capacity < list->size + 1) {
void* ptr = realloc(list->content, sizeof(void*) * list->capacity * 2);
//Note: realloc will free list->root if it succeeds.
if (!ptr) {
fprintf(stderr, "Failed to resize array with realloc() (%d bytes).\n", list->capacity * 2);
return 0;
}
list->content = ptr;
list->capacity *= 2;
}
if (list->size > 0)
{
Token* prev = list->content[list->size - 1];
token->Prev = prev;
prev->Next = token;
}
list->content[list->size] = token;
list->size++;
return 1;
}
void RemoveToken(int index, TokenList* list) {
Token* token = list->content[index];
if (token->Prev) token->Prev->Next = token->Next;
memmove(&list->content[index], &list->content[index + 1], (list->size - index) * sizeof(Token*));
list->size--;
list->content[list->size] = NULL;
}