Compare commits

73 Commits
Author SHA1 Message Date
glm94 c95d739e16 Updated the parser to handle treating variables as expressions. Updated the symbols table to handle computing the values of variables. 2023-09-07 21:29:58 -05:00
glm94 21c2bf046e Wrote the very first part of the tree walking code, more just to make sure the logic is sound. 2023-09-01 10:04:21 -05:00
glm94 b810d7803c Updated the parser to create an expression tree for symbols declared via .db directives. Scanner now see some math operators as 'punctuation'. 2023-08-31 23:27:33 -05:00
glm94 9e178ca3da Build a quick draft of how the symbol value should be computed. 2023-08-25 01:27:22 -05:00
glm94 82b2136a69 Started to work on setting up Symbol value resolution which will be based on an abstract syntax tree that can handle undefined values if unknown symbols are used in an expression. 2023-08-25 00:21:19 -05:00
glm94 4bb90da6df Make the compiler options a bit more strict about C compliance (in other words I just learned about these flags and they seem to be ideal for this project). 2023-08-25 00:15:18 -05:00
glm94 4af2527806 Updated the structure of symbols and tokens to hopefully help make computing symbol values easier. 2023-08-07 01:18:01 -05:00
glm94 65383cded2 Updated the way symbols are represented and how they are marked as resolved. 2023-07-09 22:04:55 -05:00
glm94 9c8fe17c8f Minor code cleanup and added an extra flag to the compiler. If a symbol is a label it will now hold a reference to the opcode and by extension the offset into the file it points to. 2023-07-06 19:39:03 -05:00
glm94 3031fa8b9b Code cleanup. 2023-06-29 20:53:39 -05:00
glm94 2918f2964d Fixed a bug where words weren't being written to 'memory' correctly. 2023-06-24 18:00:13 -05:00
glm94 980b0449fc Fixed a bug with the parse discarding identifier tokens when parsing a COPY instruction. Added the first batch of code to encode the final token array to a binary form. 2023-06-22 20:22:20 -05:00
glm94 6741dde7ce Hooked up the program counter in the parser to set the memory offset of symbols, well labels to be more acurrate. 2023-06-18 14:51:37 -05:00
glm94 c6e6e2c5dc Fixed a bug in the parser not setting an identifier when one is used with a CMP and fixed a bug in the example assembly program. 2023-06-18 12:11:30 -05:00
glm94 0eaa5527ea Updated the parser to be able to handle the changes made to the ISA. 2023-06-18 00:50:18 -05:00
glm94 403439def4 Updated the opcode listing and 'fixed' the code to at least compile, but obviously I'll need to rework how the new copy instructions are to be parsed. Fixed some errors in the documentation where I skipped some hex numbers and added a copy immediate op. 2023-06-17 21:15:08 -05:00
glm94 0fd6413d81 Updated the instruction set and mapped out the opcodes as well as wrote a sample program to see if the ISA was complete enough to build a trivial program. 2023-06-17 16:56:42 -05:00
glm94 a9e57132be Fixed issue #1 Scanner Bug where the start of comment would become part of an identifier or throw off number parsing. 2023-04-14 13:28:44 -05:00
glm94 aa3888b578 Fixed up the RemoveCurrentToken function which was causing all sorts of bugs, and added the ability to delcare a data byte, '.db', to have a variable's address as a value. This will be needed for building up an interrupt table at least with the version I have in mind. 2023-04-04 22:14:17 -05:00
glm94 c3d12da54a Added some ideas to the encoding scheme to be used by the processor. 2023-03-30 13:21:25 -05:00
glm94 288ba9f9cb Updated the docs to include new jump operands and setup the assembler to handle parsing all the jump instructions. 2023-03-29 22:30:12 -05:00
glm94 a9a50055a4 Updated the parser to validate most of the opcodes. NOP falls through via the default case since it takes no parameters but the jump instructions need some thought before proceeding. 2023-03-29 20:57:04 -05:00
glm94 2e888913be Added support for the STORE directive. 2023-03-29 12:57:07 -05:00
glm94 ee60d14570 Refactored the code and added expect functions to make the code flow a bit nicer. 2023-03-29 12:41:53 -05:00
glm94 fa12a04bb3 Moved some of the grunt work to their own functions to clean the code up a bit. 2023-03-28 22:14:31 -05:00
glm94 2d7aad1617 Fixed a bug with labels not advancing the parser after being processed resulting in a redefinition error. 2023-03-28 12:41:56 -05:00
glm94 ae03991fa5 Fixed some parser bugs processing the syntax for some LOAD instruction types. 2023-03-27 22:22:01 -05:00
glm94 95ed2143ea Updated the parser to handle processing a simple load byte instruction. Added the LODB opcode to the opcode enum listing and updated the Scanner to look for 'byte' and mark it as a directive. 2023-03-24 23:24:24 -05:00
glm94 f96e1c6380 Added grammar checking for DB declarations. 2023-03-08 21:38:33 -06:00
glm94 4f9f4202b2 Wrote the first batch of directive handling code, so far just handling creating variables. 2023-03-08 13:25:46 -06:00
glm94 4230708f15 Added a function to remove an item from the generic list, which is mainly used for the tokens. 2023-03-07 21:31:14 -06:00
glm94 5e9e1bdec4 Massive refactoring to simplify this whole setup. The Token List will be modified in place and reused in the Parser to normalize the list and generate a symbols table. Normalizing the token list will make sure the syntax is valid and the symbols table will have the relative offsets and length of symbol values. At least this is all the plan but one step at a time. 2023-03-06 21:51:36 -06:00
glm94 1b246e90bc Started rethinking how this machine should behave. Updated and refactoring some things with a few new ideas. 2023-02-23 22:22:07 -06:00
glm94 9a602824b6 Added all the instructions supported to a structure to help make validation of parameters easier. 2022-10-18 21:14:50 -05:00
glm94 6d6bf1cbaf Wrote the first step for the binary output. Doesn't handle instruction pointer symbols yet since those emit addresses that still need to be computed. 2022-10-10 18:36:38 +00:00
glm94 bca2ad67f1 Fixed a bug with single parameter instructions that land at the end of a file expecting a comma. 2022-10-08 22:02:51 -05:00
glm94 fd44e03b1b Updated the Parser to try and give 'AddressParameter's more meaning depending on the SymbolType. 2022-10-08 18:21:11 -05:00
glm94 48a77950be Updated the way string symbols are handled, allowing the programmer to terminate a string with any byte or none at all. 2022-10-08 16:35:05 -05:00
glm94 8b6edc39be Updated the Parser to handle the variable declaration directive. 2022-10-08 16:15:10 -05:00
glm94 c142d2c6d0 Cleaned up the Symbol struct and updated code accordingly. 2022-10-08 14:43:00 -05:00
glm94 4b08707e67 Cleaned up the list implementation so as to NOT copy items into a new buffer. 2022-10-07 20:46:55 +00:00
glm94 bdb4d2b241 Updated the way Instructions keep track of symbols. Only their name is important but we need the Symbols to be able to track which Instruction they point to (if they do that is, i.e. a label symbol). 2022-10-06 21:11:22 -05:00
glm94 d53833c25f Added some debug print out for the instruction list and a helper function to get a string for any Registers enum. 2022-10-06 21:05:57 +00:00
glm94 f38da9cc74 A bit of cleanup. 2022-10-04 21:05:15 +00:00
glm94 33309a2759 Updated the Instruction's structure and updated the Parser accordingly. 2022-10-04 20:37:47 +00:00
glm94 1f7989533b Fixed a bug in the SymbolsTable where it wouldn't update its size. 2022-10-03 20:11:04 -05:00
glm94 cd1258a724 Fixed various Parser bugs. 2022-10-03 19:58:44 -05:00
glm94 be91e3c112 Updated the parser to better handle parameters, sort of and including a string length assembly program to use as a test for the whole assembler. 2022-10-03 21:26:28 +00:00
glm94 706f480e31 The Scanner will now check for empty lines (is the previous token and the current token a new line?) and simply not emit a NewLine Token. 2022-10-03 19:40:00 +00:00
glm94 a949007c75 Updated the ISA so the program will correctly see things like 'load', 'inc' and 'dec'. 2022-10-03 16:56:39 +00:00
glm94 a0d5d62a34 Fixed a bug in the Scanner when parsing a hex number, still not super robust but it'll work. 2022-10-03 15:20:31 +00:00
glm94 f7cd87f13f Started reworking the Parser to simplify how instructions will be represented. It will act like a 'first pass' that will do grammar checks but not verify the parameters of the opcodes. 2022-10-02 22:54:18 -05:00
glm94 55f4c32764 Missed this for the Scanner fix. 2022-10-02 22:12:16 -05:00
glm94 097eca383b Fixed a bug with the Scanner adding a blank line via a single NewLine Token when a file starts with a block of comments. 2022-10-01 20:43:55 -05:00
glm94 984683bfc5 Minor adjustment to Figure 1.1 2022-09-29 22:57:12 -05:00
glm94 0f42ad2997 Updated the ISA and added links in the table of contents. 2022-09-29 22:11:19 -05:00
glm94 63438b551b Got the scanner running again. Seems to be picking up punctuation, labels, identifiers and strings as expected. 2022-09-29 21:32:08 -05:00
glm94 e94e486e18 More refinements to the loading data instructions. I am truly bad at this whole thing... 2022-09-29 20:42:37 +00:00
glm94 2b71ac05a0 Added a note about a proposal regarding assembly language design. 2022-09-27 17:46:50 +00:00
glm94 02cf72902e Updated the load instructions. 2022-09-26 22:13:48 -05:00
glm94 afa84076c8 Added a command to make generating an opcode table a one liner. Updated the ISA. 2022-09-23 19:36:27 +00:00
glm94 925f7703d6 Updated the ISA, hopefully I can get this all put together in a thought out way. 2022-09-22 20:49:20 +00:00
glm94 a82a4c23ad Updated the docs. 2022-09-21 21:44:38 -05:00
glm94 f5c82d1c36 Added a LaTex document to put in writing how the machine should behave. 2022-09-21 16:23:11 +00:00
glm94 0b58e0124f Some refactoring to update everything to use the new structures and some new considerations for how Symbols and Instructions shoudl be represented. 2022-09-15 22:14:38 -05:00
glm94 2ae60a607c Incomplete, but I want to make sure the ideas I have here don't get wiped. Sadly this commit won't compile. 2022-09-15 21:29:27 +00:00
glm94 6d17dffcdf Added a SymbolTable object (untested at the moment) and redefined the Symbol object. I think when a Symbol is made only the size of it should matter to the code that will assemble the final binary. 2022-09-08 20:56:02 +00:00
glm94 2518214585 Added a length attribute to the symbol. 2022-09-06 14:15:45 +00:00
glm94 7004fff659 This feels like a trainwreck but eh. Changed the way the opcodes are managed. Hopefully this is the right direction when I add support for multiple ASM files. 2022-09-01 18:46:59 +00:00
glm94 13f8ff194b Smoothbrain indeed... 2022-08-30 22:48:49 -05:00
glm94 dac01ffd0c Changed the TokenClass from opcode to nomic, a less smoothbrain name IMO and also frees up Opcode for a better use later on. 2022-08-30 22:06:50 -05:00
glm94 3c7056c5e6 The main function will now write out the output of the parser. 2022-08-29 21:23:02 +00:00
glm94 6c7b8d5356 Fixed a bug where CMP REG, REG wouldn't get encoded and a disassembler bug related to said CMP bug. 2022-08-29 19:29:34 +00:00
20 changed files with 1643 additions and 784 deletions
+4 -1
View File
@@ -1,3 +1,6 @@
assm assm
obj/ obj/
bin/ bin/
*.bin
docs/*
!docs/*.tex
+3 -3
View File
@@ -1,5 +1,5 @@
CC = gcc CC = gcc
CFLAGS=-g -Wall -DDEBUG -Wpedantic CFLAGS=-g -Wall -DDEBUG -Wpedantic -Wextra -Wunused-result -std=c99 -pedantic-errors
SRCDIR=src SRCDIR=src
OBJDIR=obj OBJDIR=obj
SRCS=$(wildcard $(SRCDIR)/*.c) SRCS=$(wildcard $(SRCDIR)/*.c)
@@ -11,7 +11,7 @@ BIN=$(BINDIR)/assm
all: $(BIN) all: $(BIN)
release: CFLAGS=-Wall -Wpedantic -O2 release: CFLAGS=-Wall -Wpedantic -std=c99 -pedantic-errors -O2
release: clean release: clean
release: $(BIN) release: $(BIN)
@@ -35,7 +35,7 @@ clean:
rm -rf $(BINDIR)/* $(OBJDIR)/* rm -rf $(BINDIR)/* $(OBJDIR)/*
test: test:
$(BIN) misc/test.asm $(BIN) misc/another_test.asm
disass: disass:
objdump -S --disassemble $(OBJDIR)/$(FILE).o > $(OBJDIR)/$(FILE).s objdump -S --disassemble $(OBJDIR)/$(FILE).o > $(OBJDIR)/$(FILE).s
+187
View File
@@ -0,0 +1,187 @@
\documentclass[a4paper,12pt]{book}
\usepackage{tikz}
\usepackage{hyperref}
\hypersetup{
linktoc=all
}
\title{Unnamed Machine}
\author{A Very Terrible 16-bit Machine}
\newcommand{\OpcodeTable}[5] {
\begin{tabular}{ c c c c c }
\hline
Opcode & Bit Pattern & Mnemonic & Operand 1 & Operand 2 \\
\hline\hline
#1 & #2 & #3 & #4 & #5 \\
\hline
\end{tabular}
}
\begin{document}
\maketitle
\tableofcontents
\chapter{Overview}
\section{Introduction}
This is a very poorly thought out 16-bit machine, but you've got to start somewhere. Currently debating between a CPU status register, 80x86 style, or just placing arithmetic results into a predetermined register. This is more of a loadstore architecture to try and keep the instruction set simple. The machine will be big-endian.
\section{Registers}
The following are the general purpose registers that can be used.
\begin{itemize}
\item[] R1
\item[] ...
\item[] R8
\end{itemize}
Additionally, there will be a special status register that will be a signed 16 bit register that the compare and jump instructions will read or write to when determining what action, if any, they'll take.
\section{Memory Model}
Memory will be implicitly mapped I/O. The bottom 3201 bytes of memory will be reserved for the keyboard input and graphics.
A single byte is reserved for the keyboard's input. The current key will be stored in byte 0xF37E, with the most significant bit being a flag indicating that the keyboard
is ready to be read from. This means that the character encoding is actually 7 bits.
Video memory starts at 0xF37F (62335 decimal), and every byte represents an ASCII character in monochrome.
\begin{figure}[!htb]
\centering
\begin{tikzpicture}
\fill[gray!5] (0,0)rectangle(5,10);
%\draw (0,10) .. controls (-2,6) and (-2,4) .. (0,1);
%\draw (0,10) arc (0:180:3cm);
\draw (0,10) -- (5,10);
\draw (0,1) -- node[above] {Video 0xF37F} (5,1);
\draw (0,0) -- (5,0);
\node[label=right:Top 0x0000] at (5,10) {};
\node[label=right:Bottom 0xFFFF] at (5,0) {};
\end{tikzpicture}
\caption{Memory Layout}
\end{figure}
\chapter{Instruction Set Architecture}
\section{Instruction Encoding}
The instruction encoding is fixed width to exactly 8 bites wide.
The encoding for the instructions will differ depending on if the opcode requires a register as its first operand.
For instructions that require, as their first argument, a register the encoding format will be as shown in Figure \ref{fig:WithRegister}.
The lower 3 bits may be any bit pattern as long as at least one bit is set in the upper 5 bits.
%https://tex.stackexchange.com/questions/32598/force-latex-image-to-appear-in-the-section-in-which-its-declared
\begin{figure}[!htb]
\centering
\begin{tabular}{ c c }
Instruction & Register \\
\hline
XXXX X & 000 \\
\hline
\end{tabular}
\caption{Encoding Layout, Register Required}
\label{fig:WithRegister}
\end{figure}
Figure \ref{fig:NoRegisterEncoding} shows how encoding will look for instructions that lack arguments or do not require a register.
In contrast with the previous encoding scheme the upper 5 bits MUST be zero, allowing 7 possible instructions to use this format.
All zeroes is not considered a legal instruction.
\begin{figure}[!htb]
\centering
\begin{tabular}{ c c }
Must Be Zero & Instruction \\
\hline
0000 0 & XXX \\
\hline
\end{tabular}
\caption{Encoding Layout, No Register Required}
\label{fig:NoRegisterEncoding}
\end{figure}
\section{Notes}
For the opcodes that load or store data at the assembly language level we could have the mnemonics
"store" and "load" and have the assembler pick the opcode based on the inclusion of the word "byte"
or "word" for two bytes. That would make the assembly easier to read but put a bit more work on the
assembler. The Stack Pointer will start 2 bytes above the video memory start.
\section{COPY (Copy Word from Address)}
\OpcodeTable{0x08}{0000 1000}{copya}{Register}{Address}\\[6pt]
Copy a machine word from Operand 2 into Operand 1.
\section{COPY BYTE (Copy Byte from Address)}
\OpcodeTable{0x10}{0001 0000}{copyab}{Register}{Address}\\[6pt]
Copy a byte (8 bits) from Operand 2 into Operand 1, clearing the setting the most significant bits to zero.
\section{COPY (Copy Word Indirect Address)}
\OpcodeTable{0x18}{0001 1000}{copyra}{Register}{[Register]}\\[6pt]
Copy a machine word from the address stored in Operand 2 into Operand 1.
\section{COPY BYTE (Copy Byte Indirect Address)}
\OpcodeTable{0x20}{0010 0000}{copyrab}{Register}{[Register]}\\[6pt]
Copy a byte (8 bits) from the address stored in Operand 2 into Operand 1.
\section{COPY (Copy)}
\OpcodeTable{0x28}{0010 1000}{copyrara}{[Register]}{[Register]}\\[6pt]
Copy a machine word from the address stored in Operand 2 into the address stored in Operand 1.
\section{COPY BYTE (Copy Byte)}
\OpcodeTable{0x30}{0011 0000}{copyrarab}{[Register]}{[Register]}\\[6pt]
Copy a byte (8 bits) from the address stored in Operand 2 into the address stored in Operand 1.
\section{COPY (Copy)}
\OpcodeTable{0x38}{0011 1000}{copy}{Register}{Register}\\[6pt]
Copy a machine word from Operand 2 into Operand 1.
\section{COPY (Copy Immediate)}
\OpcodeTable{0xC0}{1100 0000}{copyi}{Register}{Constant}\\[6pt]
Copy a machine word from Operand 2 into Operand 1.
\section{COPY BYTE (Copy Byte)}
\OpcodeTable{0x40}{0100 0000}{copyb}{Register}{Constant}\\[6pt]
Copy a byte (8 bits) from Operand 2 into Operand 1.
\section{CMP (Compare)}
\OpcodeTable{0x48}{0100 1000}{cmp}{Register}{Register}\\[6pt]
Compares two registers and somewhere sets a result in the status register.
\section{CMPI (Compare Immediate)}
\OpcodeTable{0x50}{0101 0000}{cmpi}{Register}{Constant}\\[6pt]
Compares an immediate 2 byte value to the contents of a register setting the status register accordingly.
\section{ADD (Add)}
\OpcodeTable{0x58}{0101 1000}{add}{Register}{Register}\\[6pt]
Performs addition on a register with a value from another (or the same) register.
\section{SUB (Subtract)}
\OpcodeTable{0x60}{0110 0000}{sub}{Register}{Register}\\[6pt]
Performs subtraction on a register with a value from another (or the same) register.
\section{AND (Logical AND)}
\OpcodeTable{0x68}{0110 1000}{and}{Register}{Register}\\[6pt]
Logical ANDs the two registers together storing the result in operand 1.
\section{XOR (Logical Exclusive OR)}
\OpcodeTable{0x70}{0111 0000}{xor}{Register}{Register}\\[6pt]
Logical XORs the two registers together storing the result in operand 1.
\section{OR (Logical OR)}
\OpcodeTable{0x78}{0111 1000}{or}{Register}{Register}\\[6pt]
Logical ORs the two registers together storing the result in operand 1.
\section{NOT (Logical Negation)}
\OpcodeTable{0x80}{1000 0000}{not}{Register}{None}\\[6pt]
Inverts the bits of the target register.
\section{SHR (Shift Right)}
\OpcodeTable{0x88}{1000 1000}{shr}{Register}{Constant}\\[6pt]
Bit-wise shifts the contents of the register right Constant number of times.
\section{SHL (Shift Left)}
\OpcodeTable{0x90}{1001 0000}{shl}{Register}{Constant}\\[6pt]
Bit-wise shifts the contents of the register left Constant number of times.
\section{INC (Increment)}
\OpcodeTable{0x98}{1001 1000}{inc}{Register}{None}\\[6pt]
Increments the contents of the register by one. Over-flows will not be reported.
\section{DEC (Decrement)}
\OpcodeTable{0xA0}{1010 0000}{dec}{Register}{None}\\[6pt]
Decrements the contents of the register by one. Under-flows will not be reported.
\section{Push}
\OpcodeTable{0xA8}{1010 1000}{push}{Register}{None}
Pushes the value of Register onto the stack, decrementing the Stack Pointer by 2.
\section{Pop}
\OpcodeTable{0xB0}{1011 0000}{pop}{Register}{None}
Pops the top of the stack into Register, incrementing the Stack Pointer by 2.
\section{JMPI (Jump Indirect)}
\OpcodeTable{0xB8}{1011 1000}{jmpi}{Register}{None}\\[6pt]
Jumps unconditionally to a memory address stored in Operand 1. Sets the Program Counter to Operand 1.
\section{JMP (Jump)}
\OpcodeTable{0x02}{0000 0010}{jmp}{Address}{None}\\[6pt]
Jumps unconditionally to a memory address. Sets the Program Counter to Address.
\section{JZ (Jump if Zero)}
\OpcodeTable{0x03}{0000 0011}{jz}{Address}{None}\\[6pt]
Jumps to a memory address if the status flag is zero. Sets the Program Counter to Address.
\section{JG (Jump if Greater Than)}
\OpcodeTable{0x04}{0000 0100}{jg}{Address}{None}\\[6pt]
Jump to the Address if the status flag is greater than zero. Sets the Program Counter to Address.
\section{JL (Jump if Less Than)}
\OpcodeTable{0x05}{0000 0101}{jl}{Address}{None}\\[6pt]
Jumps to the address if the status flag is less than zero. Sets the Program Counter to Address.
\section{NOP (No Operation)}
\OpcodeTable{0x01}{0000 0001}{nop}{None}{None}\\[6pt]
Skips a clock cycle, incrementing the program counter.
\section{CALL (Call Subroutine)}
\OpcodeTable{0x06}{0000 0110}{call}{Address}{None}\\[6pt]
Pushes the base address to the stack and sets the Program Counter to Address.
\section{RET (Return from Subroutine)}
\OpcodeTable{0x07}{0000 0111}{ret}{None}{None}\\[6pt]
Pops the stack and sets the Program Counter to that value.
\end{document}
+1 -1
View File
@@ -3,7 +3,7 @@
#include "parser.h" #include "parser.h"
void Disassemble(unsigned char image[HIGHMEMORY]); //void Disassemble(unsigned char image[HIGHMEMORY]);
#endif #endif
-21
View File
@@ -1,21 +0,0 @@
#ifndef LIST_H
#define LIST_H
#include <stddef.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#define LISTDEFAULTSIZE 4
typedef struct {
void** content;
int size;
int capacity;
} List;
List* CreateList(void);
int AddListItem(const void *, size_t, List *);
void DestroyList(List*);
#endif
+15 -21
View File
@@ -2,29 +2,23 @@
#define OPCODES_H #define OPCODES_H
#include <string.h> #include <string.h>
#include "token.h" #include <errno.h>
#include <stdio.h>
#define OPCODECOUNT 13 typedef enum {
#define REGISTERCOUNT 8 R1 = 0, R2, R3, R4, R5, R6, R7, R8 = 7
} Registers;
typedef struct { typedef enum {
char* lexeme; COPYA = 0x08, COPYAB = 0x10, COPYRA = 0x18, COPYRAB = 0x20, COPYRARA = 0x28, COPYRARAB = 0x30, COPY = 0x38, COPYB = 0x40, COPYI = 0xC0,
TokenType op; CMP = 0x48, CMPI = 0x50, ADD = 0x58, SUB = 0x60, AND = 0x68, XOR = 0x70, OR = 0x78, NOT = 0x80, SHR = 0x88, SHL = 0x90,
TokenClass parameter_one; INC = 0x98, DEC = 0xA0, PUSH = 0xA8, POP = 0xB0,
TokenClass parameter_two; JMPI = 0xB8, JMP = 0x02, JZ = 0x03, JG = 0x04, JL = 0x05, NOP = 0x01, CALL = 0x06, RET = 0x07
} Instruction; } Mnemonic;
typedef struct { int IsOpcode(const char*, Mnemonic*);
char* lexeme; int IsRegister(const char*, Registers*);
TokenType type; void GetMnemonicText(Mnemonic mnemonic, char buffer[12]);
} Register; void GetRegisterText(Registers reg, char buffer[3]);
extern Instruction instructions[OPCODECOUNT];
extern Register registers[REGISTERCOUNT];
int IsOpcode(const char*, TokenType*);
int IsRegister(const char*, TokenType*);
const Instruction* GetOpcodeDetails(TokenType);
unsigned char GetInstructionMask(TokenType type, TokenClass parameterOneType);
#endif #endif
+5 -4
View File
@@ -1,14 +1,15 @@
#ifndef PARSER_H #ifndef PARSER_H
#define PARSER_H #define PARSER_H
#include <stdint.h>
#include <stdio.h> #include <stdio.h>
#include <stdlib.h> #include <stdlib.h>
#include "list.h" #include <string.h>
#include <stdarg.h>
#include "token.h" #include "token.h"
#include "opcodes.h" #include "opcodes.h"
#include "symbols_table.h"
#define HIGHMEMORY 65535 //64KiB - 1 AKA 0xFFFF void ParseTokens(TokenList* tokens, SymbolTable** symbolsTable);
unsigned char* ParseTokens(List*);
#endif #endif
+1 -2
View File
@@ -7,9 +7,8 @@
#include <errno.h> #include <errno.h>
#include "stdlib.h" #include "stdlib.h"
#include "token.h" #include "token.h"
#include "list.h"
#include "opcodes.h" #include "opcodes.h"
List* GenerateTokenList(const char*); TokenList* GenerateTokenList(const char*);
#endif #endif
+44
View File
@@ -0,0 +1,44 @@
#ifndef SYMBOLSTABLE_H
#define SYMBOLSTABLE_H
#include <stdint.h>
#include <stdlib.h>
#include <errno.h>
#include <stdio.h>
#include <string.h>
#include <sys/types.h>
#include "opcodes.h"
#include "token.h"
#define SYMBOLSTABLE_DEFAULT_CAPACITY 128
typedef struct __token_node {
const Token* Token;
struct __token_node* Left;
struct __token_node* Right;
} TokenNode;
typedef struct _symbol {
const char* Name;
int Address;
int Length;
const Token* Token;
TokenNode* ValueExpression;
} Symbol;
typedef struct {
Symbol** Symbols;
int Size;
int Capacity;
} SymbolTable;
SymbolTable* CreateSymbolTable(void);
TokenNode* CreateTokenNode(const Token* token);
int TryGetSymbol(const char* name, const SymbolTable* table, Symbol** outSymbol);
Symbol* AddSymbolToTable(const char* name, int address, SymbolTable* table);
void FreeSymbolTable(SymbolTable* table);
void FreeSymbol(Symbol* symbol);
int SymbolResolved(const Symbol* symbol, const SymbolTable* table);
int TryGetSymbolValue(const Symbol* symbol, const SymbolTable* table, unsigned short* value);
#endif
+55 -39
View File
@@ -5,52 +5,68 @@
#include <string.h> #include <string.h>
#include <errno.h> #include <errno.h>
#include <stdio.h> #include <stdio.h>
#include "opcodes.h"
#define REGISTEROFFSET 29
typedef enum { typedef enum {
PLUS, Plus = '+',
MINUS, Minus = '-',
STAR, Star = '*',
SLASH, Slash = '/',
POWER, Power = '^',
LPARAM, LParan = '(',
RPARAM, RParan = ')',
LBRACKET, LBracket = '[',
RBracket, RBracket = ']',
COMMA, Comma = ',',
STRING, NewLine = '\n'
IDENTIFIER, } TokenPunctuation;
LABEL,
NUMBER,
LineEnd,
//Keywords
DB, ORG,
//Opcodes
COPY, ADD, SUB, JZ, INT, YLD, RET,
CMP, NOP, JMP, CALL, IN, OUT,
//Registers
R1, R2, R3, R4, R5, R6, R7, R8
} TokenType;
typedef enum { typedef enum {
None = 0, DB,
Reg = 1, Include,
Constant = 2, Byte
Address = 4, } Directive;
Opcode = 8,
Directive = 16 typedef enum {
RegisterClass = 0,
NumberClass = 1,
CharacterClass = 2,
MnemonicClass = 4,
DirectiveClass = 8,
PunctuationClass = 16,
IdentifierClass = 32,
LabelClass = 64,
AddressClass = LabelClass | IdentifierClass
} TokenClass; } TokenClass;
typedef struct { typedef struct __token {
TokenType type; char* Lemexe;
char* lexeme; TokenClass Class;
void* value; int LineNumber;
int line; int EndOfFile;
TokenClass token_class;
union {
TokenPunctuation Punctuation;
Registers Register;
Mnemonic Mnemonic;
Directive Directive;
unsigned short Number;
} Value;
struct __token* Prev;
struct __token* Next;
} Token; } Token;
Token* CreateToken(char*, void*, int, TokenType); typedef struct {
void FreeToken(Token*); Token** content;
int size;
int capacity;
} TokenList;
Token* CreateToken(int lineNumber, TokenClass tokenClass);
TokenList* CreateTokenList(void);
int AddToken(Token* token, TokenList* list);
void RemoveToken(int index, TokenList* list);
void FreeToken(Token* token);
#endif #endif
+44
View File
@@ -0,0 +1,44 @@
.db MAX_MEM 0xFFFF
.db VIDEO_MEM 0xF37F
.db MAX_LENGTH MAX_MEM; - MAX_LENGTH
.db MSG "Hello, World!", 0
__start:
copy r1, MSG ; Pointer into r1
copy r8, VIDEO_MEM
call strlen
copy r1, MSG
cmp r2, 0
jz _end
cmp r2, MAX_LENGTH
jg _end
draw_loop:
copy byte [r8], [r1]
inc r1
inc r8
dec r2
cmp r2, 0
jz _end
jmp draw_loop
_end:
jmp _end
.db NewMsg "My message", 0
; Returns the length of a NULL terminated string
; Arguments: R1 - Pointer to the string
; Returns: R2 - Contains the length of the string
strlen:
copy r2, 0 ; length
copy byte r3, [r1]
cmp r3, 0
jz end ;The string is zero length
loop:
inc r1
copy byte r3, [r1]
cmp r3, 0
jz end
inc r2
jmp loop
end:
ret
+26 -26
View File
@@ -1,27 +1,27 @@
;.org 0x100 ;.include "./another_test.asm"
.db labelsz reference
.db video_start 0xF37F
.db msg "Hello, world!", 0 .db msg "Hello, world!", 0
.db more_stuff "AA", 0 .db NOTERM "No terminating byte here"
copy r7, msg ; // .db null_byte 0
cmp r8, 45 ;1000 1111 jmp [r4]
add r4, 30 ;load r2, label
sub r1, 69 something_insance:
int 20 load r1, unknown_symbol
jz 45 load r1, 5
jmp msg .db fun_alright 0x70;does this break?
in 10 load r3, r4
call more_stuff store byte r3, r5
ret store r5, r7
yld unknown_symbol:;Does this work?
cmp r1, 5;or one after the register?
;copy r1, 45 ;//B0 = 10110000 cmp r1, r2;This might work
;copy 2, 45 and r1, r2
;loop: add r4, r7
; cmp r1, 0 inc r1 ;increment r1
; jz loop xor r1, r1 ;clear self
; add r1, 1 pop r3
;mov r1, 5 ; move the immediate value 5 into r1 jmp unknown_symbol
;add r1,5 ; add 5 into r1 jmp [r4]
;int 21 ; maybe that will call some string drawing BIOS-like routine jz [r6]
;ret ;And a comment at the end
; 4 byte header
; | Address of First Opcode (2 bytes) | End of Binary (2 bytes) |
+5 -4
View File
@@ -14,7 +14,7 @@ const unsigned char ADDRESSMASK = 0x10; //0001 0000
int IsRegisterPattern(unsigned char pattern, char* lexeme); int IsRegisterPattern(unsigned char pattern, char* lexeme);
void GetParameter(unsigned char instruction, char* text); void GetParameter(unsigned char instruction, char* text);
/*
void Disassemble(unsigned char image[HIGHMEMORY]) { void Disassemble(unsigned char image[HIGHMEMORY]) {
Image = image; Image = image;
int position = image[0] + image[1]; int position = image[0] + image[1];
@@ -32,8 +32,9 @@ void Disassemble(unsigned char image[HIGHMEMORY]) {
IsRegisterPattern(instruction & 0x07, parameter1); IsRegisterPattern(instruction & 0x07, parameter1);
unsigned char parameterType = instruction & 0x18; unsigned char parameterType = instruction & 0x18;
if (!parameterType) { if (parameterType == REGMASK) {
IsRegisterPattern(Image[position], parameter2); IsRegisterPattern(Image[position + 1], parameter2);
position += 2;
} else if (parameterType == CONSTMASK) { } else if (parameterType == CONSTMASK) {
snprintf(parameter2, sizeof(char) * 31, "%d", Image[position + 1] + Image[position + 2]); snprintf(parameter2, sizeof(char) * 31, "%d", Image[position + 1] + Image[position + 2]);
position += 3; position += 3;
@@ -116,7 +117,7 @@ void Disassemble(unsigned char image[HIGHMEMORY]) {
memset(parameter2, '\0', sizeof(char) * 32); memset(parameter2, '\0', sizeof(char) * 32);
} }
} }
*/
int IsRegisterPattern(unsigned char pattern, char* lexeme) { int IsRegisterPattern(unsigned char pattern, char* lexeme) {
if (pattern > 8) { if (pattern > 8) {
lexeme[0] = '\0'; lexeme[0] = '\0';
-66
View File
@@ -1,66 +0,0 @@
#include "../includes/list.h"
List* CreateList() {
List *new = malloc(sizeof(List));
if (!new) {
fprintf(stderr, "Failed to malloc() for new new List.\n");
return NULL;
}
new->content = malloc(sizeof(void*) * LISTDEFAULTSIZE);
if (!new->content) {
fprintf(stderr, "Failed to malloc() memory for List contents.\n");
free(new);
return NULL;
}
new->size = 0;
new->capacity = LISTDEFAULTSIZE;
return new;
}
int AddListItem(const void *value, size_t size, List* list) {
if (!list) return -1;
if (!value) return -1;
if (size == 0) return -1;
if (list->capacity < list->size + 1) {
void* ptr = realloc(list->content, sizeof(void*) * list->capacity * 2);
//Note: realloc will free list->root if it succeeds.
if (!ptr) {
fprintf(stderr, "Failed to resize array with realloc() (%d bytes).\n", list->capacity * 2);
return -1;
}
list->content = ptr;
list->capacity = list->capacity * 2;
}
void* item = calloc(1, size);
if (!item) {
fprintf(stderr, "Failed to calloc() new memory (%lu bytes).\n", size);
return -1;
}
memcpy(item, value, size);
list->content[list->size] = item;
list->size++;
return 0;
}
void DestroyList(List* list) {
if (!list) return;
for(int i = 0; i < list->size; i++) {
free(list->content[i]);
}
free(list->content);
free(list);
}
+228 -22
View File
@@ -1,12 +1,20 @@
#include <bits/types/FILE.h>
#include <stdlib.h> #include <stdlib.h>
#include <stdio.h> #include <stdio.h>
#include <string.h> #include <string.h>
#include <sysexits.h> #include <sysexits.h>
#include "../includes/list.h" #include "../includes/token.h"
#include "../includes/parser.h"
#include "../includes/futil.h" #include "../includes/futil.h"
#include "../includes/scanner.h" #include "../includes/scanner.h"
#include "../includes/disass.h" #include "../includes/parser.h"
const char* MagicStartName = "__start";
TokenList* LIST;
SymbolTable* Symbols = NULL;
unsigned char mem[128] = {0};
void print(void);
void assemble(void);
int main(int argc, char* args[]) { int main(int argc, char* args[]) {
if (argc == 1) { if (argc == 1) {
@@ -14,29 +22,227 @@ int main(int argc, char* args[]) {
return EX_USAGE; return EX_USAGE;
} }
atexit(print);
char* source_code; char* source_code;
size_t bytes_read; size_t bytes_read;
if (!ReadAllString(args[1], &source_code, &bytes_read)) return EX_IOERR; if (!ReadAllString(args[1], &source_code, &bytes_read)) return EX_IOERR;
List* list = GenerateTokenList(source_code); LIST = GenerateTokenList(source_code);
// for(int i = 0; i < list->size; i++) { ParseTokens(LIST, &Symbols);
// Token* t = (Token*) list->content[i];
// if (t->type == LineEnd) {
// if (i - 1 >= 0 && ((Token*) list->content[i - 1])->type != LineEnd)
// printf("\n");
// continue;
// }
// printf("[%i] '%s' [%i] ", t->type, t->lexeme, t->token_class);
// if (t->type == LABEL) printf("* ");
// }
unsigned char* image = ParseTokens(list);
Disassemble(image);
free(source_code); free(source_code);
}
assemble();
printf("\nBytes:\n");
for(unsigned long i = 0; i < sizeof(mem); i++) {
if (i != 0 && i % 8 == 0) printf("\n");
printf("%02X ", mem[i] & 0xFF);
}
printf("\n");
}
void print(void) {
char mnemonic[12];
printf("printing tokens...\n");
for(int i = 0; i < LIST->size; i++) {
Token* t = (Token*) LIST->content[i];
if (t->EndOfFile) {
printf("EOF\n");
break;
}
if (t->Class == PunctuationClass){
if(t->Value.Punctuation == NewLine) {
printf("<%d>\n", t->LineNumber);
continue;
}
printf("[P]%c", t->Value.Punctuation);
continue;
}
if (t->Class == LabelClass) {
printf("[L]%s*", t->Lemexe);
continue;
}
if (t->Class == IdentifierClass) {
printf("[I]%s ", t->Lemexe);
continue;
}
if (t->Class == RegisterClass) {
printf("[R]%d", t->Value.Register);
continue;
}
if (t->Class == NumberClass) {
printf("[N]%d ", t->Value.Number);
continue;
}
if (t->Class == MnemonicClass) {
GetMnemonicText(t->Value.Mnemonic, mnemonic);
printf("[M]%s ", mnemonic);
continue;
}
if (t->Class == DirectiveClass) {
printf("[D]%d ", t->Value.Directive);
}
if (t->Class == CharacterClass) {
printf("[C]'%s'", t->Lemexe);
}
}
}
int CurrentIndex = 0;
int PC2 = 0;
void WriteByte(unsigned char byte) {
mem[PC2] = byte;
PC2++;
}
void WriteWord(unsigned short word) {
mem[PC2] = (word >> 8) & 0xFF;
mem[PC2 + 1] = word & 0xFF;
PC2 += 2;
}
unsigned short GetValueFromToken(Token* token) {
if (token->Class & IdentifierClass) {
Symbol* symbol;
if (TryGetSymbol(token->Lemexe, Symbols, &symbol)) {
return symbol->Address;
}
else {
fprintf(stderr, "Failed to find symbol %s while assembling.\n", token->Lemexe);
exit(1);
}
}
else if (token->Class == NumberClass) return token->Value.Number;
fprintf(stderr, "Failed to find token '%d' while assmbling.\n", token->Class);
exit(1);
}
int IsAtEnd(void){
return CurrentIndex >= LIST->size;
}
void assemble() {
while(!IsAtEnd()) {
Token* current = LIST->content[CurrentIndex];
switch(current->Value.Mnemonic) {
case COPYA:
case COPYAB:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case COPYRA:
case COPYRAB:
case COPYRARA:
case COPYRARAB:
case COPY:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteByte(((Token*) LIST->content[CurrentIndex])->Value.Register);
break;
case COPYI: //TODO: should explicitly get number from value I feel
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case COPYB:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteByte((unsigned char)GetValueFromToken(LIST->content[CurrentIndex]));
break;
case CMP:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteByte(((Token*) LIST->content[CurrentIndex])->Value.Register);
break;
case CMPI:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case ADD:
case SUB:
case AND:
case OR:
case XOR:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteByte(((Token*) LIST->content[CurrentIndex])->Value.Register);
break;
case SHL:
case SHR:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case INC:
case DEC:
case PUSH:
case POP:
case JMPI:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
break;
case JMP:
case JZ:
case JG:
case JL:
WriteByte(current->Value.Mnemonic);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case NOP:
case RET:
WriteByte(current->Value.Mnemonic);
break;
case CALL:
WriteByte(current->Value.Mnemonic);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
default:
break;
}
CurrentIndex++;
}
}
+72 -105
View File
@@ -1,114 +1,76 @@
#include "../includes/opcodes.h" #include "../includes/opcodes.h"
#include <ctype.h>
#include <stdlib.h> #include <stdlib.h>
#include <string.h> #include <string.h>
Instruction instructions[OPCODECOUNT] = { #define OPCODECOUNT 34
{ "copy", COPY, Reg | Address, Reg | Constant | Address },
{ "add", ADD, Reg, Reg | Constant }, struct _instruction {
{ "sub", SUB, Reg, Reg | Constant }, char* Name;
{ "jz", JZ, Address, None }, Mnemonic Mnemonic;
{ "int", INT, Constant, None },
{ "yld", YLD, None, None },
{ "ret", RET, None, None },
{ "cmp", CMP, Reg, Reg | Constant },
{ "in", IN, Constant | Reg, None},
{ "out", OUT, None, None},
{ "nop", NOP, None, None},
{ "jmp", JMP, Address, None},
{ "call", CALL, Address, None}
}; };
unsigned char GetInstructionMask(TokenType type, TokenClass parameterOneType) { struct _instruction instructions[31] = {
switch (type) { { "copya", COPYA },
case COPY: { "copyab", COPYAB },
if (parameterOneType & Reg) return 0x20; //0b00100000; { "copyra", COPYRA },
{ "copyrab", COPYRAB },
return 0xA0;//0b10100000; { "copyrara", COPYRARA },
case ADD: { "copyrarab", COPYRARAB },
return 0x40;//0b01000000; { "copy", COPY },
case SUB: { "copyi", COPYI },
return 0x60;//0b01100000; { "copyb", COPYB },
case CMP: { "cmp", CMP },
return 0x80;//0b10000000; { "cmpi", CMPI },
case JZ: { "add", ADD },
return 0x01;//0b00000001; { "sub", SUB },
case INT: { "and", AND },
return 0x02; //0b00000010; { "xor", XOR },
case YLD: { "or", OR },
return 0x03; { "not", NOT },
case RET: { "shr", SHR },
return 0x04; { "shl", SHL },
case CALL: { "inc", INC },
return 0x05; { "dec", DEC },
case JMP: { "push", PUSH },
return 0x06;//0b00000110; { "pop", POP },
case IN: { "jmpi", JMPI },
return 0x07;//0b00000111; { "jmp", JMP },
case OUT: { "jz", JZ },
return 0x08;//0b00001000; { "jg", JG },
default: { "jl", JL },
return 0x00; { "nop", NOP },
} { "call", CALL },
} { "ret", RET }
//yld
/*
NOP - 0000 0000 NOP
JZ - 0000 0001 JZ Address
INT - 0000 0010 INT Constant
YLD - 0000 0011 YLD
RET - 0000 0100 RET
CALL - 0000 0101 CALL Address
JMP - 0000 0110 JMP Address
IN - 0000 0111 IN Constant
OUT - 0000 1000 OUT Constant
COPY - 0010 0XXX COPY REG, REG
- 0010 1XXX COPY REG, Constant
- 0011 0XXX COPY REG, Address
- 1010 0XXX COPY Address, REG
- 1010 1XXX COPY Address, Constant
- 1011 0XXX COPY Address, Address
ADD - 0100 0XXX ADD REG, REG
- 0100 1XXX ADD REG, Constant
SUB - 0110 0XXX SUB REG, REG
- 0110 1XXX SUB REG, Constant
CMP - 1000 0XXX CMP REG, REG
- 1000 1XXX CMP REG, Constant
- 1001 0XXX CMP REG, Address
*/
//COPY X01X XXXX
//ADD 010X XXXX
//SUB 011X XXXX
//CMP 100X XXXX
//REG XXX0 0XXX
//Constant XXX0 1XXX
//Address XXX1 0XXX
//R1 XXXX X000 -> XXXX X111 (R1 to R8)
Register registers[REGISTERCOUNT] = {
{ "r1", R1 },
{ "r2", R2 },
{ "r3", R3 },
{ "r4", R4 },
{ "r5", R5 },
{ "r6", R6 },
{ "r7", R7 },
{ "r8", R8 }
}; };
const Instruction* GetOpcodeDetails(TokenType type) { void GetMnemonicText(Mnemonic mnemonic, char buffer[12]) {
memset(buffer, '\0', 12);
for(int i = 0; i < OPCODECOUNT; i++) { for(int i = 0; i < OPCODECOUNT; i++) {
if (instructions[i].op == type) return &instructions[i]; if (instructions[i].Mnemonic == mnemonic) {
strncpy(buffer, instructions[i].Name, 11);
break;
}
} }
return NULL;
} }
int IsOpcode(const char* text, TokenType* opcode) { void GetRegisterText(Registers reg, char buffer[3]) {
memset(buffer, '\0', 3);
if (reg < R1 || reg > R8) return;
buffer[0] = 'r';
buffer[1] = reg + 49;
}
int IsOpcode(const char* text, Mnemonic* opcode) {
if (!text) return 0; if (!text) return 0;
for(int i = 0; i < OPCODECOUNT; i++) { for(unsigned long i = 0; i < sizeof(instructions) / sizeof(struct _instruction); i++) {
if (strcmp(instructions[i].lexeme, text) == 0) { if (strcmp(instructions[i].Name, text) == 0) {
if (opcode) *opcode = instructions[i].op; if (opcode) *opcode = instructions[i].Mnemonic;
return 1; return 1;
} }
} }
@@ -116,15 +78,20 @@ int IsOpcode(const char* text, TokenType* opcode) {
return 0; return 0;
} }
int IsRegister(const char* text, TokenType* reg) { int IsRegister(const char* text, Registers* reg) {
if (!text) return 0; if (!text) return 0;
for(int i = 0; i < REGISTERCOUNT; i++) { int length = strlen(text);
if (strcmp(registers[i].lexeme, text) == 0) { Registers r = R8;
*reg = registers[i].type;
return 1;
}
}
return 0; if (length != 2) return 0;
if (text[0] != 'r') return 0;
if (!isdigit(text[1])) return 0;
r = text[1] - 0x31;
if (reg) *reg = r;
return 1;
} }
+517 -402
View File
File diff suppressed because it is too large Load Diff
+167 -41
View File
@@ -1,13 +1,16 @@
#include "../includes/scanner.h" #include "../includes/scanner.h"
#include <ctype.h>
#include <stdlib.h> #include <stdlib.h>
#include <string.h> #include <string.h>
#include <limits.h>
const char* SourceCode; const char* SourceCode;
int Line = 0; int Line = 1;
int Position = 0; int Position = 0;
int SourceLength = 0; int SourceLength = 0;
int ScannerAtEnd(void); int ScannerAtEnd(void);
int IsPunctuation(char); int IsPunctuation(char);
int IsWhiteSpace(char);
char PeekScanner(void); char PeekScanner(void);
char PeekAheadScanner(void); char PeekAheadScanner(void);
void AdvanceScanner(void); void AdvanceScanner(void);
@@ -18,8 +21,8 @@ Token* ParseNumber(void);
Token* ParsePunctuation(char); Token* ParsePunctuation(char);
void IgnoreLine(void); void IgnoreLine(void);
List* GenerateTokenList(const char* source) { TokenList* GenerateTokenList(const char* source) {
List* tokens = CreateList(); TokenList* tokens = CreateTokenList();
Token* token = NULL; Token* token = NULL;
if (!tokens) return NULL; if (!tokens) return NULL;
@@ -34,41 +37,60 @@ List* GenerateTokenList(const char* source) {
case ' ': case ' ':
case '\r': case '\r':
case '\t': case '\t':
case '\v':
case '\f':
AdvanceScanner(); AdvanceScanner();
break; //Ignore whitespace break; //Ignore whitespace
case '\n': case '\n':
token = CreateToken("^", NULL, Line, LineEnd); if (tokens->size > 0) {
AddListItem(token, sizeof(Token), tokens); token = tokens->content[tokens->size - 1];
if (token->Class == PunctuationClass && token->Value.Punctuation == NewLine) {
AdvanceScanner();
Line++;
continue;
}
}
token = CreateToken(Line, PunctuationClass);
token->Value.Punctuation = NewLine;
AddToken(token, tokens);
AdvanceScanner(); AdvanceScanner();
Line++; Line++;
break; break;
case ';': case ';':
IgnoreLine(); IgnoreLine();
if (tokens->size == 0) {
//There's nothing here so that means this is some comments block at the start of the file.
AdvanceScanner(); //Consume the actual new line char.
Line++;
}
break; break;
case '.': //directive like ".org" or ".db" case '.': //directive like ".include" or ".db"
token = ParseDirective(); token = ParseDirective();
if (token) AddListItem(token, sizeof(Token), tokens); if (token) AddToken(token, tokens);
break; break;
case '"': case '"':
token = ParseString(); token = ParseString();
if (token) AddListItem(token, sizeof(Token), tokens); if (token) AddToken(token, tokens);
break; break;
default: default:
if (isdigit(c)) { if (isdigit(c)) {
AddListItem(ParseNumber(), sizeof(Token), tokens); AddToken(ParseNumber(), tokens);
break; break;
} }
if (IsPunctuation(c)) { if (IsPunctuation(c)) {
AdvanceScanner(); AdvanceScanner();
AddListItem(ParsePunctuation(c), sizeof(Token), tokens); AddToken(ParsePunctuation(c), tokens);
break; break;
} }
token = ParseIdentifier(); token = ParseIdentifier();
if (token) AddListItem(token, sizeof(Token), tokens); if (token) AddToken(token, tokens);
break; break;
} }
@@ -76,6 +98,21 @@ List* GenerateTokenList(const char* source) {
token = NULL; token = NULL;
} }
if (tokens->size > 0) {
token = tokens->content[tokens->size - 1];
if (token->Class == PunctuationClass && token->Value.Punctuation == NewLine) {
//If the last token is a line break, remove it as its not too meaningful.
tokens->size--;
}
}
token = CreateToken(Line, PunctuationClass);
token->EndOfFile = 1;
AddToken(token, tokens);
return tokens; return tokens;
} }
@@ -85,9 +122,21 @@ Token* ParseNumber(void) {
while(!ScannerAtEnd() && isdigit(PeekScanner())) while(!ScannerAtEnd() && isdigit(PeekScanner()))
AdvanceScanner(); AdvanceScanner();
if (PeekScanner() == 'x' && isdigit(PeekAheadScanner())) { if (tolower(PeekScanner()) == 'x') {
AdvanceScanner(); //Consume the 'x' char ahead = tolower(PeekAheadScanner());
while(!ScannerAtEnd() && isdigit(PeekScanner())) AdvanceScanner();
if ((ahead >= 'a' && ahead <= 'f') || isdigit(ahead)) {
AdvanceScanner(); //Consume the 'x'
while(!ScannerAtEnd() && PeekScanner() != '\n' && !IsWhiteSpace(PeekScanner()) && PeekScanner() != ';') {
if (isdigit(PeekScanner())) {
AdvanceScanner();
continue;
}
if (tolower(PeekScanner()) >= 'a' && tolower(PeekScanner()) <= 'f') AdvanceScanner();
}
}
} }
int length = Position - start; int length = Position - start;
@@ -95,24 +144,26 @@ Token* ParseNumber(void) {
if (length == 0) return NULL; if (length == 0) return NULL;
char* lexeme = calloc(sizeof(char), length + 1); char* lexeme = calloc(sizeof(char), length + 1);
long* value = calloc(1, sizeof(long));
if (!lexeme) { if (!lexeme) {
fprintf(stderr, "Failed to calloc space for number. %s.\n", strerror(errno)); fprintf(stderr, "Failed to calloc space for number. %s.\n", strerror(errno));
return NULL; return NULL;
} }
if (!value) { memcpy(lexeme, &SourceCode[start], length);
free(lexeme);
fprintf(stderr, "Failed to calloc space for the raw numeric value of a token. %s.\n", strerror(errno)); Token* token = CreateToken(Line, NumberClass);
return NULL;
token->Value.Number = strtol(lexeme, NULL, 0);
if (errno != 0) {
fprintf(stderr, "[Error] Line %d: Invalid number detected. %s.\n", Line, strerror(errno));
exit(1);
} }
memcpy(lexeme, &SourceCode[start], length); token->Lemexe = lexeme;
// Setting the base to zero means the function will pick the base.
*value = strtol(lexeme, NULL, 0);
return CreateToken(lexeme, value, Line, NUMBER); return token;
} }
Token* ParseDirective(void) { Token* ParseDirective(void) {
@@ -133,19 +184,23 @@ Token* ParseDirective(void) {
memcpy(directive, &SourceCode[start], length); memcpy(directive, &SourceCode[start], length);
Token* token = CreateToken(Line, DirectiveClass);
if (strcmp(directive, ".db") == 0) { if (strcmp(directive, ".db") == 0) {
return CreateToken(directive, directive, Line, DB); token->Value.Directive = DB;
return token;
} }
else if (strcmp(directive, ".org") == 0) { else if (strcmp(directive, ".include") == 0) {
fprintf(stderr, "[Warning] Org is not a supported directive.\n"); token->Value.Directive = Include;
free(directive);
IgnoreLine(); return token;
return NULL;
} }
fprintf(stderr, "[Error] Unknown assembler directive '%s'\n", directive); fprintf(stderr, "[Error] Unknown assembler directive '%s'\n", directive);
free(directive); free(directive);
free(token);
IgnoreLine(); IgnoreLine();
@@ -178,7 +233,9 @@ Token* ParseString(void) {
AdvanceScanner(); //Consume the trailing double quote. AdvanceScanner(); //Consume the trailing double quote.
Token* token = CreateToken(lexeme, lexeme, Line, STRING); Token* token = CreateToken(Line, CharacterClass);
token->Lemexe = lexeme;
return token; return token;
} }
@@ -186,7 +243,7 @@ Token* ParseString(void) {
Token* ParseIdentifier(void) { Token* ParseIdentifier(void) {
int start = Position; int start = Position;
while(!ScannerAtEnd() && !IsPunctuation(PeekScanner()) && PeekScanner() != ' ' && PeekScanner() != '\n') { while(!ScannerAtEnd() && !IsPunctuation(PeekScanner()) && PeekScanner() != ' ' && PeekScanner() != '\n' && PeekScanner() != ';') {
AdvanceScanner(); AdvanceScanner();
} }
@@ -194,7 +251,8 @@ Token* ParseIdentifier(void) {
if (length == 0) return NULL; if (length == 0) return NULL;
TokenType type; Mnemonic mnemonics;
Registers reg;
char* lexeme = calloc(sizeof(char), length + 1); char* lexeme = calloc(sizeof(char), length + 1);
if (!lexeme) { if (!lexeme) {
@@ -204,14 +262,41 @@ Token* ParseIdentifier(void) {
memcpy(lexeme, &SourceCode[start], length); memcpy(lexeme, &SourceCode[start], length);
if (IsOpcode(lexeme, &type)) return CreateToken(lexeme, lexeme, Line, type); if (IsOpcode(lexeme, &mnemonics)) {
if (IsRegister(lexeme, &type)) return CreateToken(lexeme, lexeme, Line, type); Token* token = CreateToken(Line, MnemonicClass);
token->Value.Mnemonic = mnemonics;
return token;
}
if (IsRegister(lexeme, &reg)) {
Token* token = CreateToken(Line, RegisterClass);
token->Value.Register = reg;
return token;
}
if (lexeme[length - 1] == ':') { if (lexeme[length - 1] == ':') {
lexeme[length - 1] = '\0'; //Bit hacky, but this makes the parser's job a bit easier. lexeme[length - 1] = '\0'; //Bit hacky, but this makes the parser's job a bit easier.
return CreateToken(lexeme, lexeme, Line, LABEL); Token* token = CreateToken(Line, LabelClass);
token->Lemexe = lexeme;
return token;
}
if (strcmp(lexeme, "byte") == 0) {
Token* token = CreateToken(Line, DirectiveClass);
token->Lemexe = lexeme;
token->Value.Directive = Byte;
return token;
} }
return CreateToken(lexeme, lexeme, Line, IDENTIFIER); Token* token = CreateToken(Line, IdentifierClass);
token->Lemexe = lexeme;
return token;
} }
char PeekScanner(void) { char PeekScanner(void) {
@@ -243,6 +328,23 @@ int IsPunctuation(char c) {
case '(': case '(':
case ')': case ')':
case ',': case ',':
case '-':
case '+':
case '*':
case '/':
return 1;
default:
return 0;
}
}
int IsWhiteSpace(char c) {
switch(c) {
case ' ':
case '\t':
case '\v':
case '\f':
case '\r':
return 1; return 1;
default: default:
return 0; return 0;
@@ -250,18 +352,42 @@ int IsPunctuation(char c) {
} }
Token* ParsePunctuation(char c) { Token* ParsePunctuation(char c) {
TokenPunctuation punctuation;
switch(c) { switch(c) {
case '[': case '[':
return CreateToken("[", NULL, Line, LBRACKET); punctuation = LBracket;
break;
case ']': case ']':
return CreateToken("]", NULL, Line, RBracket); punctuation = RBracket;
break;
case '(': case '(':
return CreateToken("(", NULL, Line, LPARAM); punctuation = LParan;
break;
case ')': case ')':
return CreateToken(")", NULL, Line, RPARAM); punctuation = RParan;
break;
case ',': case ',':
return CreateToken(",", NULL, Line, COMMA); punctuation = Comma;
break;
case '-':
punctuation = Minus;
break;
case '+':
punctuation = Plus;
break;
case '*':
punctuation = Star;
break;
case '/':
punctuation = Slash;
break;
default: default:
return NULL; return NULL;
} }
Token* token = CreateToken(Line, PunctuationClass);
token->Value.Punctuation = punctuation;
return token;
} }
+202
View File
@@ -0,0 +1,202 @@
#include "../includes/symbols_table.h"
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
int SymbolResolved(const Symbol* symbol, const SymbolTable* table) {
if (!symbol) return 0;
const Token* token = symbol->Token;
if (!token) return 0;
if (token->Class == LabelClass || token->Class == CharacterClass || token->Class == MnemonicClass) return 1;
return TryGetSymbolValue(symbol, table, NULL);
}
SymbolTable* CreateSymbolTable(void){
SymbolTable* table = calloc(1, sizeof(SymbolTable));
if (!table) {
fprintf(stderr, "Failed to calloc memory for a SymbolTable. %s.\n", strerror(errno));
return NULL;
}
table->Symbols = calloc(SYMBOLSTABLE_DEFAULT_CAPACITY, sizeof(Symbol*));
if (!table->Symbols) {
fprintf(stderr, "Failed to calloc Symbol list. %s.\n", strerror(errno));
free(table);
return NULL;
}
table->Capacity = SYMBOLSTABLE_DEFAULT_CAPACITY;
table->Size = 0;
return table;
}
Symbol* CreateSymbol(const char* name, int address) {
Symbol* symbol = calloc(1, sizeof(Symbol));
if (!symbol) {
fprintf(stderr, "Failed to create Symbol '%s'. %s.\n", name, strerror(errno));
return NULL;
}
symbol->Address = address;
symbol->Name = name;
//symbol->Type = RefUnknown;
return symbol;
}
int TryGetSymbol(const char* name, const SymbolTable* table, Symbol** outSymbol) {
*outSymbol = NULL;
if (!name || !table) return 0;
for(int i = 0; i < table->Size; i++) {
if (strcmp(table->Symbols[i]->Name, name) == 0) {
*outSymbol = table->Symbols[i];
return 1;
}
}
return 0;
}
Symbol* AddSymbolToTable(const char* name, int address, SymbolTable* table) {
//if (!name || !value || !table || length == 0) return NULL;
for(int i = 0; i < table->Size; i++) {
if (strcmp(table->Symbols[i]->Name, name) == 0) {
//TODO: Do we update or throw some kind of an error?
return table->Symbols[i];
}
}
if (table->Capacity < table->Size + 1) {
Symbol** newBlock = realloc(table->Symbols, sizeof(Symbol*) * table->Capacity * 2);
if (!newBlock) {
fprintf(stderr, "Failed to realloc space for a new symbol '%s'. %s.\n", name, strerror(errno));
return NULL;
}
table->Capacity *= 2;
table->Symbols = newBlock;
}
Symbol* symbol = CreateSymbol(name, address);
table->Symbols[table->Size] = symbol;
table->Size++;
return symbol;
}
void FreeSymbolTable(SymbolTable* table) {
if (!table) return;
for(int i = 0; i < table->Size; i++) FreeSymbol(table->Symbols[i]);
free(table);
}
void FreeSymbol(Symbol* symbol) {
if (!symbol) return;
free(symbol);
}
int TryGetTokenNodeValue(const TokenNode* node, const SymbolTable* table, unsigned short* value);
unsigned short DoOp(unsigned short left, unsigned short right, TokenPunctuation op);
int TryGetSymbolValue(const Symbol* symbol, const SymbolTable* table, unsigned short* value) {
if (!symbol || !symbol->Token || !table) return 0;
TokenNode* root = symbol->ValueExpression;
Symbol* s = NULL;
if (value) *value = 0;
if (!root || !root->Token) return 0;
switch (root->Token->Class) {
case NumberClass:
if (value) *value = root->Token->Value.Number;
return 1;
case IdentifierClass:
if (!TryGetSymbol(root->Token->Lemexe, table, &s)) return 0;
return TryGetTokenNodeValue(s->ValueExpression, table, value);
case PunctuationClass:
return TryGetTokenNodeValue(root, table, value);
default:
return 0;
}
}
int TryGetTokenNodeValue(const TokenNode* node, const SymbolTable* table, unsigned short* value) {
const Token* token = node->Token;
Symbol* symbol = { 0 };
if (token->Class == NumberClass) {
if (value) *value = token->Value.Number;
return 1;
}
if (token->Class == IdentifierClass) {
if (!TryGetSymbol(token->Lemexe, table, &symbol)) return 0;
return TryGetTokenNodeValue(symbol->ValueExpression, table, value);
}
if (token->Class != PunctuationClass) return 0;
unsigned short left = 0;
unsigned short right = 0;
if (!TryGetTokenNodeValue(node->Left, table, &left)) return 0;
if (!TryGetTokenNodeValue(node->Right, table, &right)) return 0;
if (value) *value = DoOp(left, right, token->Value.Punctuation);
return 1;
}
TokenNode* CreateTokenNode(const Token* token){
TokenNode* node = calloc(1, sizeof(TokenNode));
if (!node) {
fprintf(stderr, "Failed to calloc memory for a TokenNode. %s.\n", strerror(errno));
return NULL;
}
node->Token = token;
return node;
}
unsigned short DoOp(unsigned short left, unsigned short right, TokenPunctuation op) {
switch (op) {
case Plus:
return left + right;
case Minus:
return left - right;
case Star:
return left * right;
case Slash:
return left / right;
default:
break;
}
return 0;
}
+67 -26
View File
@@ -1,8 +1,8 @@
#include "../includes/token.h" #include "../includes/token.h"
TokenClass GetTokenClass(TokenType); #define LISTDEFAULTSIZE 32
Token* CreateToken(char* lexeme, void* value, int lineNumber, TokenType type) { Token* CreateToken(int lineNumber, TokenClass tokenClass) {
Token* token = calloc(1, sizeof(Token)); Token* token = calloc(1, sizeof(Token));
if (!token) { if (!token) {
@@ -10,36 +10,77 @@ Token* CreateToken(char* lexeme, void* value, int lineNumber, TokenType type) {
return NULL; return NULL;
} }
token->type = type; token->Class = tokenClass;
token->line = lineNumber; token->LineNumber = lineNumber;
token->lexeme = lexeme;
token->value = value;
token->token_class = GetTokenClass(type);
return token; return token;
} }
TokenClass GetTokenClass(TokenType type) {
if (type >= R1 && type <= R8) return Reg;
if (type >= COPY && type <= OUT) return Opcode;
if (type >= DB && type <= ORG) return Directive;
switch(type) {
case STRING:
case IDENTIFIER:
case LABEL:
return Address;
case NUMBER:
return Constant;
default:
return None;
}
}
void FreeToken(Token* token) { void FreeToken(Token* token) {
if (!token) return; if (!token) return;
if (token->value && (token->type >= STRING || token->type == NUMBER)) free(token->value);
free(token); free(token);
}
TokenList* CreateTokenList(void) {
TokenList *new = malloc(sizeof(TokenList));
if (!new) {
fprintf(stderr, "Failed to malloc() for new new List.\n");
return NULL;
}
new->content = calloc(LISTDEFAULTSIZE, sizeof(void*));
if (!new->content) {
fprintf(stderr, "Failed to malloc() memory for List contents.\n");
free(new);
return NULL;
}
new->size = 0;
new->capacity = LISTDEFAULTSIZE;
return new;
}
int AddToken(Token* token, TokenList* list) {
if (!list || !token) return 0;
if (list->capacity < list->size + 1) {
void* ptr = realloc(list->content, sizeof(void*) * list->capacity * 2);
//Note: realloc will free list->root if it succeeds.
if (!ptr) {
fprintf(stderr, "Failed to resize array with realloc() (%d bytes).\n", list->capacity * 2);
return 0;
}
list->content = ptr;
list->capacity *= 2;
}
if (list->size > 0)
{
Token* prev = list->content[list->size - 1];
token->Prev = prev;
prev->Next = token;
}
list->content[list->size] = token;
list->size++;
return 1;
}
void RemoveToken(int index, TokenList* list) {
Token* token = list->content[index];
if (token->Prev) token->Prev->Next = token->Next;
memmove(&list->content[index], &list->content[index + 1], (list->size - index) * sizeof(Token*));
list->size--;
list->content[list->size] = NULL;
} }