43 Commits
Author SHA1 Message Date
glm94 c95d739e16 Updated the parser to handle treating variables as expressions. Updated the symbols table to handle computing the values of variables. 2023-09-07 21:29:58 -05:00
glm94 21c2bf046e Wrote the very first part of the tree walking code, more just to make sure the logic is sound. 2023-09-01 10:04:21 -05:00
glm94 b810d7803c Updated the parser to create an expression tree for symbols declared via .db directives. Scanner now see some math operators as 'punctuation'. 2023-08-31 23:27:33 -05:00
glm94 9e178ca3da Build a quick draft of how the symbol value should be computed. 2023-08-25 01:27:22 -05:00
glm94 82b2136a69 Started to work on setting up Symbol value resolution which will be based on an abstract syntax tree that can handle undefined values if unknown symbols are used in an expression. 2023-08-25 00:21:19 -05:00
glm94 4bb90da6df Make the compiler options a bit more strict about C compliance (in other words I just learned about these flags and they seem to be ideal for this project). 2023-08-25 00:15:18 -05:00
glm94 4af2527806 Updated the structure of symbols and tokens to hopefully help make computing symbol values easier. 2023-08-07 01:18:01 -05:00
glm94 65383cded2 Updated the way symbols are represented and how they are marked as resolved. 2023-07-09 22:04:55 -05:00
glm94 9c8fe17c8f Minor code cleanup and added an extra flag to the compiler. If a symbol is a label it will now hold a reference to the opcode and by extension the offset into the file it points to. 2023-07-06 19:39:03 -05:00
glm94 3031fa8b9b Code cleanup. 2023-06-29 20:53:39 -05:00
glm94 2918f2964d Fixed a bug where words weren't being written to 'memory' correctly. 2023-06-24 18:00:13 -05:00
glm94 980b0449fc Fixed a bug with the parse discarding identifier tokens when parsing a COPY instruction. Added the first batch of code to encode the final token array to a binary form. 2023-06-22 20:22:20 -05:00
glm94 6741dde7ce Hooked up the program counter in the parser to set the memory offset of symbols, well labels to be more acurrate. 2023-06-18 14:51:37 -05:00
glm94 c6e6e2c5dc Fixed a bug in the parser not setting an identifier when one is used with a CMP and fixed a bug in the example assembly program. 2023-06-18 12:11:30 -05:00
glm94 0eaa5527ea Updated the parser to be able to handle the changes made to the ISA. 2023-06-18 00:50:18 -05:00
glm94 403439def4 Updated the opcode listing and 'fixed' the code to at least compile, but obviously I'll need to rework how the new copy instructions are to be parsed. Fixed some errors in the documentation where I skipped some hex numbers and added a copy immediate op. 2023-06-17 21:15:08 -05:00
glm94 0fd6413d81 Updated the instruction set and mapped out the opcodes as well as wrote a sample program to see if the ISA was complete enough to build a trivial program. 2023-06-17 16:56:42 -05:00
glm94 a9e57132be Fixed issue #1 Scanner Bug where the start of comment would become part of an identifier or throw off number parsing. 2023-04-14 13:28:44 -05:00
glm94 aa3888b578 Fixed up the RemoveCurrentToken function which was causing all sorts of bugs, and added the ability to delcare a data byte, '.db', to have a variable's address as a value. This will be needed for building up an interrupt table at least with the version I have in mind. 2023-04-04 22:14:17 -05:00
glm94 c3d12da54a Added some ideas to the encoding scheme to be used by the processor. 2023-03-30 13:21:25 -05:00
glm94 288ba9f9cb Updated the docs to include new jump operands and setup the assembler to handle parsing all the jump instructions. 2023-03-29 22:30:12 -05:00
glm94 a9a50055a4 Updated the parser to validate most of the opcodes. NOP falls through via the default case since it takes no parameters but the jump instructions need some thought before proceeding. 2023-03-29 20:57:04 -05:00
glm94 2e888913be Added support for the STORE directive. 2023-03-29 12:57:07 -05:00
glm94 ee60d14570 Refactored the code and added expect functions to make the code flow a bit nicer. 2023-03-29 12:41:53 -05:00
glm94 fa12a04bb3 Moved some of the grunt work to their own functions to clean the code up a bit. 2023-03-28 22:14:31 -05:00
glm94 2d7aad1617 Fixed a bug with labels not advancing the parser after being processed resulting in a redefinition error. 2023-03-28 12:41:56 -05:00
glm94 ae03991fa5 Fixed some parser bugs processing the syntax for some LOAD instruction types. 2023-03-27 22:22:01 -05:00
glm94 95ed2143ea Updated the parser to handle processing a simple load byte instruction. Added the LODB opcode to the opcode enum listing and updated the Scanner to look for 'byte' and mark it as a directive. 2023-03-24 23:24:24 -05:00
glm94 f96e1c6380 Added grammar checking for DB declarations. 2023-03-08 21:38:33 -06:00
glm94 4f9f4202b2 Wrote the first batch of directive handling code, so far just handling creating variables. 2023-03-08 13:25:46 -06:00
glm94 4230708f15 Added a function to remove an item from the generic list, which is mainly used for the tokens. 2023-03-07 21:31:14 -06:00
glm94 5e9e1bdec4 Massive refactoring to simplify this whole setup. The Token List will be modified in place and reused in the Parser to normalize the list and generate a symbols table. Normalizing the token list will make sure the syntax is valid and the symbols table will have the relative offsets and length of symbol values. At least this is all the plan but one step at a time. 2023-03-06 21:51:36 -06:00
glm94 1b246e90bc Started rethinking how this machine should behave. Updated and refactoring some things with a few new ideas. 2023-02-23 22:22:07 -06:00
glm94 9a602824b6 Added all the instructions supported to a structure to help make validation of parameters easier. 2022-10-18 21:14:50 -05:00
glm94 6d6bf1cbaf Wrote the first step for the binary output. Doesn't handle instruction pointer symbols yet since those emit addresses that still need to be computed. 2022-10-10 18:36:38 +00:00
glm94 bca2ad67f1 Fixed a bug with single parameter instructions that land at the end of a file expecting a comma. 2022-10-08 22:02:51 -05:00
glm94 fd44e03b1b Updated the Parser to try and give 'AddressParameter's more meaning depending on the SymbolType. 2022-10-08 18:21:11 -05:00
glm94 48a77950be Updated the way string symbols are handled, allowing the programmer to terminate a string with any byte or none at all. 2022-10-08 16:35:05 -05:00
glm94 8b6edc39be Updated the Parser to handle the variable declaration directive. 2022-10-08 16:15:10 -05:00
glm94 c142d2c6d0 Cleaned up the Symbol struct and updated code accordingly. 2022-10-08 14:43:00 -05:00
glm94 4b08707e67 Cleaned up the list implementation so as to NOT copy items into a new buffer. 2022-10-07 20:46:55 +00:00
glm94 bdb4d2b241 Updated the way Instructions keep track of symbols. Only their name is important but we need the Symbols to be able to track which Instruction they point to (if they do that is, i.e. a label symbol). 2022-10-06 21:11:22 -05:00
glm94 d53833c25f Added some debug print out for the instruction list and a helper function to get a string for any Registers enum. 2022-10-06 21:05:57 +00:00
17 changed files with 1162 additions and 576 deletions
+3 -3
View File
@@ -1,5 +1,5 @@
CC = gcc
CFLAGS=-g -Wall -DDEBUG -Wpedantic
CFLAGS=-g -Wall -DDEBUG -Wpedantic -Wextra -Wunused-result -std=c99 -pedantic-errors
SRCDIR=src
OBJDIR=obj
SRCS=$(wildcard $(SRCDIR)/*.c)
@@ -11,7 +11,7 @@ BIN=$(BINDIR)/assm
all: $(BIN)
release: CFLAGS=-Wall -Wpedantic -O2
release: CFLAGS=-Wall -Wpedantic -std=c99 -pedantic-errors -O2
release: clean
release: $(BIN)
@@ -35,7 +35,7 @@ clean:
rm -rf $(BINDIR)/* $(OBJDIR)/*
test:
$(BIN) misc/test.asm
$(BIN) misc/another_test.asm
disass:
objdump -S --disassemble $(OBJDIR)/$(FILE).o > $(OBJDIR)/$(FILE).s
+98 -55
View File
@@ -8,12 +8,12 @@
\title{Unnamed Machine}
\author{A Very Terrible 16-bit Machine}
\newcommand{\OpcodeTable}[4] {
\begin{tabular}{ c c c c }
\newcommand{\OpcodeTable}[5] {
\begin{tabular}{ c c c c c }
\hline
Opcode & Mnemonic & Operand 1 & Operand 2 \\
Opcode & Bit Pattern & Mnemonic & Operand 1 & Operand 2 \\
\hline\hline
#1 & #2 & #3 & #4 \\
#1 & #2 & #3 & #4 & #5 \\
\hline
\end{tabular}
}
@@ -53,92 +53,135 @@
\end{figure}
\chapter{Instruction Set Architecture}
\section{Instruction Encoding}
Instructions are fixed to exactly one byte (8 bits).
Instructions that work with two operands the register for operand one will be encoded in the three least significant bits. So an instruction with format XXXX X000 will use Register 1 and so forth all the way to XXXX X111, which will be Register 8.
%https://tex.stackexchange.com/questions/32598/force-latex-image-to-appear-in-the-section-in-which-its-declared
The instruction encoding is fixed width to exactly 8 bites wide.
The encoding for the instructions will differ depending on if the opcode requires a register as its first operand.
For instructions that require, as their first argument, a register the encoding format will be as shown in Figure \ref{fig:WithRegister}.
The lower 3 bits may be any bit pattern as long as at least one bit is set in the upper 5 bits.
%https://tex.stackexchange.com/questions/32598/force-latex-image-to-appear-in-the-section-in-which-its-declared
\begin{figure}[!htb]
\centering
\begin{tabular}{ c c }
\hline
Instruction & Register \\
\hline\hline
0000 0 & 000 \\
\hline
XXXX X & 000 \\
\hline
\end{tabular}
\caption{Encoding Layout}
\caption{Encoding Layout, Register Required}
\label{fig:WithRegister}
\end{figure}
Figure \ref{fig:NoRegisterEncoding} shows how encoding will look for instructions that lack arguments or do not require a register.
In contrast with the previous encoding scheme the upper 5 bits MUST be zero, allowing 7 possible instructions to use this format.
All zeroes is not considered a legal instruction.
\begin{figure}[!htb]
\centering
\begin{tabular}{ c c }
Must Be Zero & Instruction \\
\hline
0000 0 & XXX \\
\hline
\end{tabular}
\caption{Encoding Layout, No Register Required}
\label{fig:NoRegisterEncoding}
\end{figure}
\section{Notes}
For the opcodes that load or store data at the assembly language level we could have the mnemonics
"store" and "load" and have the assembler pick the opcode based on the inclusion of the word "byte"
or "word" for two bytes. That would make the assembly easier to read but put a bit more work on the
assembler.
\section{STOB (Store Byte)}
\OpcodeTable{0x20}{stob}{Register}{Address}\\[6pt]
Stores a single byte (the lower nibble) from a register to a memory address.
\section{STOW (Store Machine Word)}
\OpcodeTable{0x20}{stow}{Register}{Address}\\[6pt]
Stores a machine word from a register to a memory address.
\section{LAA (Load Absolute Address)}
\OpcodeTable{0x00}{laa}{Register}{Address}\\[6pt]
Loads an address (2 bytes) into the register.
\section{LODB (Load Byte)}
\OpcodeTable{0x20}{lodb}{Register}{Register}\\[6pt]
Loads a single byte into a register from a memory address in the second operand register, zeroing out the high nibble.
\section{LODW (Load Machine Word)}
\OpcodeTable{0x20}{lodw}{Register}{Register}\\[6pt]
Loads a word into a register from a memory address in the second operand register.
\section{LODWI (Load Immediate Word)}
\OpcodeTable{0x00}{lodwi}{Register}{Constant}\\[6pt]
Loads an immediate machine word into the register clearing the high nibble if the value is less then 256.
assembler. The Stack Pointer will start 2 bytes above the video memory start.
\section{COPY (Copy Word from Address)}
\OpcodeTable{0x08}{0000 1000}{copya}{Register}{Address}\\[6pt]
Copy a machine word from Operand 2 into Operand 1.
\section{COPY BYTE (Copy Byte from Address)}
\OpcodeTable{0x10}{0001 0000}{copyab}{Register}{Address}\\[6pt]
Copy a byte (8 bits) from Operand 2 into Operand 1, clearing the setting the most significant bits to zero.
\section{COPY (Copy Word Indirect Address)}
\OpcodeTable{0x18}{0001 1000}{copyra}{Register}{[Register]}\\[6pt]
Copy a machine word from the address stored in Operand 2 into Operand 1.
\section{COPY BYTE (Copy Byte Indirect Address)}
\OpcodeTable{0x20}{0010 0000}{copyrab}{Register}{[Register]}\\[6pt]
Copy a byte (8 bits) from the address stored in Operand 2 into Operand 1.
\section{COPY (Copy)}
\OpcodeTable{0x28}{0010 1000}{copyrara}{[Register]}{[Register]}\\[6pt]
Copy a machine word from the address stored in Operand 2 into the address stored in Operand 1.
\section{COPY BYTE (Copy Byte)}
\OpcodeTable{0x30}{0011 0000}{copyrarab}{[Register]}{[Register]}\\[6pt]
Copy a byte (8 bits) from the address stored in Operand 2 into the address stored in Operand 1.
\section{COPY (Copy)}
\OpcodeTable{0x38}{0011 1000}{copy}{Register}{Register}\\[6pt]
Copy a machine word from Operand 2 into Operand 1.
\section{COPY (Copy Immediate)}
\OpcodeTable{0xC0}{1100 0000}{copyi}{Register}{Constant}\\[6pt]
Copy a machine word from Operand 2 into Operand 1.
\section{COPY BYTE (Copy Byte)}
\OpcodeTable{0x40}{0100 0000}{copyb}{Register}{Constant}\\[6pt]
Copy a byte (8 bits) from Operand 2 into Operand 1.
\section{CMP (Compare)}
\OpcodeTable{0x20}{cmp}{Register}{Register}\\[6pt]
\OpcodeTable{0x48}{0100 1000}{cmp}{Register}{Register}\\[6pt]
Compares two registers and somewhere sets a result in the status register.
\section{CMPI (Compare Immediate)}
\OpcodeTable{0x00}{cmpi}{Register}{Constant}\\[6pt]
\OpcodeTable{0x50}{0101 0000}{cmpi}{Register}{Constant}\\[6pt]
Compares an immediate 2 byte value to the contents of a register setting the status register accordingly.
\section{ADD (Add)}
\OpcodeTable{0x20}{add}{Register}{Register}\\[6pt]
\OpcodeTable{0x58}{0101 1000}{add}{Register}{Register}\\[6pt]
Performs addition on a register with a value from another (or the same) register.
\section{SUB (Subtract)}
\OpcodeTable{0x20}{sub}{Register}{Register}\\[6pt]
\OpcodeTable{0x60}{0110 0000}{sub}{Register}{Register}\\[6pt]
Performs subtraction on a register with a value from another (or the same) register.
\section{JMP (Jump)}
\OpcodeTable{0x20}{jmp}{Address}{None}\\[6pt]
Jumps unconditionally to a memory address.
\section{JZ (Jump if Zero)}
\OpcodeTable{0x20}{jz}{Register}{None}\\[6pt]
Jumps to a memory address if the status flag is zero.
\section{JG (Jump if Greater Than)}
\OpcodeTable{0x70}{jg}{Address}{None}\\[6pt]
Jump to the Address if the status flag is greater than zero.
\section{JL (Jump if Less Than)}
\OpcodeTable{0x00}{jl}{Address}{None}\\[6pt]
Jumps to the address if the status flag is less than zero.
\section{AND (Logical AND)}
\OpcodeTable{0x00}{and}{Register}{Register}\\[6pt]
\OpcodeTable{0x68}{0110 1000}{and}{Register}{Register}\\[6pt]
Logical ANDs the two registers together storing the result in operand 1.
\section{XOR (Logical Exclusive OR)}
\OpcodeTable{0x00}{xor}{Register}{Register}\\[6pt]
\OpcodeTable{0x70}{0111 0000}{xor}{Register}{Register}\\[6pt]
Logical XORs the two registers together storing the result in operand 1.
\section{OR (Logical OR)}
\OpcodeTable{0x00}{or}{Register}{Register}\\[6pt]
\OpcodeTable{0x78}{0111 1000}{or}{Register}{Register}\\[6pt]
Logical ORs the two registers together storing the result in operand 1.
\section{NOT (Logical Negation)}
\OpcodeTable{0x00}{not}{Register}{None}\\[6pt]
\OpcodeTable{0x80}{1000 0000}{not}{Register}{None}\\[6pt]
Inverts the bits of the target register.
\section{SHR (Shift Right)}
\OpcodeTable{0x00}{shr}{Register}{Constant}\\[6pt]
\OpcodeTable{0x88}{1000 1000}{shr}{Register}{Constant}\\[6pt]
Bit-wise shifts the contents of the register right Constant number of times.
\section{SHL (Shift Left)}
\OpcodeTable{0x00}{shl}{Register}{Constant}\\[6pt]
\OpcodeTable{0x90}{1001 0000}{shl}{Register}{Constant}\\[6pt]
Bit-wise shifts the contents of the register left Constant number of times.
\section{INC (Increment)}
\OpcodeTable{0x00}{inc}{Register}{None}\\[6pt]
\OpcodeTable{0x98}{1001 1000}{inc}{Register}{None}\\[6pt]
Increments the contents of the register by one. Over-flows will not be reported.
\section{DEC (Decrement)}
\OpcodeTable{0x00}{dec}{Register}{None}\\[6pt]
\OpcodeTable{0xA0}{1010 0000}{dec}{Register}{None}\\[6pt]
Decrements the contents of the register by one. Under-flows will not be reported.
\section{Push}
\OpcodeTable{0xA8}{1010 1000}{push}{Register}{None}
Pushes the value of Register onto the stack, decrementing the Stack Pointer by 2.
\section{Pop}
\OpcodeTable{0xB0}{1011 0000}{pop}{Register}{None}
Pops the top of the stack into Register, incrementing the Stack Pointer by 2.
\section{JMPI (Jump Indirect)}
\OpcodeTable{0xB8}{1011 1000}{jmpi}{Register}{None}\\[6pt]
Jumps unconditionally to a memory address stored in Operand 1. Sets the Program Counter to Operand 1.
\section{JMP (Jump)}
\OpcodeTable{0x02}{0000 0010}{jmp}{Address}{None}\\[6pt]
Jumps unconditionally to a memory address. Sets the Program Counter to Address.
\section{JZ (Jump if Zero)}
\OpcodeTable{0x03}{0000 0011}{jz}{Address}{None}\\[6pt]
Jumps to a memory address if the status flag is zero. Sets the Program Counter to Address.
\section{JG (Jump if Greater Than)}
\OpcodeTable{0x04}{0000 0100}{jg}{Address}{None}\\[6pt]
Jump to the Address if the status flag is greater than zero. Sets the Program Counter to Address.
\section{JL (Jump if Less Than)}
\OpcodeTable{0x05}{0000 0101}{jl}{Address}{None}\\[6pt]
Jumps to the address if the status flag is less than zero. Sets the Program Counter to Address.
\section{NOP (No Operation)}
\OpcodeTable{0x00}{nop}{None}{None}\\[6pt]
\OpcodeTable{0x01}{0000 0001}{nop}{None}{None}\\[6pt]
Skips a clock cycle, incrementing the program counter.
\section{CALL (Call Subroutine)}
\OpcodeTable{0x06}{0000 0110}{call}{Address}{None}\\[6pt]
Pushes the base address to the stack and sets the Program Counter to Address.
\section{RET (Return from Subroutine)}
\OpcodeTable{0x07}{0000 0111}{ret}{None}{None}\\[6pt]
Pops the stack and sets the Program Counter to that value.
\end{document}
-21
View File
@@ -1,21 +0,0 @@
#ifndef LIST_H
#define LIST_H
#include <stddef.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
#define LISTDEFAULTSIZE 4
typedef struct {
void** content;
int size;
int capacity;
} List;
List* CreateList(void);
int AddListItem(const void *, size_t, List *);
void DestroyList(List*);
#endif
+8 -32
View File
@@ -2,47 +2,23 @@
#define OPCODES_H
#include <string.h>
#include "symbols_table.h"
#include <errno.h>
#include <stdio.h>
typedef enum {
R1, R2, R3, R4, R5, R6, R7, R8
R1 = 0, R2, R3, R4, R5, R6, R7, R8 = 7
} Registers;
typedef enum {
ADD, SUB, JZ, INT, YLD, RET,
CMP, NOP, JMP, CALL, LOAD, JE, INC, DEC, LOADB
COPYA = 0x08, COPYAB = 0x10, COPYRA = 0x18, COPYRAB = 0x20, COPYRARA = 0x28, COPYRARAB = 0x30, COPY = 0x38, COPYB = 0x40, COPYI = 0xC0,
CMP = 0x48, CMPI = 0x50, ADD = 0x58, SUB = 0x60, AND = 0x68, XOR = 0x70, OR = 0x78, NOT = 0x80, SHR = 0x88, SHL = 0x90,
INC = 0x98, DEC = 0xA0, PUSH = 0xA8, POP = 0xB0,
JMPI = 0xB8, JMP = 0x02, JZ = 0x03, JG = 0x04, JL = 0x05, NOP = 0x01, CALL = 0x06, RET = 0x07
} Mnemonic;
typedef enum {
NoParameter,
ConstantParameter,
AddressParameter,
RegisterParameter
} OpcodeParameter;
typedef struct {
OpcodeParameter ParameterType;
OpcodeParameter InterpretedAs;
union {
int Number; //ConstanrParam
Registers Register; //Register param
Symbol* Symbol; //Address param
} Value;
} Parameter;
typedef struct {
Mnemonic Mnemonic;
Parameter* ParameterOne;
Parameter* ParameterTwo;
} Instruction;
Instruction* CreateInstruction(Mnemonic mnemonic);
Parameter* CreateParameter(OpcodeParameter parameterType);
void FreeInstruction(Instruction* instruction);
int IsOpcode(const char*, Mnemonic*);
int IsRegister(const char*, Registers*);
void GetMnemonicText(Mnemonic mnemonic, char buffer[12]);
unsigned char GetInstructionMask(Mnemonic type);
void GetRegisterText(Registers reg, char buffer[3]);
#endif
+1 -9
View File
@@ -6,18 +6,10 @@
#include <stdlib.h>
#include <string.h>
#include <stdarg.h>
#include "list.h"
#include "token.h"
#include "opcodes.h"
#include "symbols_table.h"
//#define HIGHMEMORY 65535 //64KiB - 1 AKA 0xFFFF
typedef struct {
List* Instructions;
SymbolTable* SymbolsTable;
} IRState;
IRState* ParseTokens(List*);
void ParseTokens(TokenList* tokens, SymbolTable** symbolsTable);
#endif
+1 -2
View File
@@ -7,9 +7,8 @@
#include <errno.h>
#include "stdlib.h"
#include "token.h"
#include "list.h"
#include "opcodes.h"
List* GenerateTokenList(const char*);
TokenList* GenerateTokenList(const char*);
#endif
+19 -15
View File
@@ -1,28 +1,29 @@
#ifndef SYMBOLSTABLE_H
#define SYMBOLSTABLE_H
#include <stdint.h>
#include <stdlib.h>
#include <errno.h>
#include <stdio.h>
#include <string.h>
#include <sys/types.h>
#include "opcodes.h"
#include "token.h"
#define SYMBOLSTABLE_DEFAULT_CAPACITY 128
typedef enum {
ValueAt,
Address
} SymbolType;
typedef struct __token_node {
const Token* Token;
struct __token_node* Left;
struct __token_node* Right;
} TokenNode;
typedef struct {
char* Name;
typedef struct _symbol {
const char* Name;
int Address;
int Length;
int Resolved;
union {
char* Text;
int Number;
//Instruction Instruction;
} Value;
const Token* Token;
TokenNode* ValueExpression;
} Symbol;
typedef struct {
@@ -32,9 +33,12 @@ typedef struct {
} SymbolTable;
SymbolTable* CreateSymbolTable(void);
Symbol* TryGetSymbol(char* name, SymbolTable* table);
Symbol* AddSymbolToTable(char* name, void* value, int length, SymbolTable* table);
TokenNode* CreateTokenNode(const Token* token);
int TryGetSymbol(const char* name, const SymbolTable* table, Symbol** outSymbol);
Symbol* AddSymbolToTable(const char* name, int address, SymbolTable* table);
void FreeSymbolTable(SymbolTable* table);
void FreeSymbol(Symbol* symbol);
int SymbolResolved(const Symbol* symbol, const SymbolTable* table);
int TryGetSymbolValue(const Symbol* symbol, const SymbolTable* table, unsigned short* value);
#endif
+15 -3
View File
@@ -23,7 +23,7 @@ typedef enum {
typedef enum {
DB,
Origin,
Include,
Byte
} Directive;
@@ -39,7 +39,7 @@ typedef enum {
AddressClass = LabelClass | IdentifierClass
} TokenClass;
typedef struct {
typedef struct __token {
char* Lemexe;
TokenClass Class;
int LineNumber;
@@ -50,11 +50,23 @@ typedef struct {
Registers Register;
Mnemonic Mnemonic;
Directive Directive;
int Number;
unsigned short Number;
} Value;
struct __token* Prev;
struct __token* Next;
} Token;
typedef struct {
Token** content;
int size;
int capacity;
} TokenList;
Token* CreateToken(int lineNumber, TokenClass tokenClass);
TokenList* CreateTokenList(void);
int AddToken(Token* token, TokenList* list);
void RemoveToken(int index, TokenList* list);
void FreeToken(Token* token);
#endif
+44
View File
@@ -0,0 +1,44 @@
.db MAX_MEM 0xFFFF
.db VIDEO_MEM 0xF37F
.db MAX_LENGTH MAX_MEM; - MAX_LENGTH
.db MSG "Hello, World!", 0
__start:
copy r1, MSG ; Pointer into r1
copy r8, VIDEO_MEM
call strlen
copy r1, MSG
cmp r2, 0
jz _end
cmp r2, MAX_LENGTH
jg _end
draw_loop:
copy byte [r8], [r1]
inc r1
inc r8
dec r2
cmp r2, 0
jz _end
jmp draw_loop
_end:
jmp _end
.db NewMsg "My message", 0
; Returns the length of a NULL terminated string
; Arguments: R1 - Pointer to the string
; Returns: R2 - Contains the length of the string
strlen:
copy r2, 0 ; length
copy byte r3, [r1]
cmp r3, 0
jz end ;The string is zero length
loop:
inc r1
copy byte r3, [r1]
cmp r3, 0
jz end
inc r2
jmp loop
end:
ret
+25 -25
View File
@@ -1,27 +1,27 @@
;.include "./another_test.asm"
.db labelsz reference
.db video_start 0xF37F
.db msg "Hello, world!", 0
load r1, [msg] ; Because the lod* instructions can't load an address
; from anything but a register, this becomes laa r1, [msg]
;Routine: string_length
;In: String address in r1
;Out: Length in R2
string_length:
loadb r3, [r1] ; Load the byte from the address in R1, into R3
load r2, 0 ; String length
cmp r3, 0 ; Is R3 a null byte?
je end
inc r2
start:
inc r1 ; next char
loadb r3, [r1] ; Load the next character byte into R3
cmp r3, 0 ; Null byte?
je end
inc r2 ; Nope, increment the length counter
jmp start
end:
ret
.db NOTERM "No terminating byte here"
.db null_byte 0
jmp [r4]
;load r2, label
something_insance:
load r1, unknown_symbol
load r1, 5
.db fun_alright 0x70;does this break?
load r3, r4
store byte r3, r5
store r5, r7
unknown_symbol:;Does this work?
cmp r1, 5;or one after the register?
cmp r1, r2;This might work
and r1, r2
add r4, r7
inc r1 ;increment r1
xor r1, r1 ;clear self
pop r3
jmp unknown_symbol
jmp [r4]
jz [r6]
;And a comment at the end
-66
View File
@@ -1,66 +0,0 @@
#include "../includes/list.h"
List* CreateList() {
List *new = malloc(sizeof(List));
if (!new) {
fprintf(stderr, "Failed to malloc() for new new List.\n");
return NULL;
}
new->content = malloc(sizeof(void*) * LISTDEFAULTSIZE);
if (!new->content) {
fprintf(stderr, "Failed to malloc() memory for List contents.\n");
free(new);
return NULL;
}
new->size = 0;
new->capacity = LISTDEFAULTSIZE;
return new;
}
int AddListItem(const void *value, size_t size, List* list) {
if (!list) return -1;
if (!value) return -1;
if (size == 0) return -1;
if (list->capacity < list->size + 1) {
void* ptr = realloc(list->content, sizeof(void*) * list->capacity * 2);
//Note: realloc will free list->root if it succeeds.
if (!ptr) {
fprintf(stderr, "Failed to resize array with realloc() (%d bytes).\n", list->capacity * 2);
return -1;
}
list->content = ptr;
list->capacity = list->capacity * 2;
}
void* item = calloc(1, size);
if (!item) {
fprintf(stderr, "Failed to calloc() new memory (%lu bytes).\n", size);
return -1;
}
memcpy(item, value, size);
list->content[list->size] = item;
list->size++;
return 0;
}
void DestroyList(List* list) {
if (!list) return;
for(int i = 0; i < list->size; i++) {
free(list->content[i]);
}
free(list->content);
free(list);
}
+176 -25
View File
@@ -3,10 +3,18 @@
#include <stdio.h>
#include <string.h>
#include <sysexits.h>
#include "../includes/list.h"
#include "../includes/parser.h"
#include "../includes/token.h"
#include "../includes/futil.h"
#include "../includes/scanner.h"
#include "../includes/parser.h"
const char* MagicStartName = "__start";
TokenList* LIST;
SymbolTable* Symbols = NULL;
unsigned char mem[128] = {0};
void print(void);
void assemble(void);
int main(int argc, char* args[]) {
if (argc == 1) {
@@ -14,16 +22,36 @@ int main(int argc, char* args[]) {
return EX_USAGE;
}
atexit(print);
char* source_code;
size_t bytes_read;
if (!ReadAllString(args[1], &source_code, &bytes_read)) return EX_IOERR;
List* list = GenerateTokenList(source_code);
char mnemonic[12];
LIST = GenerateTokenList(source_code);
ParseTokens(LIST, &Symbols);
for(int i = 0; i < list->size; i++) {
Token* t = (Token*) list->content[i];
free(source_code);
assemble();
printf("\nBytes:\n");
for(unsigned long i = 0; i < sizeof(mem); i++) {
if (i != 0 && i % 8 == 0) printf("\n");
printf("%02X ", mem[i] & 0xFF);
}
printf("\n");
}
void print(void) {
char mnemonic[12];
printf("printing tokens...\n");
for(int i = 0; i < LIST->size; i++) {
Token* t = (Token*) LIST->content[i];
if (t->EndOfFile) {
printf("EOF\n");
@@ -32,20 +60,16 @@ int main(int argc, char* args[]) {
if (t->Class == PunctuationClass){
if(t->Value.Punctuation == NewLine) {
printf("\n");
printf("<%d>\n", t->LineNumber);
continue;
}
printf("<%d>", t->LineNumber);
printf("%c ", t->Value.Punctuation);
printf("[P]%c", t->Value.Punctuation);
continue;
}
printf("<%d>", t->LineNumber);
if (t->Class == LabelClass) {
printf("[L]%s* ", t->Lemexe);
printf("[L]%s*", t->Lemexe);
continue;
}
@@ -76,22 +100,149 @@ int main(int argc, char* args[]) {
}
if (t->Class == CharacterClass) {
printf("%s ", t->Lemexe);
printf("[C]'%s'", t->Lemexe);
}
}
}
int CurrentIndex = 0;
int PC2 = 0;
void WriteByte(unsigned char byte) {
mem[PC2] = byte;
PC2++;
}
void WriteWord(unsigned short word) {
mem[PC2] = (word >> 8) & 0xFF;
mem[PC2 + 1] = word & 0xFF;
PC2 += 2;
}
unsigned short GetValueFromToken(Token* token) {
if (token->Class & IdentifierClass) {
Symbol* symbol;
if (TryGetSymbol(token->Lemexe, Symbols, &symbol)) {
return symbol->Address;
}
else {
fprintf(stderr, "Failed to find symbol %s while assembling.\n", token->Lemexe);
exit(1);
}
}
else if (token->Class == NumberClass) return token->Value.Number;
IRState* image = ParseTokens(list);
//free(image);
//Disassemble(image);
//for(int i = 0; i < image->Opcodes->size; i++) {
// printf("%d\n", ((IROpcode*) image->Opcodes->content[i])->Mnemonic);
//}
fprintf(stderr, "Failed to find token '%d' while assmbling.\n", token->Class);
//FILE* program = fopen("program.bin", "w+b");
exit(1);
}
//fwrite(image, 1, image[2] + image[3], program);
int IsAtEnd(void){
return CurrentIndex >= LIST->size;
}
//fclose(program);
void assemble() {
while(!IsAtEnd()) {
Token* current = LIST->content[CurrentIndex];
switch(current->Value.Mnemonic) {
case COPYA:
case COPYAB:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case COPYRA:
case COPYRAB:
case COPYRARA:
case COPYRARAB:
case COPY:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteByte(((Token*) LIST->content[CurrentIndex])->Value.Register);
break;
case COPYI: //TODO: should explicitly get number from value I feel
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case COPYB:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteByte((unsigned char)GetValueFromToken(LIST->content[CurrentIndex]));
break;
case CMP:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteByte(((Token*) LIST->content[CurrentIndex])->Value.Register);
break;
case CMPI:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case ADD:
case SUB:
case AND:
case OR:
case XOR:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteByte(((Token*) LIST->content[CurrentIndex])->Value.Register);
break;
case SHL:
case SHR:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case INC:
case DEC:
case PUSH:
case POP:
case JMPI:
CurrentIndex++;
WriteByte(current->Value.Mnemonic | ((Token*) LIST->content[CurrentIndex])->Value.Register);
break;
case JMP:
case JZ:
case JG:
case JL:
WriteByte(current->Value.Mnemonic);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
break;
case NOP:
case RET:
WriteByte(current->Value.Mnemonic);
break;
case CALL:
WriteByte(current->Value.Mnemonic);
CurrentIndex++;
WriteWord(GetValueFromToken(LIST->content[CurrentIndex]));
default:
break;
}
CurrentIndex++;
}
}
free(source_code);
}
+49 -155
View File
@@ -1,99 +1,50 @@
#include "../includes/opcodes.h"
#include <ctype.h>
#include <stdlib.h>
#include <string.h>
#define OPCODECOUNT 15
#define OPCODECOUNT 34
struct _instruction {
char* Name;
Mnemonic Mnemonic;
};
struct _instruction instructions[OPCODECOUNT] = {
{ "add", ADD },//, Reg, Reg | Constant },
{ "sub", SUB },//, Reg, Reg | Constant },
{ "jz", JZ },//, Address, None },
{ "int", INT },//, Constant, None },
{ "yld", YLD },//, None, None },
{ "ret", RET },//, None, None },
{ "cmp", CMP },//, Reg, Reg | Constant },
{ "inc", INC },//, Constant | Reg, None},
{ "dec", DEC },//, None, None},
{ "nop", NOP },//, None, None},
{ "jmp", JMP },//, Address, None},
{ "call", CALL },//, Address, None}
{ "load", LOAD },
{ "je", JE },
{ "loadb", LOADB}
struct _instruction instructions[31] = {
{ "copya", COPYA },
{ "copyab", COPYAB },
{ "copyra", COPYRA },
{ "copyrab", COPYRAB },
{ "copyrara", COPYRARA },
{ "copyrarab", COPYRARAB },
{ "copy", COPY },
{ "copyi", COPYI },
{ "copyb", COPYB },
{ "cmp", CMP },
{ "cmpi", CMPI },
{ "add", ADD },
{ "sub", SUB },
{ "and", AND },
{ "xor", XOR },
{ "or", OR },
{ "not", NOT },
{ "shr", SHR },
{ "shl", SHL },
{ "inc", INC },
{ "dec", DEC },
{ "push", PUSH },
{ "pop", POP },
{ "jmpi", JMPI },
{ "jmp", JMP },
{ "jz", JZ },
{ "jg", JG },
{ "jl", JL },
{ "nop", NOP },
{ "call", CALL },
{ "ret", RET }
//yld
};
Instruction* CreateInstruction(Mnemonic mnemonic) {
Instruction* instruction = calloc(1, sizeof(Instruction));
if (!instruction) {
fprintf(stderr, "Failed to calloc room for an Instruction. %s.\n", strerror(errno));
return NULL;
}
instruction->Mnemonic = mnemonic;
// instruction->ParameterOne = CreateParameter(NoParameter);
// instruction->ParameterTwo = CreateParameter(NoParameter);
return instruction;
}
Parameter* CreateParameter(OpcodeParameter parameterType) {
Parameter* param = calloc(1, sizeof(Parameter));
if (!param) {
fprintf(stderr, "Failed to calloc room for a Parameter. %s,\n", strerror(errno));
return NULL;
}
param->ParameterType = parameterType;
return param;
}
void FreeInstruction(Instruction* instruction) {
if (!instruction) return;
free(instruction);
}
unsigned char GetInstructionMask(Mnemonic type) {
switch (type) {
case LOAD:
return 0x20; //0b00100000; ORing 0x80 marks the first parameter as an address
case ADD:
return 0x40;//0b01000000;
case SUB:
return 0x60;//0b01100000;
case CMP:
return 0x80;//0b10000000;
case JZ:
return 0x01;//0b00000001;
case INT:
return 0x02; //0b00000010;
case YLD:
return 0x03;
case RET:
return 0x04;
case CALL:
return 0x05;
case JMP:
return 0x06;//0b00000110;
case INC:
return 0x07;//0b00000111;
case DEC:
return 0x08;//0b00001000;
default:
return 0x00;
}
}
void GetMnemonicText(Mnemonic mnemonic, char buffer[12]) {
memset(buffer, '\0', 12);
@@ -105,62 +56,19 @@ void GetMnemonicText(Mnemonic mnemonic, char buffer[12]) {
}
}
/*
NOP - 0000 0000 NOP
JZ - 0000 0001 JZ Address
INT - 0000 0010 INT Constant
YLD - 0000 0011 YLD
RET - 0000 0100 RET
CALL - 0000 0101 CALL Address
JMP - 0000 0110 JMP Address
IN - 0000 0111 IN Constant
OUT - 0000 1000 OUT Constant
COPY - 0010 0XXX COPY REG, REG
- 0010 1XXX COPY REG, Constant
- 0011 0XXX COPY REG, Address
- 1010 0XXX COPY Address, REG
- 1010 1XXX COPY Address, Constant
- 1011 0XXX COPY Address, Address
ADD - 0100 0XXX ADD REG, REG
- 0100 1XXX ADD REG, Constant
SUB - 0110 0XXX SUB REG, REG
- 0110 1XXX SUB REG, Constant
CMP - 1000 0XXX CMP REG, REG
- 1000 1XXX CMP REG, Constant
- 1001 0XXX CMP REG, Address
*/
//COPY X01X XXXX
//ADD 010X XXXX
//SUB 011X XXXX
//CMP 100X XXXX
//REG XXX0 0XXX
//Constant XXX0 1XXX
//Address XXX1 0XXX
//R1 XXXX X000 -> XXXX X111 (R1 to R8)
void GetRegisterText(Registers reg, char buffer[3]) {
memset(buffer, '\0', 3);
// Register registers[REGISTERCOUNT] = {
// { "r1", R1 },
// { "r2", R2 },
// { "r3", R3 },
// { "r4", R4 },
// { "r5", R5 },
// { "r6", R6 },
// { "r7", R7 },
// { "r8", R8 }
// };
if (reg < R1 || reg > R8) return;
// const Instruction* GetOpcodeDetails(Mnemonic type) {
// for(int i = 0; i < OPCODECOUNT; i++) {
// if (instructions[i].op == type) return &instructions[i];
// }
// return NULL;
// }
buffer[0] = 'r';
buffer[1] = reg + 49;
}
int IsOpcode(const char* text, Mnemonic* opcode) {
if (!text) return 0;
for(int i = 0; i < OPCODECOUNT; i++) {
for(unsigned long i = 0; i < sizeof(instructions) / sizeof(struct _instruction); i++) {
if (strcmp(instructions[i].Name, text) == 0) {
if (opcode) *opcode = instructions[i].Mnemonic;
return 1;
@@ -179,25 +87,11 @@ int IsRegister(const char* text, Registers* reg) {
if (length != 2) return 0;
if (text[0] != 'r') return 0;
switch(text[1]) {
case '1':
r--;
case '2':
r--;
case '3':
r--;
case '4':
r--;
case '5':
r--;
case '6':
r--;
case '7':
r--;
case '8':
if (reg) *reg = r;
return 1;
default:
return 0;
}
if (!isdigit(text[1])) return 0;
r = text[1] - 0x31;
if (reg) *reg = r;
return 1;
}
+503 -137
View File
@@ -1,213 +1,503 @@
#include "../includes/parser.h"
#include <stdio.h>
#include <stdlib.h>
#include <string.h>
const List* TokensList;
typedef enum {
NoOptions = 0, ForwardParser = 1, RemoveExpected = 2
} ExpectOptions;
TokenList* TokensList;
int CurrentToken = 0;
void PrintSymbols(void);
void HandleOperation(void);
void RemoveCurrentToken(void);
void HandleAssemblerDirective(void);
Token* ExpectMnemonic(void);
void AdvanceParser(void);
void IgnoreParserLine(void);
Token* PeekToken(void);
int ParserAtEnd(void);
IRState MachineState;
IRState* ParseTokens(List* tokens) {
if (!tokens) return NULL;
void ExpectPuncuation(TokenPunctuation punctuation, ExpectOptions options);
void ExpectRegister(void);
Symbol* ExpectIdentifier(ExpectOptions options, int expectUnresolved);
void ExpectCharacterClass(Symbol* symbol);
Token* ExpectLineEndOrFileEnd(ExpectOptions options);
Token* ExpectMathOperator(ExpectOptions options);
Token* ExpectMathOperand(ExpectOptions options);
SymbolTable* SymbolsTable;
TokenNode* ParseSymbolExpression(void);
//int HeapSize = 0;
int PC = 0;
void ParseTokens(TokenList* tokens, SymbolTable** symbols) {
if (!tokens) return;
if (!symbols) return;
TokensList = tokens;
MachineState.SymbolsTable = CreateSymbolTable();
MachineState.Instructions = CreateList();
if (!*symbols) *symbols = CreateSymbolTable();
SymbolsTable = *symbols;
while(!ParserAtEnd()) {
Token* t = PeekToken();
if (!t || t->EndOfFile) break;
switch(t->Class) {
case DirectiveClass: //Maybe these should be ignored, let another process handle that.
IgnoreParserLine();
case DirectiveClass:
HandleAssemblerDirective();
break;
case MnemonicClass:
HandleOperation();
ExpectMnemonic();
break;
case LabelClass:
AdvanceParser();
AddSymbolToTable(t->Lemexe, NULL, strlen(t->Lemexe), MachineState.SymbolsTable)->Resolved = 1;
AdvanceParser(); //Consume the NewLine
{
Symbol* symbol;
int found = TryGetSymbol(t->Lemexe, SymbolsTable, &symbol);
if (found && SymbolResolved(symbol, SymbolsTable)) {
fprintf(stderr, "[Error] Line %d: Redefinition of symbol '%s'\n", t->LineNumber, t->Lemexe);
exit(1);
}
if (!found) symbol = AddSymbolToTable(t->Lemexe, PC, SymbolsTable);
symbol->Address = PC;
RemoveCurrentToken(); //label
ExpectLineEndOrFileEnd(RemoveExpected);
symbol->Token = ExpectMnemonic();
}
break;
default:
fprintf(stderr, "[Warning] Line %d: Syntax error, expected start of expression, got '%c' [%d].\n", t->LineNumber, t->Value.Punctuation, t->Class);
//exit(1);
IgnoreParserLine();
exit(1);
break;
}
}
if (MachineState.SymbolsTable->Size > 0) PrintSymbols();
for(int i = 0; i < MachineState.Instructions->size; i++) {
char mn[12];
Instruction* ins = MachineState.Instructions->content[i];
GetMnemonicText(ins->Mnemonic, mn);
printf("%s\n", mn);
}
return &MachineState;
PrintSymbols();
}
Parameter* GetParameterType() {
Token* token = PeekToken();
Symbol* symbol = NULL;
Parameter* param = CreateParameter(NoParameter);
void HandleAssemblerDirective() {
Token* directive = PeekToken();
switch(token->Class) {
case RegisterClass:
param->ParameterType = RegisterParameter;
param->InterpretedAs = RegisterParameter;
switch(directive->Value.Directive){
case DB:
{
RemoveCurrentToken();
param->Value.Register = token->Value.Register;
Symbol* symbol = ExpectIdentifier(RemoveExpected, 1);
if (PeekToken()->Class == CharacterClass) {
ExpectCharacterClass(symbol);
}
else if (PeekToken()->Class == NumberClass || PeekToken()->Class == IdentifierClass) {
symbol->Length = 2;
symbol->ValueExpression = ParseSymbolExpression();
}
}
break;
default:
fprintf(stderr, "Synatx error on line %d, %s.\n", directive->LineNumber, directive->Lemexe);
exit(1);
}
}
Token* ExpectMnemonic(void) {
Token* opcode = PeekToken();
if (opcode->Class != MnemonicClass) {
fprintf(stderr, "Syntax error on line %d: expected mnemonic.\n", opcode->LineNumber);
exit(1);
}
AdvanceParser();
switch(opcode->Value.Mnemonic) {
case COPY:
{
Token* arg = PeekToken();
int isByte = 0;
if (arg->Class == DirectiveClass) {
if (arg->Value.Directive != Byte) {
fprintf(stderr, "Syntax error on line %d: expected keyword 'byte'.\n", opcode->LineNumber);
exit(1);
}
RemoveCurrentToken(); //byte
isByte = 1;
arg = PeekToken();
}
PC++;
if (arg->Class == RegisterClass) {
ExpectRegister();
ExpectPuncuation(Comma, RemoveExpected);
arg = PeekToken();
if (arg->Class == NumberClass) {
opcode->Value.Mnemonic = isByte ? COPYB : COPYI;
opcode->Lemexe = isByte ? "copyb" : "copyi";
isByte ? PC++ : (PC += 2);
AdvanceParser();
}
else if (arg->Class == RegisterClass) {
if (isByte) {
fprintf(stderr, "Syntax error on line %d: unexpected modifier 'Byte' for Register to Register copy.\n", opcode->LineNumber);
exit(1);
}
ExpectRegister();
opcode->Value.Mnemonic = COPY;
opcode->Lemexe = "copy";
PC++;
}
else if (arg->Class & IdentifierClass) {
opcode->Value.Mnemonic = isByte ? COPYAB : COPYA;
opcode->Lemexe = isByte ? "copyab" : "copya";
PC += 2;
ExpectIdentifier(ForwardParser, 0);
}
else {
ExpectPuncuation(LBracket, RemoveExpected);
ExpectRegister();
ExpectPuncuation(RBracket, RemoveExpected);
opcode->Value.Mnemonic = isByte ? COPYRAB : COPYRA;
opcode->Lemexe = isByte ? "copyrab" : "copyra";
PC++;
}
}
else {
ExpectPuncuation(LBracket, RemoveExpected);
ExpectRegister();
ExpectPuncuation(RBracket, RemoveExpected);
ExpectPuncuation(Comma, RemoveExpected);
ExpectPuncuation(LBracket, RemoveExpected);
ExpectRegister();
ExpectPuncuation(RBracket, RemoveExpected);
opcode->Value.Mnemonic = isByte ? COPYRARAB : COPYRARA;
opcode->Lemexe = isByte ? "copyrarab" : "copyrara";
PC++;
}
}
break;
case CMP:
{
ExpectRegister();
ExpectPuncuation(Comma, RemoveExpected);
TokenClass class = PeekToken()->Class;
PC++;
if (class == NumberClass) {
opcode->Value.Mnemonic = CMPI;
opcode->Lemexe = "CMPI";
AdvanceParser();
PC += 2;
}
else if (class & IdentifierClass) {
opcode->Value.Mnemonic = CMPI;
opcode->Lemexe = "CMPI";
ExpectIdentifier(ForwardParser, 0);
PC += 2;
}
else {
ExpectRegister();
PC++;
}
}
break;
case ADD:
case SUB:
case AND:
case OR:
case XOR:
PC++;
ExpectRegister();
ExpectPuncuation(Comma, RemoveExpected);
ExpectRegister();
PC++;
break;
case SHL:
case SHR:
PC++;
ExpectRegister();
ExpectPuncuation(Comma, RemoveExpected);
if (PeekToken()->Class != NumberClass) {
fprintf(stderr, "[Line %d] Syntax error, expected numeric literal.\n", PeekToken()->LineNumber);
exit(12);
}
AdvanceParser();
return param;
case LabelClass:
case IdentifierClass:
param->ParameterType = AddressParameter;
param->InterpretedAs = AddressParameter;
PC += 2;
break;
case INC:
case DEC:
case PUSH:
case POP:
PC++;
ExpectRegister();
break;
case JMP:
{
PC++;
symbol = TryGetSymbol(token->Lemexe, MachineState.SymbolsTable);
Token* arg = PeekToken();
if (!symbol) symbol = AddSymbolToTable(token->Lemexe, token->Lemexe, strlen(token->Lemexe), MachineState.SymbolsTable);
if (arg->Class & IdentifierClass) {
ExpectIdentifier(ForwardParser, 0);
param->Value.Symbol = symbol;
opcode->Value.Mnemonic = JMP;
opcode->Lemexe = "jmp";
AdvanceParser();
PC += 2;
}
else
{
ExpectPuncuation(LBracket, RemoveExpected);
return param;
case PunctuationClass:
if (token->Value.Punctuation != LBracket) {
fprintf(stderr, "[Error] Line %d: Expected opening bracket\n", token->LineNumber);
exit(1);
ExpectRegister();
ExpectPuncuation(RBracket, RemoveExpected);
opcode->Value.Mnemonic = JMPI;
opcode->Lemexe = "jmpi";
}
}
break;
case JZ:
{
PC++;
AdvanceParser(); // [
ExpectIdentifier(ForwardParser, 0);
token = PeekToken();
PC += 2;
if (token->Class == NumberClass) {
param->ParameterType = ConstantParameter;
param->InterpretedAs = AddressParameter;
param->Value.Number = token->Value.Number;
opcode->Value.Mnemonic = JZ;
opcode->Lemexe = "jz";
}
break;
case JG:
{
PC++;
ExpectIdentifier(ForwardParser, 0);
PC += 2;
opcode->Value.Mnemonic = JG;
opcode->Lemexe = "jg";
}
else if (token->Class == RegisterClass) {
param->ParameterType = RegisterParameter;
param->InterpretedAs = AddressParameter;
param->Value.Register = token->Value.Register;
break;
case JL:
{
PC++;
ExpectIdentifier(ForwardParser, 0);
PC += 2;
opcode->Value.Mnemonic = JL;
opcode->Lemexe = "jl";
}
else if (token->Class & AddressClass) {
param->ParameterType = AddressParameter;
param->ParameterType = AddressParameter;
break;
case CALL:
{
PC++;
symbol = TryGetSymbol(token->Lemexe, MachineState.SymbolsTable);
Token* arg = PeekToken();
if (!symbol) symbol = AddSymbolToTable(token->Lemexe, NULL, strlen(token->Lemexe), MachineState.SymbolsTable);
if ((arg->Class & IdentifierClass) == 0) {
fprintf(stderr, "Syntax error on line %d: expected subroutine call target.\n", opcode->LineNumber);
param->Value.Symbol = symbol;
exit(1);
}
ExpectIdentifier(ForwardParser, 0);
PC += 2;
}
else {
fprintf(stderr, "[Error] Line %d: Expected identifier, constant number or register.\n", token->LineNumber);
IgnoreParserLine();
//return NULL;
exit(1);
}
AdvanceParser(); //Consume the parameter it self.
token = PeekToken();
if (token->Class != PunctuationClass || token->Value.Punctuation != RBracket) {
fprintf(stderr, "[Error] Line %d: Expected closing bracket.\n", token->LineNumber);
exit(1);
}
AdvanceParser(); // ]
return param;
case NumberClass:
param->ParameterType = ConstantParameter;
param->InterpretedAs = ConstantParameter;
param->Value.Number = token->Value.Number;
AdvanceParser();
return param;
break;
default:
break;
}
return param;
ExpectLineEndOrFileEnd(ForwardParser);
return opcode;
}
void HandleOperation() {
void ExpectPuncuation(TokenPunctuation punctuation, ExpectOptions options) {
Token* token = PeekToken();
char mn[12];
Instruction* ins = CreateInstruction(token->Value.Mnemonic);
GetMnemonicText(ins->Mnemonic, mn);
AdvanceParser();
//No parameters here, we have a line break.
if (PeekToken()->EndOfFile || (PeekToken()->Class == PunctuationClass && PeekToken()->Value.Punctuation == NewLine)) {
goto InsertInstruction;
}
ins->ParameterOne = GetParameterType();
if (ins->ParameterOne->ParameterType == NoParameter) {
//This would be an error if the next token isn't a line break.
if (PeekToken()->Class != PunctuationClass || PeekToken()->Value.Punctuation != NewLine) {
fprintf(stderr, "[Error] Line %d: Syntax error, expected line break but got %s.\n", token->LineNumber, token->Lemexe);
exit(1);
}
goto InsertInstruction;
}
if (PeekToken()->Class == PunctuationClass && PeekToken()->Value.Punctuation == NewLine) {
goto InsertInstruction;
}
if (PeekToken()->Class != PunctuationClass || PeekToken()->Value.Punctuation != Comma) {
//Syntax error
fprintf(stderr, "[Error] Line %d: Expected comma.\n", token->LineNumber);
if (token->Class != PunctuationClass || token->Value.Punctuation != punctuation) {
fprintf(stderr, "[Line %d] Syntac error, expected %c.\n", token->LineNumber, punctuation);
exit(1);
}
AdvanceParser(); //Consume the comma
if (options & RemoveExpected) RemoveCurrentToken();
if (options & ForwardParser) AdvanceParser();
}
ins->ParameterTwo = GetParameterType();
void ExpectRegister() {
Token* token = PeekToken();
InsertInstruction:
if (token->Class != RegisterClass) {
fprintf(stderr, "[Line %d] Syntax error, expected Register.\n", token->LineNumber);
exit(2);
}
AddListItem(ins, sizeof(Instruction), MachineState.Instructions);
AdvanceParser();
}
AdvanceParser(); //Consume the line break
Symbol* ExpectIdentifier(ExpectOptions options, int expectUnresolved) {
Token* token = PeekToken();
free(ins);
if (token->Class != IdentifierClass) {
fprintf(stderr, "[Line %d] Expected Identifier.\n", token->LineNumber);
exit(3);
}
Symbol* symbol;
int found = TryGetSymbol(token->Lemexe, SymbolsTable, &symbol);
if (found && TryGetSymbolValue(symbol, SymbolsTable, NULL) && expectUnresolved) {
fprintf(stderr, "[Line %d] Redefinition of symbol '%s'.\n", token->LineNumber, token->Lemexe);
exit(4);
}
if (!found) symbol = AddSymbolToTable(token->Lemexe, PC, SymbolsTable);
if (expectUnresolved) symbol->Token = token;
if (options & RemoveExpected) RemoveCurrentToken();
if (options & ForwardParser) AdvanceParser();
return symbol;
}
void ExpectCharacterClass(Symbol* symbol) {
Token* token = PeekToken();
if (token->Class != CharacterClass) {
fprintf(stderr, "[Line %d] Syntax error, expected character string.\n", token->LineNumber);
exit(5);
}
symbol->Length = strlen(token->Lemexe);
symbol->Token = token;
RemoveCurrentToken(); //Remove the string declared by this DB command.
token = PeekToken();
if (token->EndOfFile) return;
if (token->Class == PunctuationClass && token->Value.Punctuation == NewLine) {
RemoveCurrentToken();
return;
}
ExpectPuncuation(Comma, RemoveExpected);
token = PeekToken();
if (token->Class != NumberClass) {
fprintf(stderr, "[Line %d] Syntax error, expected terminating byte.\n", token->LineNumber);
exit(7);
}
RemoveCurrentToken(); //Remove terminating byte.
//TODO: add the raw value of the byte to the end of the string, but for now just pretend all numbers are zero.
symbol->Length++;
ExpectLineEndOrFileEnd(RemoveExpected);
}
Token* ExpectLineEndOrFileEnd(ExpectOptions options) {
Token* token = PeekToken();
if (token->EndOfFile) return token;
if (token->Class != PunctuationClass || token->Value.Punctuation != NewLine) {
fprintf(stderr, "[Line %d] Syntax error, expected line break.\n", token->LineNumber);
exit(9);
}
if (options & RemoveExpected) RemoveCurrentToken();
if (options & ForwardParser) AdvanceParser();
return token;
}
void PrintSymbols(void) {
char mn[12];
printf("-----SYMBOLS-----\n");
for(int i = 0; i < MachineState.SymbolsTable->Size; i++) {
Symbol* symbol = MachineState.SymbolsTable->Symbols[i];
for(int i = 0; i < SymbolsTable->Size; i++) {
Symbol* symbol = SymbolsTable->Symbols[i];
unsigned short value = 0;
int resolved = SymbolResolved(symbol, SymbolsTable);
printf("[Resolved? %d] %s\n", symbol->Resolved, symbol->Name);
if (resolved) TryGetSymbolValue(symbol, SymbolsTable, &value);
printf("[%s] %s [Value: %d]", resolved == 0 ? "Unresolved" : "Resolved", symbol->Name, value);
if (symbol->Token && symbol->Token->Class == MnemonicClass)
{
GetMnemonicText(symbol->Token->Value.Mnemonic, mn);
printf(" -> [%s]", mn);
}
if (symbol->Token) printf(" Line %d", symbol->Token->LineNumber);
printf("\n");
}
printf("-----SYMBOLS-----\n");
}
@@ -239,4 +529,80 @@ void IgnoreParserLine(void) {
AdvanceParser();
}
}
void RemoveCurrentToken(void) {
RemoveToken(CurrentToken, TokensList);
}
TokenNode* ParseSymbolExpression(void) {
TokenNode* root = CreateTokenNode(ExpectMathOperand(RemoveExpected));
while(!ParserAtEnd()) {
if (PeekToken()->Class == PunctuationClass && PeekToken()->Value.Punctuation == NewLine) {
ExpectPuncuation(NewLine, RemoveExpected);
break;
}
TokenNode* value = CreateTokenNode(ExpectMathOperand(RemoveExpected));
TokenNode* operation = CreateTokenNode(ExpectMathOperator(RemoveExpected));
operation->Left = root;
operation->Right = value;
root = operation;
}
return root;
}
Token* ExpectMathOperator(ExpectOptions options) {
if (ParserAtEnd()) {
fprintf(stderr, "Syntax error, expected math operator but found unexpected end of file.\n");
exit (1);
}
Token* current = PeekToken();
if (current->Class != PunctuationClass) {
fprintf(stderr, "[Line %d] Syntax error, expected math operator.\n", current->LineNumber);
exit (1);
}
switch (current->Value.Punctuation) {
case Plus:
case Minus:
case Star:
case Slash:
case Power:
break;
default:
fprintf(stderr, "[Line %d] Syntax error, expected math operator 2.\n", current->LineNumber);
exit(1);
}
if (options & RemoveExpected) RemoveCurrentToken();
if (options & ForwardParser) AdvanceParser();
return current;
}
Token* ExpectMathOperand(ExpectOptions options) {
if (ParserAtEnd()) {
fprintf(stderr, "Syntax error, expected math operand but found unexpected end of file.\n");
exit (1);
}
Token* token = PeekToken();
if (token->Class != NumberClass && token->Class != IdentifierClass) {
fprintf(stderr, "[Line %d] Syntax error, expected a number or identifier.\n", token->LineNumber);
exit (1);
}
if (options & RemoveExpected) RemoveCurrentToken();
if (options & ForwardParser) AdvanceParser();
return token;
}
+41 -18
View File
@@ -21,8 +21,8 @@ Token* ParseNumber(void);
Token* ParsePunctuation(char);
void IgnoreLine(void);
List* GenerateTokenList(const char* source) {
List* tokens = CreateList();
TokenList* GenerateTokenList(const char* source) {
TokenList* tokens = CreateTokenList();
Token* token = NULL;
if (!tokens) return NULL;
@@ -54,7 +54,7 @@ List* GenerateTokenList(const char* source) {
token = CreateToken(Line, PunctuationClass);
token->Value.Punctuation = NewLine;
AddListItem(token, sizeof(Token), tokens);
AddToken(token, tokens);
AdvanceScanner();
Line++;
break;
@@ -67,30 +67,30 @@ List* GenerateTokenList(const char* source) {
Line++;
}
break;
case '.': //directive like ".org" or ".db"
case '.': //directive like ".include" or ".db"
token = ParseDirective();
if (token) AddListItem(token, sizeof(Token), tokens);
if (token) AddToken(token, tokens);
break;
case '"':
token = ParseString();
if (token) AddListItem(token, sizeof(Token), tokens);
if (token) AddToken(token, tokens);
break;
default:
if (isdigit(c)) {
AddListItem(ParseNumber(), sizeof(Token), tokens);
AddToken(ParseNumber(), tokens);
break;
}
if (IsPunctuation(c)) {
AdvanceScanner();
AddListItem(ParsePunctuation(c), sizeof(Token), tokens);
AddToken(ParsePunctuation(c), tokens);
break;
}
token = ParseIdentifier();
if (token) AddListItem(token, sizeof(Token), tokens);
if (token) AddToken(token, tokens);
break;
}
@@ -111,7 +111,7 @@ List* GenerateTokenList(const char* source) {
token->EndOfFile = 1;
AddListItem(token, sizeof(Token), tokens);
AddToken(token, tokens);
return tokens;
}
@@ -128,7 +128,7 @@ Token* ParseNumber(void) {
if ((ahead >= 'a' && ahead <= 'f') || isdigit(ahead)) {
AdvanceScanner(); //Consume the 'x'
while(!ScannerAtEnd() && PeekScanner() != '\n' && !IsWhiteSpace(PeekScanner())) {
while(!ScannerAtEnd() && PeekScanner() != '\n' && !IsWhiteSpace(PeekScanner()) && PeekScanner() != ';') {
if (isdigit(PeekScanner())) {
AdvanceScanner();
continue;
@@ -191,17 +191,16 @@ Token* ParseDirective(void) {
return token;
}
else if (strcmp(directive, ".org") == 0) {
fprintf(stderr, "[Warning] Org is not a supported directive.\n");
free(directive);
free(token);
IgnoreLine();
return NULL;
else if (strcmp(directive, ".include") == 0) {
token->Value.Directive = Include;
return token;
}
fprintf(stderr, "[Error] Unknown assembler directive '%s'\n", directive);
free(directive);
free(token);
IgnoreLine();
@@ -244,7 +243,7 @@ Token* ParseString(void) {
Token* ParseIdentifier(void) {
int start = Position;
while(!ScannerAtEnd() && !IsPunctuation(PeekScanner()) && PeekScanner() != ' ' && PeekScanner() != '\n') {
while(!ScannerAtEnd() && !IsPunctuation(PeekScanner()) && PeekScanner() != ' ' && PeekScanner() != '\n' && PeekScanner() != ';') {
AdvanceScanner();
}
@@ -284,6 +283,14 @@ Token* ParseIdentifier(void) {
return token;
}
if (strcmp(lexeme, "byte") == 0) {
Token* token = CreateToken(Line, DirectiveClass);
token->Lemexe = lexeme;
token->Value.Directive = Byte;
return token;
}
Token* token = CreateToken(Line, IdentifierClass);
@@ -321,6 +328,10 @@ int IsPunctuation(char c) {
case '(':
case ')':
case ',':
case '-':
case '+':
case '*':
case '/':
return 1;
default:
return 0;
@@ -359,6 +370,18 @@ Token* ParsePunctuation(char c) {
case ',':
punctuation = Comma;
break;
case '-':
punctuation = Minus;
break;
case '+':
punctuation = Plus;
break;
case '*':
punctuation = Star;
break;
case '/':
punctuation = Slash;
break;
default:
return NULL;
}
+114 -10
View File
@@ -3,6 +3,18 @@
#include <stdlib.h>
#include <string.h>
int SymbolResolved(const Symbol* symbol, const SymbolTable* table) {
if (!symbol) return 0;
const Token* token = symbol->Token;
if (!token) return 0;
if (token->Class == LabelClass || token->Class == CharacterClass || token->Class == MnemonicClass) return 1;
return TryGetSymbolValue(symbol, table, NULL);
}
SymbolTable* CreateSymbolTable(void){
SymbolTable* table = calloc(1, sizeof(SymbolTable));
@@ -25,7 +37,7 @@ SymbolTable* CreateSymbolTable(void){
return table;
}
Symbol* CreateSymbol(char* name, void* value, int length) {
Symbol* CreateSymbol(const char* name, int address) {
Symbol* symbol = calloc(1, sizeof(Symbol));
if (!symbol) {
@@ -34,24 +46,29 @@ Symbol* CreateSymbol(char* name, void* value, int length) {
return NULL;
}
symbol->Length = length;
symbol->Address = address;
symbol->Name = name;
//symbol->Value = value;
//symbol->Type = RefUnknown;
return symbol;
}
Symbol* TryGetSymbol(char* name, SymbolTable* table) {
if (!name || !table) return NULL;
int TryGetSymbol(const char* name, const SymbolTable* table, Symbol** outSymbol) {
*outSymbol = NULL;
if (!name || !table) return 0;
for(int i = 0; i < table->Size; i++) {
if (strcmp(table->Symbols[i]->Name, name) == 0) return table->Symbols[i];
if (strcmp(table->Symbols[i]->Name, name) == 0) {
*outSymbol = table->Symbols[i];
return 1;
}
}
return NULL;
return 0;
}
Symbol* AddSymbolToTable(char* name, void* value, int length, SymbolTable* table) {
Symbol* AddSymbolToTable(const char* name, int address, SymbolTable* table) {
//if (!name || !value || !table || length == 0) return NULL;
for(int i = 0; i < table->Size; i++) {
@@ -63,7 +80,7 @@ Symbol* AddSymbolToTable(char* name, void* value, int length, SymbolTable* table
}
if (table->Capacity < table->Size + 1) {
Symbol** newBlock = realloc(table->Symbols, sizeof(Symbol*) * table->Capacity * 2);//calloc(table->Size * 2, sizeof(Symbol*));
Symbol** newBlock = realloc(table->Symbols, sizeof(Symbol*) * table->Capacity * 2);
if (!newBlock) {
fprintf(stderr, "Failed to realloc space for a new symbol '%s'. %s.\n", name, strerror(errno));
@@ -75,7 +92,7 @@ Symbol* AddSymbolToTable(char* name, void* value, int length, SymbolTable* table
table->Symbols = newBlock;
}
Symbol* symbol = CreateSymbol(name, value, length);
Symbol* symbol = CreateSymbol(name, address);
table->Symbols[table->Size] = symbol;
table->Size++;
@@ -95,4 +112,91 @@ void FreeSymbol(Symbol* symbol) {
if (!symbol) return;
free(symbol);
}
int TryGetTokenNodeValue(const TokenNode* node, const SymbolTable* table, unsigned short* value);
unsigned short DoOp(unsigned short left, unsigned short right, TokenPunctuation op);
int TryGetSymbolValue(const Symbol* symbol, const SymbolTable* table, unsigned short* value) {
if (!symbol || !symbol->Token || !table) return 0;
TokenNode* root = symbol->ValueExpression;
Symbol* s = NULL;
if (value) *value = 0;
if (!root || !root->Token) return 0;
switch (root->Token->Class) {
case NumberClass:
if (value) *value = root->Token->Value.Number;
return 1;
case IdentifierClass:
if (!TryGetSymbol(root->Token->Lemexe, table, &s)) return 0;
return TryGetTokenNodeValue(s->ValueExpression, table, value);
case PunctuationClass:
return TryGetTokenNodeValue(root, table, value);
default:
return 0;
}
}
int TryGetTokenNodeValue(const TokenNode* node, const SymbolTable* table, unsigned short* value) {
const Token* token = node->Token;
Symbol* symbol = { 0 };
if (token->Class == NumberClass) {
if (value) *value = token->Value.Number;
return 1;
}
if (token->Class == IdentifierClass) {
if (!TryGetSymbol(token->Lemexe, table, &symbol)) return 0;
return TryGetTokenNodeValue(symbol->ValueExpression, table, value);
}
if (token->Class != PunctuationClass) return 0;
unsigned short left = 0;
unsigned short right = 0;
if (!TryGetTokenNodeValue(node->Left, table, &left)) return 0;
if (!TryGetTokenNodeValue(node->Right, table, &right)) return 0;
if (value) *value = DoOp(left, right, token->Value.Punctuation);
return 1;
}
TokenNode* CreateTokenNode(const Token* token){
TokenNode* node = calloc(1, sizeof(TokenNode));
if (!node) {
fprintf(stderr, "Failed to calloc memory for a TokenNode. %s.\n", strerror(errno));
return NULL;
}
node->Token = token;
return node;
}
unsigned short DoOp(unsigned short left, unsigned short right, TokenPunctuation op) {
switch (op) {
case Plus:
return left + right;
case Minus:
return left - right;
case Star:
return left * right;
case Slash:
return left / right;
default:
break;
}
return 0;
}
+65
View File
@@ -1,5 +1,7 @@
#include "../includes/token.h"
#define LISTDEFAULTSIZE 32
Token* CreateToken(int lineNumber, TokenClass tokenClass) {
Token* token = calloc(1, sizeof(Token));
@@ -18,4 +20,67 @@ void FreeToken(Token* token) {
if (!token) return;
free(token);
}
TokenList* CreateTokenList(void) {
TokenList *new = malloc(sizeof(TokenList));
if (!new) {
fprintf(stderr, "Failed to malloc() for new new List.\n");
return NULL;
}
new->content = calloc(LISTDEFAULTSIZE, sizeof(void*));
if (!new->content) {
fprintf(stderr, "Failed to malloc() memory for List contents.\n");
free(new);
return NULL;
}
new->size = 0;
new->capacity = LISTDEFAULTSIZE;
return new;
}
int AddToken(Token* token, TokenList* list) {
if (!list || !token) return 0;
if (list->capacity < list->size + 1) {
void* ptr = realloc(list->content, sizeof(void*) * list->capacity * 2);
//Note: realloc will free list->root if it succeeds.
if (!ptr) {
fprintf(stderr, "Failed to resize array with realloc() (%d bytes).\n", list->capacity * 2);
return 0;
}
list->content = ptr;
list->capacity *= 2;
}
if (list->size > 0)
{
Token* prev = list->content[list->size - 1];
token->Prev = prev;
prev->Next = token;
}
list->content[list->size] = token;
list->size++;
return 1;
}
void RemoveToken(int index, TokenList* list) {
Token* token = list->content[index];
if (token->Prev) token->Prev->Next = token->Next;
memmove(&list->content[index], &list->content[index + 1], (list->size - index) * sizeof(Token*));
list->size--;
list->content[list->size] = NULL;
}