all repos — nand2tetris @ aba8a91d9655ef4ca547d4d1eaa7dec05441bdf2

my nand2tetris progress

projects/06/assembler1/assembler1.c

 1
 2
 3
 4
 5
 6
 7
 8
 9
 10
 11
 12
 13
 14
 15
 16
 17
 18
 19
 20
 21
 22
 23
 24
 25
 26
 27
 28
 29
 30
 31
 32
 33
 34
 35
 36
 37
 38
 39
 40
 41
 42
 43
 44
 45
 46
 47
 48
 49
 50
 51
 52
 53
 54
 55
 56
 57
 58
 59
 60
 61
 62
 63
 64
 65
 66
 67
 68
 69
 70
 71
 72
 73
 74
 75
 76
 77
 78
 79
 80
 81
 82
 83
 84
 85
 86
 87
 88
 89
 90
 91
 92
 93
 94
 95
 96
 97
 98
 99
 100
 101
 102
 103
 104
 105
 106
 107
 108
 109
 110
 111
 112
 113
 114
 115
 116
 117
 118
 119
 120
 121
 122
 123
 124
 125
 126
 127
 128
 129
 130
 131
 132
 133
 134
 135
 136
 137
 138
 139
 140
 141
 142
 143
 144
 145
 146
 147
 148
 149
 150
 151
 152
 153
 154
 155
 156
 157
 158
 159
 160
 161
 162
 163
 164
 165
 166
 167
 168
 169
 170
 171
 172
 173
 174
 175
 176
 177
 178
 179
 180
 181
 182
 183
 184
 185
 186
 187
 188
 189
 190
 191
 192
 193
 194
 195
 196
#include <stdbool.h>
#include <stdint.h>
#include <stdio.h>
#include <stdlib.h>
#include <string.h>

#include "../bindump.h"

#define DEBUG(...)              printf(__VA_ARGS__)
#define die(err_msg)            perror(err_msg); exit(-1)
#define alert(...)              fprintf(stderr, __VA_ARGS__)

#define MAX_LINE_LEN            256


uint32_t myatoi(const char *a_field_str)
{
	size_t i;
	uint32_t ret = 0;

	for (i = 0; i < 5 && '0' <= a_field_str[i] && a_field_str[i] <= '9'; ++i) {
		ret = (ret * 10) + (a_field_str[i] - 0x30);
	}

	return ret;
}

bool parse_a_type(const char *line, uint16_t *instruction)
{
	char c, a_field_str[6];  // TODO: eventually factor out use of array
	uint32_t a_field = 0;
	size_t i, a = 0;

	if (line[0] != '@') {
		alert("error: A-type instruction doesn't start with @\n");
		return false;
	}

	if (line[1] == '\0') {
		alert("error: A-type instruction empty after @\n");
		return false;
	}

	for (i = 1; (c = line[i]) != '\0' && a < 6; ++i) {
		if ('0' <= c && c <= '9') {
			if (a > 4) {
				alert("error: @<number> too long\n");
				return false;
			}
			a_field_str[a] = c; // get number
			a++;
		} else if ((c == ' ' || c == '\t' || c == '/') && i > 1) {
			break;
		} else {       // any other character
			alert("syntax error: invalid char '%c' found after @\n",
			      c);
			return false;
		}
	}

	a_field_str[a] = '\0';   // exit

	// TODO: extension: support negative numbers
	a_field = myatoi(a_field_str);
	if (a_field > 32767) {
		alert("error: %u > 32767, too large\n", a_field);
		return false;
	}

	*instruction = 0x0000 | (uint16_t) a_field;
	return true; // STUB, A-type MSB == 0 anyway
}

bool parse_c_type(const char *line, uint16_t *instruction)
{
	// Note: could do 'jump' fields via lookup tables once
	// they're extracted from the line
	*instruction = 0x8888; // STUB, TODO implement
	return true;
}

// does not care about line line length; exits at first newline or after
// relevant portion parsed (allows for syntactically-incorrect lines, I know)
bool parse_next_instruction(const char *line, uint16_t *instruction)
{
	bool ret;
	char c;
	size_t i = 0;

	while ((c = line[i]) != '\0') {
		if (c == ' ' || c == '\t')
			;  // skip any whitespace at start of line
		else if (c == '@') {
			ret = parse_a_type(&line[i], instruction);
			break;
		} else if (c >= '!' && c < '~') {
			ret = parse_c_type(&line[i], instruction);
			break;
		} else {
			alert("syntax error: line '%s' incorrectly formatted\n",
			      line);
		}

		++i;
	}

	return ret;
}

// return false for comment or invalid assembly instruction
bool parse_line(const char *line, size_t line_len, uint16_t *instruction)
{
	char c;
	bool slash_found = false;
	size_t i;

	if (line_len == 0 || line_len == 1)
		return false;

	// filter out comment lines
	//for (i = 0; (c = line[i]) != NULL; ++i) {
	for (i = 0; i < line_len; ++i) {
		c = line[i];

		if (c == ' ' || c == '\t') {
			continue;
		} else if (c == '/') {
			if (slash_found) {
				// second slash means this is a comment
				return false;
			}
			slash_found = true;
			continue;
		} else if (slash_found) {
			// this char not slash, but previous was: invalid syntax
			// TODO: add line, column numbers
			alert("syntax error: found '/', comments need '//'\n");
			return false;
		} else {
			// non-whitespace/slash char discovered
			break;
		}
	}

	// comment not found, so attempting to parse instruction
	return parse_next_instruction(line, instruction);
}


char *usage_msg = "Usage: assembler1 [path/to/file.asm]\n";

int main(int argc, char *argv[])
{
	bool result = false; 
	uint16_t instruction;
	char in_line[MAX_LINE_LEN];
	size_t in_line_len, i, count;
	char *in_file_path;
	FILE *fp;

	if (argc != 2) {                            // requires 1 argument
		die(usage_msg);
		exit(-1);
	}

	in_file_path = argv[1];
	fp = fopen(in_file_path, "r");
	if (fp == NULL) {
		die("failed to open file for reading\n");
		exit(-1);
	}

	count = 1;
	while (fgets(in_line, MAX_LINE_LEN, fp) != NULL) {
		in_line_len = strlen(in_line);

		for (i = 0; i < in_line_len; ++i) { // remove newlines
			if (in_line[i] == '\n') {
				in_line[i] = '\0';
				break;
			}
		}

		DEBUG("%lu|%s\n", count, in_line);
		result = parse_line(in_line, in_line_len, &instruction);
		if (result) {
			DEBUG("instruction: 0x%x  |  ", instruction);
			bindump_word16(instruction); // output instruction as binary
			putchar('\n');
		}
	++count;
	}

	return 0;
}