Page Menu
Home
Phorge
Search
Configure Global Search
Log In
Files
F85803733
No One
Temporary
Actions
View File
Edit File
Delete File
View Transforms
Subscribe
Award Token
Flag For Later
Size
41 KB
Referenced Files
None
Subscribers
None
View Options
diff --git a/Makefile b/Makefile
index fe8ef2c..ac1d0f7 100644
--- a/Makefile
+++ b/Makefile
@@ -1,93 +1,102 @@
MIX = mix
-MYHTMLEX_CFLAGS = -g -O2 -std=c99 -pedantic -Wcomment -Wall
+MYHTMLEX_CFLAGS = -g -O2 -std=c99 -pedantic -Wcomment -Wextra -Wno-old-style-declaration -Wall
# we need to compile position independent code
MYHTMLEX_CFLAGS += -fpic -DPIC
# For some reason __erl_errno is undefined unless _REENTRANT is defined
MYHTMLEX_CFLAGS += -D_REENTRANT
# myhtmlex is using stpcpy, as defined in gnu string.h
# MYHTMLEX_CFLAGS += -D_GNU_SOURCE
# base on the same posix c source as myhtml
# MYHTMLEX_CFLAGS += -D_POSIX_C_SOURCE=199309
# turn warnings into errors
# MYHTMLEX_CFLAGS += -Werror
# ignore unused variables
# MYHTMLEX_CFLAGS += -Wno-unused-variable
# ignore unused parameter warnings
MYHTMLEX_CFLAGS += -Wno-unused-parameter
# set erlang include path
ERLANG_PATH = $(shell erl -eval 'io:format("~s", [lists:concat([code:root_dir(), "/erts-", erlang:system_info(version)])])' -s init stop -noshell)
MYHTMLEX_CFLAGS += -I$(ERLANG_PATH)/include
# expecting myhtml as a submodule in c_src/
# that way we can pin a version and package the whole thing in hex
# hex does not allow for non-app related dependencies.
MYHTML_PATH = c_src/myhtml
MYHTML_STATIC = $(MYHTML_PATH)/lib/libmyhtml_static.a
MYHTMLEX_CFLAGS += -I$(MYHTML_PATH)/include
# avoid undefined reference errors to phtread_mutex_trylock
MYHTMLEX_CFLAGS += -lpthread
# that would be used for a dynamically linked build
# MYHTMLEX_CFLAGS += -L$(MYHTML_PATH)/lib
MYHTMLEX_LDFLAGS = -shared
# C-Node
ERL_INTERFACE = $(wildcard $(ERLANG_PATH)/../lib/erl_interface-*)
CNODE_CFLAGS = $(MYHTMLEX_CFLAGS)
CNODE_CFLAGS += -L$(ERL_INTERFACE)/lib
CNODE_CFLAGS += -I$(ERL_INTERFACE)/include
-CNODE_CFLAGS += -lerl_interface -lei -pthread
+
+CNODE_LDFLAGS =
+
+ifeq ($(OTP22_DEF),YES)
+ CNODE_CFLAGS += -DOTP_22_OR_NEWER
+else
+ CNODE_LDFLAGS += -lerl_interface
+endif
+
+CNODE_LDFLAGS += -lei -pthread
# enumerate docker build tests
BUILD_TESTS := $(patsubst %.dockerfile, %.dockerfile.PHONY, $(wildcard ./build-test/*.dockerfile))
# platform specific environment
UNAME = $(shell uname -s)
ifeq ($(UNAME_S),Darwin)
MYHTMLEX_LDFLAGS += -dynamiclib -undefined dynamic_lookup
else
# myhtmlex is using stpcpy, as defined in gnu string.h
MYHTMLEX_CFLAGS += -D_GNU_SOURCE
# base on the same posix c source as myhtml
# MYHTMLEX_CFLAGS += -D_POSIX_C_SOURCE=199309
endif
.PHONY: all
all: myhtmlex
myhtmlex: priv/myhtml_worker
$(MIX) compile
$(MYHTML_STATIC): $(MYHTML_PATH)
$(MAKE) -C $(MYHTML_PATH) library MyCORE_BUILD_WITHOUT_THREADS=YES
priv/myhtml_worker: c_src/myhtml_worker.c $(MYHTML_STATIC)
- $(CC) -o $@ $< $(MYHTML_STATIC) $(CNODE_CFLAGS)
+ $(CC) -o $@ $< $(MYHTML_STATIC) $(CNODE_CFLAGS) $(CNODE_LDFLAGS)
clean: clean-myhtml
$(RM) -r priv/myhtmlex*
$(RM) priv/myhtml_worker
$(RM) myhtmlex-*.tar
$(RM) -r package-test
clean-myhtml:
$(MAKE) -C $(MYHTML_PATH) clean
# publishing the package and docs separately is required
# otherwise the build artifacts are included in the package
# and the tarball gets too big to be published
publish: clean
$(MIX) hex.publish package
$(MIX) hex.publish docs
test:
$(MIX) test
build-tests: test $(BUILD_TESTS)
%.dockerfile.PHONY: %.dockerfile
docker build -f $< .
diff --git a/c_src/myhtml_worker.c b/c_src/myhtml_worker.c
index bbc6d44..10cd038 100644
--- a/c_src/myhtml_worker.c
+++ b/c_src/myhtml_worker.c
@@ -1,432 +1,513 @@
#include <stdlib.h>
#include <stdbool.h>
#include <stdio.h>
#include <string.h>
+#include <stdarg.h>
#include <unistd.h>
#include <sys/types.h>
#include <sys/socket.h>
#include <netinet/in.h>
#include <arpa/inet.h>
#include <errno.h>
#include <ctype.h>
-#include "erl_interface.h"
#include "ei.h"
-
-#include "tstack.h"
+#ifndef OTP_22_OR_NEWER
+# include "erl_interface.h"
+#endif
#include <myhtml/myhtml.h>
#include <myhtml/mynamespace.h>
-#define BUFFER_SIZE 4096
+#include "tstack.h"
+
+#ifdef __GNUC__
+# define AFP(x, y) __attribute__((format (printf, x, y)))
+#else
+# define AFP(x, y)
+#endif
+
+#ifdef __GNUC__
+# define NORETURN __attribute__((noreturn))
+#else
+# define NORETURN
+#endif
typedef struct _state_t {
int fd;
- myhtml_t* myhtml;
+ myhtml_t * myhtml;
+ ei_cnode ec;
+ bool looping;
+ ei_x_buff buffer;
} state_t;
-void
-handle_emsg(state_t* state, ErlMessage* emsg);
-void
-handle_send(state_t* state, ErlMessage* emsg);
-ETERM*
-decode(state_t* state, ErlMessage* emsg, ETERM* bin, ETERM* args);
-ETERM*
-build_tree(myhtml_tree_t* tree, myhtml_tree_node_t* node, unsigned char* parse_flags);
-ETERM*
-build_node_attrs(myhtml_tree_t* tree, myhtml_tree_node_t* node);
-ETERM*
-err_term(const char* error_atom);
-unsigned char
-read_parse_flags(ETERM* list);
-static inline char *
-lowercase(char* c);
-
-const unsigned char FLAG_HTML_ATOMS = 1 << 0;
-const unsigned char FLAG_NIL_SELF_CLOSING = 1 << 1;
-const unsigned char FLAG_COMMENT_TUPLE3 = 1 << 2;
-
-int main(int argc, char **argv) {
- if (argc != 5 || !strcmp(argv[1],"-h") || !strcmp(argv[1],"--help")) {
- printf("\nUsage: ./priv/cnode_server <sname> <hostname> <cookie> <tname>\n\n");
- printf(" sname the short name you want this c-node to connect as\n");
- printf(" hostname the hostname\n");
- printf(" cookie the authentication cookie\n");
- printf(" tname the target node short name to connect to");
- return 0;
- }
+typedef enum parse_flags_e {
+ FLAG_HTML_ATOMS = 1 << 0,
+ FLAG_NIL_SELF_CLOSING = 1 << 1,
+ FLAG_COMMENT_TUPLE3 = 1 << 2
+} parse_flags_t;
+
+static void handle_emsg(state_t * state, erlang_msg * emsg);
+static void handle_send(state_t * state, erlang_msg * emsg);
+static void err_term(ei_x_buff * response, const char * error_atom);
+static parse_flags_t decode_parse_flags(state_t * state, int arity);
+static void decode(state_t * state, ei_x_buff * response, const char * bin_data, size_t bin_size, parse_flags_t parse_flags);
+
+static void build_tree(ei_x_buff * response, myhtml_tree_t * tree, myhtml_tree_node_t * node, parse_flags_t parse_flags);
+static void prepare_node_attrs(ei_x_buff * response, myhtml_tree_node_t * node);
+
+static inline char * lowercase(char * c);
+
+static void panic(const char *fmt, ...) AFP(1, 2);
+static void panic(const char *fmt, ...) {
+ char buf[4096];
+ va_list va;
+
+ va_start (va, fmt);
+ vsnprintf (buf, sizeof buf, fmt, va);
+ va_end (va);
+
+ fprintf (stderr, "myhtml worker: error: %s\n", buf);
+ exit (EXIT_FAILURE);
+}
+
+static void usage (void) NORETURN;
+static void usage (void) {
+ fputs ("usage: myhtml_worker sname hostname cookie tname\n\n"
+ " sname the short name you want this c-node to connect as\n"
+ " hostname the hostname\n"
+ " cookie the authentication cookie\n"
+ " tname the target node short name to connect to\n", stderr);
+ exit (EXIT_FAILURE);
+}
+
+int main(int argc, const char *argv[]) {
+#ifdef OTP_22_OR_NEWER
+ // initialize erlang client library
+ ei_init ();
+#else
+ erl_init (NULL, -1);
+#endif
+
+ if (argc != 5)
+ usage ();
+
+ const char *sname = argv[1];
+ const char *hostname = argv[2];
+ const char *cookie = argv[3];
+ const char *tname = argv[4];
- char *sname = argv[1];
- char *hostname = argv[2];
- char *cookie = argv[3];
- char *tname = argv[4];
char full_name[1024];
char target_node[1024];
- snprintf(full_name, sizeof full_name, "%s@%s", sname, hostname);
- snprintf(target_node, sizeof target_node, "%s@%s", tname, hostname);
+ snprintf (full_name, sizeof full_name, "%s@%s", sname, hostname);
+ snprintf (target_node, sizeof target_node, "%s@%s", tname, hostname);
struct in_addr addr;
addr.s_addr = htonl(INADDR_ANY);
// fd to erlang node
- state_t* state = (state_t*)malloc(sizeof(state_t));
- bool looping = true;
- int buffer_size = BUFFER_SIZE;
- unsigned char* bufferpp = (unsigned char*)malloc(BUFFER_SIZE);
- ErlMessage emsg;
-
- // initialize all of Erl_Interface
- erl_init(NULL, 0);
+ state_t* state = calloc (1, sizeof(state_t));
+ state->looping = true;
+ ei_x_new (&state->buffer);
// initialize this node
- printf("initialising %s\n", full_name); fflush(stdout);
- if ( erl_connect_xinit(hostname, sname, full_name, &addr, cookie, 0) == -1 )
- erl_err_quit("error erl_connect_init");
+ printf ("initialising %s\n", full_name);
+ if (ei_connect_xinit (&state->ec, hostname, sname, full_name, &addr, cookie, 0) == -1)
+ panic ("ei_connect_xinit failed.");
// connect to target node
- printf("connecting to %s\n", target_node); fflush(stdout);
- if ((state->fd = erl_connect(target_node)) < 0)
- erl_err_quit("erl_connect");
+ printf ("connecting to %s\n", target_node);
+ if ((state->fd = ei_connect (&state->ec, target_node)) < 0)
+ panic ("ei_connect failed.");
- state->myhtml = myhtml_create();
- myhtml_init(state->myhtml, MyHTML_OPTIONS_DEFAULT, 1, 0);
+ state->myhtml = myhtml_create ();
+ myhtml_init (state->myhtml, MyHTML_OPTIONS_DEFAULT, 1, 0);
// signal to stdout that we are ready
- printf("%s ready\n", full_name); fflush(stdout);
+ printf ("%s ready\n", full_name);
+ fflush (stdout);
- while (looping)
+ while (state->looping)
{
- // erl_xreceive_msg adapts the buffer width
- switch( erl_xreceive_msg(state->fd, &bufferpp, &buffer_size, &emsg) )
- // erl_receive_msg, uses a fixed buffer width
- /* switch( erl_receive_msg(state->fd, buffer, BUFFER_SIZE, &emsg) ) */
+ erlang_msg emsg;
+
+ switch (ei_xreceive_msg (state->fd, &emsg, &state->buffer))
{
case ERL_TICK:
- // ignore
break;
case ERL_ERROR:
- // On failure, the function returns ERL_ERROR and sets erl_errno to one of:
- //
- // EMSGSIZE
- // Buffer is too small.
- // ENOMEM
- // No more memory is available.
- // EIO
- // I/O error.
- //
- // TODO: what is the correct reaction?
- looping = false;
+ panic ("ei_xreceive_msg: %s\n", strerror (erl_errno));
break;
default:
- handle_emsg(state, &emsg);
+ handle_emsg (state, &emsg);
+ break;
}
}
- // shutdown: free all erlang terms still around
- erl_eterm_release();
- free(bufferpp);
-
- myhtml_destroy(state->myhtml);
- free(state);
+ // shutdown: free all state
+ ei_x_free (&state->buffer);
+ myhtml_destroy (state->myhtml);
+ free (state);
return EXIT_SUCCESS;
}
-void
-handle_emsg(state_t* state, ErlMessage* emsg)
+// handle an erlang_msg structure and call handle_send() if relevant
+static void handle_emsg (state_t * state, erlang_msg * emsg)
{
- switch(emsg->type)
+ state->buffer.index = 0;
+
+ switch (emsg->msgtype)
{
case ERL_REG_SEND:
case ERL_SEND:
- handle_send(state, emsg);
+ handle_send (state, emsg);
break;
case ERL_LINK:
case ERL_UNLINK:
break;
case ERL_EXIT:
break;
}
- // its our responsibility to free these pointers
- erl_free_compound(emsg->msg);
- erl_free_compound(emsg->to);
- erl_free_compound(emsg->from);
}
-void
-handle_send(state_t* state, ErlMessage* emsg)
+// handle ERL_SEND message type.
+// we expect a tuple with arity of 3 in state->buffer.
+// we expect the first argument to be an atom (`decode`),
+// the second argument to be the HTML payload, and the
+// third argument to be the argument list.
+// any other message: respond with an {error, unknown_call} tuple.
+static void handle_send (state_t * state, erlang_msg * emsg)
{
- ETERM *decode_pattern = erl_format("{decode, Bin, Args}");
- ETERM *response;
+ // response holds our response, prepare it
+ ei_x_buff response;
+
+ ei_x_new (&response);
- if (erl_match(decode_pattern, emsg->msg))
+ // check the protocol version, if it's unsupported, panic
+ int version;
+ if (ei_decode_version (state->buffer.buff, &state->buffer.index, &version) < 0)
+ panic ("malformed message - bad version (%d).", version);
+
+ // decode the tuple header, make sure we have an arity of 3.
+ int arity;
+ if (ei_decode_tuple_header (state->buffer.buff, &state->buffer.index, &arity) < 0 || arity != 3)
{
- ETERM *bin = erl_var_content(decode_pattern, "Bin");
- ETERM *args = erl_var_content(decode_pattern, "Args");
+ err_term (&response, "badmatch");
+ goto out;
+ }
- response = decode(state, emsg, bin, args);
+ // the tuple should begin with a `decode` atom.
+ char atom[MAXATOMLEN];
+ if (ei_decode_atom (state->buffer.buff, &state->buffer.index, atom) < 0)
+ {
+ err_term (&response, "badmatch");
+ goto out;
+ }
- // free allocated resources
- erl_free_term(bin);
- erl_free_term(args);
+ if (strcmp (atom, "decode"))
+ {
+ err_term (&response, "unknown_call");
+ goto out;
}
- else
+
+ // the next argument should be a binary, allocate it dynamically.
+ int bin_type, bin_size;
+ if (ei_get_type (state->buffer.buff, &state->buffer.index, &bin_type, &bin_size) < 0)
+ panic ("failed to decode binary size in message");
+
+ // verify the type
+ if (bin_type != ERL_BINARY_EXT)
{
- response = err_term("unknown_call");
- return;
+ err_term (&response, "badmatch");
+ goto out;
}
- // send response
- erl_send(state->fd, emsg->from, response);
+ // decode the binary
+ char * bin_data = calloc (1, bin_size + 1);
+ if (ei_decode_binary (state->buffer.buff, &state->buffer.index, bin_data, NULL) < 0)
+ panic ("failed to decode binary in message");
- // free allocated resources
- erl_free_compound(response);
- erl_free_term(decode_pattern);
+ // next should be the options list
+ if (ei_decode_list_header (state->buffer.buff, &state->buffer.index, &arity) < 0)
+ panic ("failed to decode options list header in message");
- // free the free-list
- erl_eterm_release();
+ parse_flags_t parse_flags = decode_parse_flags (state, arity);
+ decode (state, &response, bin_data, bin_size, parse_flags);
+
+ free (bin_data);
+
+out:
+ // send response
+ ei_send (state->fd, &emsg->from, response.buff, response.buffsz);
+
+ // free response
+ ei_x_free (&response);
return;
}
-ETERM*
-err_term(const char* error_atom)
+static void err_term (ei_x_buff * response, const char * error_atom)
{
- /* ETERM* tuple2[] = {erl_mk_atom("error"), erl_mk_atom(error_atom)}; */
- /* return erl_mk_tuple(tuple2, 2); */
- return erl_format("{error, ~w}", erl_mk_atom(error_atom));
+ response->index = 0;
+ ei_x_encode_version (response);
+ ei_x_encode_tuple_header (response, 2);
+ ei_x_encode_atom (response, "error");
+ ei_x_encode_atom (response, error_atom);
}
-ETERM*
-decode(state_t* state, ErlMessage* emsg, ETERM* bin, ETERM* args)
+static parse_flags_t decode_parse_flags (state_t * state, int arity)
{
- unsigned char parse_flags = 0;
+ parse_flags_t parse_flags = 0;
- if (!ERL_IS_BINARY(bin) || !ERL_IS_LIST(args))
+ for (int i = 0; i < arity; i++)
{
- return err_term("badarg");
+ char atom[MAXATOMLEN];
+
+ if (ei_decode_atom (state->buffer.buff, &state->buffer.index, atom) < 0)
+ continue;
+
+ if (! strcmp ("html_atoms", atom))
+ parse_flags |= FLAG_HTML_ATOMS;
+ else if (! strcmp ("nil_self_closing", atom))
+ parse_flags |= FLAG_NIL_SELF_CLOSING;
+ else if (! strcmp ("comment_tuple3", atom))
+ parse_flags |= FLAG_COMMENT_TUPLE3;
}
- // get contents of binary argument
- char* binary = (char*)ERL_BIN_PTR(bin);
- size_t binary_len = ERL_BIN_SIZE(bin);
+ return parse_flags;
+}
- myhtml_tree_t* tree = myhtml_tree_create();
- myhtml_tree_init(tree, state->myhtml);
+static void decode (state_t * state, ei_x_buff * response, const char * bin_data, size_t bin_size, parse_flags_t parse_flags)
+{
+ myhtml_tree_t * tree = myhtml_tree_create ();
+ myhtml_tree_init (tree, state->myhtml);
+ myhtml_tree_parse_flags_set (tree, MyHTML_TREE_PARSE_FLAGS_WITHOUT_DOCTYPE_IN_TREE);
// parse tree
- mystatus_t status = myhtml_parse(tree, MyENCODING_UTF_8, binary, binary_len);
+ mystatus_t status = myhtml_parse (tree, MyENCODING_UTF_8, bin_data, bin_size);
if (status != MyHTML_STATUS_OK)
{
- return err_term("myhtml_parse_failed");
+ err_term (response, "myhtml_parse_failed");
+ return;
}
- // read parse flags
- parse_flags = read_parse_flags(args);
-
// build tree
- myhtml_tree_node_t *root = myhtml_tree_get_document(tree);
- ETERM* result = build_tree(tree, myhtml_node_last_child(root), &parse_flags);
- myhtml_tree_destroy(tree);
+ myhtml_tree_node_t * root = myhtml_tree_get_document (tree);
+ build_tree (response, tree, root->child, parse_flags);
+ myhtml_tree_destroy (tree);
+}
+
+// a tag is sent as a tuple:
+// - a string or atom for the tag name
+// - an attribute list
+// - a children list
+// in this function, we prepare the atom and complete attribute list
+static void prepare_tag_header (ei_x_buff * response, const char * tag_string, myhtml_tree_node_t * node, parse_flags_t parse_flags)
+{
+ myhtml_tag_id_t tag_id = myhtml_node_tag_id (node);
+ myhtml_namespace_t tag_ns = myhtml_node_namespace (node);
+
+ ei_x_encode_tuple_header (response, 3);
- return result;
+ if (! (parse_flags & FLAG_HTML_ATOMS) || (tag_id == MyHTML_TAG__UNDEF || tag_id == MyHTML_TAG_LAST_ENTRY || tag_ns != MyHTML_NAMESPACE_HTML))
+ ei_x_encode_binary (response, tag_string, strlen (tag_string));
+ else
+ ei_x_encode_atom (response, tag_string);
+
+ prepare_node_attrs (response, node);
}
-unsigned char
-read_parse_flags(ETERM* list)
+// prepare an attribute node
+static void prepare_node_attrs(ei_x_buff * response, myhtml_tree_node_t * node)
{
- unsigned char parse_flags = 0;
- ETERM *flag;
- ETERM *html_atoms = erl_mk_atom("html_atoms");
- ETERM *nil_self_closing = erl_mk_atom("nil_self_closing");
- ETERM *comment_tuple3 = erl_mk_atom("comment_tuple3");
-
- for (; !ERL_IS_EMPTY_LIST(list); list = ERL_CONS_TAIL(list)) {
- flag = ERL_CONS_HEAD(list);
- if (erl_match(html_atoms, flag))
- {
- parse_flags |= FLAG_HTML_ATOMS;
- }
- else if (erl_match(nil_self_closing, flag))
- {
- parse_flags |= FLAG_NIL_SELF_CLOSING;
- }
- else if (erl_match(comment_tuple3, flag))
- {
- parse_flags |= FLAG_COMMENT_TUPLE3;
- }
- }
+ myhtml_tree_attr_t * attr;
- erl_free_term(html_atoms);
- erl_free_term(nil_self_closing);
- erl_free_term(comment_tuple3);
+ for (attr = myhtml_node_attribute_first (node); attr != NULL; attr = myhtml_attribute_next (attr))
+ {
+ size_t attr_name_len;
+ const char *attr_name = myhtml_attribute_key (attr, &attr_name_len);
+ size_t attr_value_len;
+ const char *attr_value = myhtml_attribute_value (attr, &attr_value_len);
- return parse_flags;
+ /* guard against poisoned attribute nodes */
+ if (! attr_name_len)
+ continue;
+
+ ei_x_encode_list_header (response, 1);
+ ei_x_encode_tuple_header (response, 2);
+ ei_x_encode_binary (response, attr_name, attr_name_len);
+
+ if (attr_value_len)
+ ei_x_encode_binary (response, attr_value, attr_value_len);
+ else
+ ei_x_encode_binary (response, attr_name, attr_name_len);
+ }
+
+ ei_x_encode_empty_list (response);
}
-ETERM* build_tree(myhtml_tree_t* tree, myhtml_tree_node_t* node, unsigned char* parse_flags)
+// dump a comment node
+static void prepare_comment (ei_x_buff * response, const char * node_comment, size_t comment_len, parse_flags_t parse_flags)
{
- ETERM* result;
- ETERM* atom_nil = erl_mk_atom("nil");
- ETERM* empty_list = erl_mk_empty_list();
- myhtml_tree_node_t* prev_node = NULL;
+ ei_x_encode_tuple_header (response, parse_flags & FLAG_COMMENT_TUPLE3 ? 3 : 2);
+ ei_x_encode_atom (response, "comment");
- tstack stack;
- tstack_init(&stack, 30);
- for(myhtml_tree_node_t* current_node = node;;) {
- ETERM* children;
+ if (parse_flags & FLAG_COMMENT_TUPLE3)
+ ei_x_encode_list_header (response, 0);
- // If we are going up the tree, get the children from the stack
- if (prev_node && !(current_node->next == prev_node || current_node->parent == prev_node)) {
- children = tstack_pop(&stack);
- // Else, try to go down the tree
- } else if(current_node->last_child) {
- tstack_push(&stack, erl_mk_empty_list());
+ ei_x_encode_binary (response, node_comment, comment_len);
+}
- prev_node = current_node;
- current_node = current_node->last_child;
+#ifdef DEBUG_LIST_MANIP
- continue;
- } else {
- if ((myhtml_node_is_close_self(current_node) || myhtml_node_is_void_element(current_node))
- && (*parse_flags & FLAG_NIL_SELF_CLOSING)) {
- children = atom_nil;
- } else {
- children = empty_list;
- }
- }
+#define EMIT_LIST_HDR \
+ printf ("list hdr for node %p\n", current_node); \
+ fflush (stdout); \
+ ei_x_encode_list_header (response, 1)
+
+#define EMIT_EMPTY_LIST_HDR \
+ printf ("list empty for node %p\n", current_node); \
+ fflush (stdout); \
+ ei_x_encode_list_header (response, 0)
+
+#define EMIT_LIST_TAIL \
+ printf ("list tail for node %p\n", current_node); \
+ fflush (stdout); \
+ ei_x_encode_empty_list (response)
+
+#else
- myhtml_tag_id_t tag_id = myhtml_node_tag_id(current_node);
- myhtml_namespace_t tag_ns = myhtml_node_namespace(current_node);
+#define EMIT_LIST_HDR ei_x_encode_list_header (response, 1)
+#define EMIT_EMPTY_LIST_HDR ei_x_encode_list_header (response, 0)
+#define EMIT_LIST_TAIL ei_x_encode_empty_list (response)
+
+#endif
+
+static void build_tree (ei_x_buff * response, myhtml_tree_t * tree, myhtml_tree_node_t * node, parse_flags_t parse_flags)
+{
+ myhtml_tree_node_t * current_node = node;
+
+ tstack stack;
+ tstack_init (&stack, 30);
+
+ // ok we're going to send an actual response so start encoding it
+ response->index = 0;
+ ei_x_encode_version (response);
+
+ while (current_node != NULL)
+ {
+ myhtml_tag_id_t tag_id = myhtml_node_tag_id (current_node);
+ myhtml_namespace_t tag_ns = myhtml_node_namespace (current_node);
if (tag_id == MyHTML_TAG__TEXT)
{
size_t text_len;
+ const char * node_text = myhtml_node_text (current_node, &text_len);
- const char* node_text = myhtml_node_text(current_node, &text_len);
- result = erl_mk_binary(node_text, text_len);
+ EMIT_LIST_HDR;
+ ei_x_encode_binary (response, node_text, text_len);
}
else if (tag_id == MyHTML_TAG__COMMENT)
{
size_t comment_len;
- const char* node_comment = myhtml_node_text(current_node, &comment_len);
-
- // For <!----> myhtml_node_text will return a null pointer, which will make erl_format segfault
- ETERM* comment = erl_mk_binary(node_comment ? node_comment : "", comment_len);
+ const char* node_comment = myhtml_node_text (current_node, &comment_len);
- if (*parse_flags & FLAG_COMMENT_TUPLE3)
- {
- result = erl_format("{comment, [], ~w}", comment);
- }
- else
- {
- result = erl_format("{comment, ~w}", comment);
- }
+ EMIT_LIST_HDR;
+ prepare_comment (response, node_comment, comment_len, parse_flags);
}
else
{
- ETERM* tag;
- ETERM* attrs;
-
// get name of tag
size_t tag_name_len;
- const char *tag_name = myhtml_tag_name_by_id(tree, tag_id, &tag_name_len);
+ const char *tag_name = myhtml_tag_name_by_id (tree, tag_id, &tag_name_len);
// get namespace of tag
size_t tag_ns_len;
- const char *tag_ns_name_ptr = myhtml_namespace_name_by_id(tag_ns, &tag_ns_len);
+ const char *tag_ns_name_ptr = myhtml_namespace_name_by_id (tag_ns, &tag_ns_len);
char buffer [tag_ns_len + tag_name_len + 2];
char *tag_string = buffer;
- size_t tag_string_len;
if (tag_ns != MyHTML_NAMESPACE_HTML)
{
// tag_ns_name_ptr is unmodifyable, copy it in our tag_ns_buffer to make it modifyable.
// +1 because myhtml uses strlen for length returned, which doesn't include the null-byte
// https://github.com/lexborisov/myhtml/blob/0ade0e564a87f46fd21693a7d8c8d1fa09ffb6b6/source/myhtml/mynamespace.c#L80
char tag_ns_buffer[tag_ns_len + 1];
- strncpy(tag_ns_buffer, tag_ns_name_ptr, sizeof(tag_ns_buffer));
- lowercase(tag_ns_buffer);
+ strncpy (tag_ns_buffer, tag_ns_name_ptr, sizeof tag_ns_buffer);
+ lowercase (tag_ns_buffer);
- tag_string_len = tag_ns_len + tag_name_len + 1; // +1 for colon
- snprintf(tag_string, sizeof(buffer), "%s:%s", tag_ns_buffer, tag_name);
+ snprintf (tag_string, sizeof buffer, "%s:%s", tag_ns_buffer, tag_name);
}
else
{
- strncpy(tag_string, tag_name, sizeof(buffer));
- tag_string_len = tag_name_len;
+ // strncpy length does not contain null, so blank the buffer before copying
+ // and limit the copy length to buffer size minus one for safety.
+ memset (tag_string, '\0', sizeof buffer);
+ strncpy (tag_string, tag_name, sizeof buffer - 1);
}
- // attributes
- attrs = build_node_attrs(tree, current_node);
+ if (response->index > 1)
+ {
+ EMIT_LIST_HDR;
+ }
- if (!(*parse_flags & FLAG_HTML_ATOMS) || (tag_id == MyHTML_TAG__UNDEF || tag_id == MyHTML_TAG_LAST_ENTRY || tag_ns != MyHTML_NAMESPACE_HTML))
- tag = erl_mk_binary(tag_string, tag_string_len);
- else
- tag = erl_mk_atom(tag_string);
+ prepare_tag_header (response, tag_string, current_node, parse_flags);
- result = erl_format("{~w, ~w, ~w}", tag, attrs, children);
- }
+ if (current_node->child)
+ {
+ tstack_push (&stack, current_node);
+ current_node = current_node->child;
- if (stack.used == 0) {
- tstack_free(&stack);
- break;
- } else {
- tstack_push(&stack, erl_cons(result, tstack_pop(&stack)));
- prev_node = current_node;
- current_node = current_node->prev ? current_node->prev : current_node->parent;
+ continue;
+ }
+ else
+ {
+ if (parse_flags & FLAG_NIL_SELF_CLOSING && (myhtml_node_is_close_self(current_node) || myhtml_node_is_void_element(current_node)))
+ {
+#ifdef DEBUG_LIST_MANIP
+ printf ("self-closing tag %s emit nil?\n", tag_string); fflush (stdout);
+#endif
+ ei_x_encode_atom (response, "nil");
+ }
+ else
+ {
+ EMIT_EMPTY_LIST_HDR;
+ }
+ }
}
- }
-
- erl_free_term(atom_nil);
-
- return result;
-}
-
-ETERM*
-build_node_attrs(myhtml_tree_t* tree, myhtml_tree_node_t* node)
-{
- myhtml_tree_attr_t* attr;
-
- ETERM* list = erl_mk_empty_list();
-
- for (attr = myhtml_node_attribute_last(node); attr != NULL; attr = myhtml_attribute_prev(attr))
- {
- ETERM* name;
- ETERM* value;
- ETERM* attr_tuple;
- size_t attr_name_len;
- const char *attr_name = myhtml_attribute_key(attr, &attr_name_len);
- size_t attr_value_len;
- const char *attr_value = myhtml_attribute_value(attr, &attr_value_len);
-
- /* guard against poisoned attribute nodes */
- if (! attr_name_len)
- continue;
-
- name = erl_mk_binary(attr_name, attr_name_len);
- value = attr_value_len ? erl_mk_binary(attr_value, attr_value_len) : name;
+ if (current_node->next)
+ current_node = current_node->next;
+ else
+ {
+ while (! current_node->next && stack.used != 0)
+ {
+ EMIT_LIST_TAIL;
+ current_node = tstack_pop (&stack);
+ }
- /* ETERM* tuple2[] = {name, value}; */
- /* attr_tuple = erl_mk_tuple(tuple2, 2); */
- attr_tuple = erl_format("{~w, ~w}", name, value);
+ if (current_node->next)
+ current_node = current_node->next;
+ }
- list = erl_cons(attr_tuple, list);
+ // are we at root?
+ if (current_node == node)
+ break;
}
- return list;
+ tstack_free (&stack);
}
-static inline char*
-lowercase(char* c)
+static inline char * lowercase(char* c)
{
- char* p = c;
- while(*p)
+ char * p = c;
+
+ while (*p)
{
- *p = tolower((unsigned char)*p);
+ *p = tolower ((unsigned char) *p);
p++;
}
+
return c;
}
-
diff --git a/c_src/tstack.h b/c_src/tstack.h
index 48141bc..bd8d51d 100644
--- a/c_src/tstack.h
+++ b/c_src/tstack.h
@@ -1,39 +1,38 @@
#ifndef TSTACK_H
#define TSTACK_H
-#include "ei.h"
#define GROW_BY 30
typedef struct {
- ETERM* *data;
+ myhtml_tree_node_t **data;
size_t used;
size_t size;
} tstack;
void tstack_init(tstack *stack, size_t initial_size) {
- stack->data = (ETERM **) malloc(initial_size * sizeof(ETERM*));
+ stack->data = (myhtml_tree_node_t **) malloc(initial_size * sizeof(myhtml_tree_node_t *));
stack->used = 0;
stack->size = initial_size;
}
void tstack_free(tstack *stack) {
free(stack->data);
}
void tstack_resize(tstack *stack, size_t new_size) {
- stack->data = (ETERM **)realloc(stack->data, new_size * sizeof(ETERM*));
+ stack->data = (myhtml_tree_node_t **) realloc(stack->data, new_size * sizeof(myhtml_tree_node_t *));
stack->size = new_size;
}
-void tstack_push(tstack *stack, ETERM* element) {
+void tstack_push(tstack *stack, myhtml_tree_node_t * element) {
if(stack->used == stack->size) {
tstack_resize(stack, stack->size + GROW_BY);
}
- stack->data[stack->used++] = element;
+ stack->data[stack->used++] = element;
}
-ETERM* tstack_pop(tstack *stack) {
+myhtml_tree_node_t* tstack_pop(tstack *stack) {
return stack->data[--(stack->used)];
}
#endif
diff --git a/mix.exs b/mix.exs
index 311b65c..c9dfa7c 100644
--- a/mix.exs
+++ b/mix.exs
@@ -1,123 +1,136 @@
defmodule Myhtmlex.Mixfile do
use Mix.Project
def project do
[
app: :myhtmlex,
version: "0.2.1",
elixir: "~> 1.5",
deps: deps(),
package: package(),
compilers: [:myhtmlex_make] ++ Mix.compilers(),
build_embedded: Mix.env() == :prod,
start_permanent: Mix.env() == :prod,
name: "Myhtmlex",
description: """
A module to decode HTML into a tree,
porting all properties of the underlying
library myhtml, being fast and correct
in regards to the html spec.
""",
docs: docs()
]
end
def package do
[
maintainers: ["Lukas Rieder"],
licenses: ["GNU LGPL"],
links: %{
"Github" => "https://git.pleroma.social/pleroma/myhtmlex",
"Issues" => "https://git.pleroma.social/pleroma/myhtmlex/issues",
"MyHTML" => "https://github.com/lexborisov/myhtml"
},
files: [
"lib",
"c_src",
"priv/.gitignore",
"test",
"Makefile",
"mix.exs",
"README.md",
"LICENSE"
]
]
end
def application do
[
extra_applications: [:logger],
mod: {Myhtmlex.Safe, []},
# used to detect conflicts with other applications named processes
registered: [Myhtmlex.Safe.Cnode, Myhtmlex.Safe.Supervisor],
env: [
mode: Myhtmlex.Safe
]
]
end
defp deps do
[
# documentation helpers
{:ex_doc, ">= 0.0.0", only: :dev},
# benchmarking helpers
{:benchfella, "~> 0.3.0", only: :dev},
# cnode helpers
{:nodex,
git: "https://git.pleroma.social/pleroma/nodex",
ref: "cb6730f943cfc6aad674c92161be23a8411f15d1"}
]
end
defp docs do
[
main: "Myhtmlex"
]
end
end
defmodule Mix.Tasks.Compile.MyhtmlexMake do
@artifacts [
"priv/myhtml_worker"
]
def find_make do
_make_cmd =
System.get_env("MAKE") ||
case :os.type() do
{:unix, :freebsd} -> "gmake"
{:unix, :openbsd} -> "gmake"
{:unix, :netbsd} -> "gmake"
{:unix, :dragonfly} -> "gmake"
_ -> "make"
end
end
+ defp otp_version do
+ :erlang.system_info(:otp_release)
+ |> to_string()
+ |> String.to_integer()
+ end
+
+ defp otp_22_or_newer? do
+ otp_version() >= 22
+ end
+
def run(_) do
make_cmd = find_make()
if match?({:win32, _}, :os.type()) do
IO.warn("Windows is not yet a target.")
exit(1)
else
{result, _error_code} =
System.cmd(
make_cmd,
@artifacts,
stderr_to_stdout: true,
- env: [{"MIX_ENV", to_string(Mix.env())}]
+ env: [
+ {"MIX_ENV", to_string(Mix.env())},
+ {"OTP22_DEF", (otp_22_or_newer?() && "YES") || "NO"}
+ ]
)
IO.binwrite(result)
end
:ok
end
def clean() do
make_cmd = find_make()
{result, _error_code} = System.cmd(make_cmd, ["clean"], stderr_to_stdout: true)
Mix.shell().info(result)
:ok
end
end
diff --git a/test/myhtmlex.safe_test.exs b/test/myhtmlex.safe_test.exs
index 0054815..122cd9a 100644
--- a/test/myhtmlex.safe_test.exs
+++ b/test/myhtmlex.safe_test.exs
@@ -1,7 +1,140 @@
defmodule Myhtmlex.SafeTest do
- use MyhtmlexSharedTests, module: Myhtmlex.Safe
+ use ExUnit.Case
+ doctest Myhtmlex
test "doesn't segfault when <!----> is encountered" do
assert {"html", _attrs, _children} = Myhtmlex.decode("<div> <!----> </div>")
end
+
+ test "builds a tree, formatted like mochiweb by default" do
+ assert {"html", [],
+ [
+ {"head", [], []},
+ {"body", [],
+ [
+ {"br", [], []}
+ ]}
+ ]} = Myhtmlex.decode("<br>")
+ end
+
+ test "builds a tree, html tags as atoms" do
+ assert {:html, [],
+ [
+ {:head, [], []},
+ {:body, [],
+ [
+ {:br, [], []}
+ ]}
+ ]} = Myhtmlex.decode("<br>", format: [:html_atoms])
+ end
+
+ test "builds a tree, nil self closing" do
+ assert {"html", [],
+ [
+ {"head", [], []},
+ {"body", [],
+ [
+ {"br", [], nil},
+ {"esi:include", [], nil}
+ ]}
+ ]} = Myhtmlex.decode("<br><esi:include />", format: [:nil_self_closing])
+ end
+
+ test "builds a tree, multiple format options" do
+ assert {:html, [],
+ [
+ {:head, [], []},
+ {:body, [],
+ [
+ {:br, [], nil}
+ ]}
+ ]} = Myhtmlex.decode("<br>", format: [:html_atoms, :nil_self_closing])
+ end
+
+ test "attributes" do
+ assert {:html, [],
+ [
+ {:head, [], []},
+ {:body, [],
+ [
+ {:span, [{"id", "test"}, {"class", "foo garble"}], []}
+ ]}
+ ]} =
+ Myhtmlex.decode(~s'<span id="test" class="foo garble"></span>',
+ format: [:html_atoms]
+ )
+ end
+
+ test "single attributes" do
+ assert {:html, [],
+ [
+ {:head, [], []},
+ {:body, [],
+ [
+ {:button, [{"disabled", "disabled"}, {"class", "foo garble"}], []}
+ ]}
+ ]} =
+ Myhtmlex.decode(~s'<button disabled class="foo garble"></span>',
+ format: [:html_atoms]
+ )
+ end
+
+ test "text nodes" do
+ assert {:html, [],
+ [
+ {:head, [], []},
+ {:body, [],
+ [
+ "text node"
+ ]}
+ ]} = Myhtmlex.decode(~s'<body>text node</body>', format: [:html_atoms])
+ end
+
+ test "broken input" do
+ assert {:html, [],
+ [
+ {:head, [], []},
+ {:body, [],
+ [
+ {:a, [{"<", "<"}], [" asdf"]}
+ ]}
+ ]} = Myhtmlex.decode(~s'<a <> asdf', format: [:html_atoms])
+ end
+
+ test "namespaced tags" do
+ assert {:html, [],
+ [
+ {:head, [], []},
+ {:body, [],
+ [
+ {"svg:svg", [],
+ [
+ {"svg:path", [], []},
+ {"svg:a", [], []}
+ ]}
+ ]}
+ ]} = Myhtmlex.decode(~s'<svg><path></path><a></a></svg>', format: [:html_atoms])
+ end
+
+ test "custom namespaced tags" do
+ assert {:html, [],
+ [
+ {:head, [], []},
+ {:body, [],
+ [
+ {"esi:include", [], nil}
+ ]}
+ ]} = Myhtmlex.decode(~s'<esi:include />', format: [:html_atoms, :nil_self_closing])
+ end
+
+ test "html comments" do
+ assert {:html, [],
+ [
+ {:head, [], []},
+ {:body, [],
+ [
+ comment: " a comment "
+ ]}
+ ]} = Myhtmlex.decode(~s'<body><!-- a comment --></body>', format: [:html_atoms])
+ end
end
diff --git a/test/myhtmlex_shared_tests.ex b/test/myhtmlex_shared_tests.ex
deleted file mode 100644
index 7ba8d79..0000000
--- a/test/myhtmlex_shared_tests.ex
+++ /dev/null
@@ -1,152 +0,0 @@
-defmodule MyhtmlexSharedTests do
- defmacro __using__(opts) do
- module = Keyword.fetch!(opts, :module)
-
- quote do
- use ExUnit.Case
- doctest Myhtmlex
-
- setup_all(_) do
- Application.put_env(:myhtmlex, :mode, unquote(module))
- :ok
- end
-
- test "builds a tree, formatted like mochiweb by default" do
- assert {"html", [],
- [
- {"head", [], []},
- {"body", [],
- [
- {"br", [], []}
- ]}
- ]} = Myhtmlex.decode("<br>")
- end
-
- test "builds a tree, html tags as atoms" do
- assert {:html, [],
- [
- {:head, [], []},
- {:body, [],
- [
- {:br, [], []}
- ]}
- ]} = Myhtmlex.decode("<br>", format: [:html_atoms])
- end
-
- test "builds a tree, nil self closing" do
- assert {"html", [],
- [
- {"head", [], []},
- {"body", [],
- [
- {"br", [], nil},
- {"esi:include", [], nil}
- ]}
- ]} = Myhtmlex.decode("<br><esi:include />", format: [:nil_self_closing])
- end
-
- test "builds a tree, multiple format options" do
- assert {:html, [],
- [
- {:head, [], []},
- {:body, [],
- [
- {:br, [], nil}
- ]}
- ]} = Myhtmlex.decode("<br>", format: [:html_atoms, :nil_self_closing])
- end
-
- test "attributes" do
- assert {:html, [],
- [
- {:head, [], []},
- {:body, [],
- [
- {:span, [{"id", "test"}, {"class", "foo garble"}], []}
- ]}
- ]} =
- Myhtmlex.decode(~s'<span id="test" class="foo garble"></span>',
- format: [:html_atoms]
- )
- end
-
- test "single attributes" do
- assert {:html, [],
- [
- {:head, [], []},
- {:body, [],
- [
- {:button, [{"disabled", "disabled"}, {"class", "foo garble"}], []}
- ]}
- ]} =
- Myhtmlex.decode(~s'<button disabled class="foo garble"></span>',
- format: [:html_atoms]
- )
- end
-
- test "text nodes" do
- assert {:html, [],
- [
- {:head, [], []},
- {:body, [],
- [
- "text node"
- ]}
- ]} = Myhtmlex.decode(~s'<body>text node</body>', format: [:html_atoms])
- end
-
- test "broken input" do
- assert {:html, [],
- [
- {:head, [], []},
- {:body, [],
- [
- {:a, [{"<", "<"}], [" asdf"]}
- ]}
- ]} = Myhtmlex.decode(~s'<a <> asdf', format: [:html_atoms])
- end
-
- test "namespaced tags" do
- assert {:html, [],
- [
- {:head, [], []},
- {:body, [],
- [
- {"svg:svg", [],
- [
- {"svg:path", [], []},
- {"svg:a", [], []}
- ]}
- ]}
- ]} = Myhtmlex.decode(~s'<svg><path></path><a></a></svg>', format: [:html_atoms])
- end
-
- test "custom namespaced tags" do
- assert {:html, [],
- [
- {:head, [], []},
- {:body, [],
- [
- {"esi:include", [], nil}
- ]}
- ]} =
- Myhtmlex.decode(~s'<esi:include />', format: [:html_atoms, :nil_self_closing])
- end
-
- test "html comments" do
- assert {:html, [],
- [
- {:head, [], []},
- {:body, [],
- [
- comment: " a comment "
- ]}
- ]} = Myhtmlex.decode(~s'<body><!-- a comment --></body>', format: [:html_atoms])
- end
- end
-
- # quote
- end
-
- # defmacro __using__
-end
diff --git a/test/test_helper.exs b/test/test_helper.exs
index 8efc5f3..869559e 100644
--- a/test/test_helper.exs
+++ b/test/test_helper.exs
@@ -1,3 +1 @@
-Code.require_file("myhtmlex_shared_tests.ex", "test")
-
ExUnit.start()
File Metadata
Details
Attached
Mime Type
text/x-diff
Expires
Fri, Oct 9, 11:42 AM (1 d, 22 h)
Storage Engine
blob
Storage Format
Raw Data
Storage Handle
1783171
Default Alt Text
(41 KB)
Attached To
Mode
R16 fast_html
Attached
Detach File
Event Timeline
Log In to Comment