Index: soft/giet_vm/applications/classif/Makefile
===================================================================
--- soft/giet_vm/applications/classif/Makefile	(revision 707)
+++ soft/giet_vm/applications/classif/Makefile	(revision 708)
@@ -2,5 +2,5 @@
 APP_NAME = classif
 
-OBJS= main.o 
+OBJS= classif.o 
 
 LIBS= -L../../build/libs -luser
Index: soft/giet_vm/applications/classif/classif.c
===================================================================
--- soft/giet_vm/applications/classif/classif.c	(revision 708)
+++ soft/giet_vm/applications/classif/classif.c	(revision 708)
@@ -0,0 +1,663 @@
+///////////////////////////////////////////////////////////////////////////////////////
+// File   : classif.c
+// Date   : november 2014
+// author : Alain Greiner
+///////////////////////////////////////////////////////////////////////////////////////
+// This multi-threaded application takes a stream of Gigabit Ethernet packets,
+// and makes packet analysis and classification, based on the source MAC address.
+// It uses the NIC peripheral, and the distributed kernel chbufs accessed by the CMA 
+// component to receive and send packets on the Gigabit Ethernet port. 
+//
+// It can run on architectures containing up to 16 * 16 clusters,
+// and from 3 to 8 processors per cluster.
+//
+// It requires one shared TTY terminal.
+//
+// This application is described as a TCG (Thread and Communication Graph) 
+// containing (N+2) threads per cluster.
+// - one "load" thread
+// - one "store" thread
+// - N "analyse" threads
+// The containers are distributed (N+2 containers per cluster):
+// - one RX container (part of the kernel rx_chbuf), in the kernel heap.
+// - one TX container (part of the kernel tx-chbuf), in the kernel heap.
+// - N working containers (one per analysis thread), in the user heap.
+// In each cluster, the "load", analysis" and "store" threads communicates through
+// three local MWMR fifos: 
+// - fifo_l2a : tranfer a full container from "load" to "analyse" thread.
+// - fifo_a2s : transfer a full container from "analyse" to "store" thread.
+// - fifo_s2l : transfer an empty container from "store" to "load" thread.
+// For each fifo, one item is a 32 bits word defining the index of an
+// available working container.
+// The pointers on the working containers, and the pointers on the MWMR fifos
+// are global arrays stored in cluster[0][0].
+//
+// The main thread exit after global initialisation, and launching the other threads: 
+// It does not use the pthread_join() construct. It is executed on P[0,0,1],
+// toavoid overload on P[0,0,0].
+//
+// Initialisation is made in two steps:
+//
+// 1) The global, shared, variables are initialised by the main thread:
+//    - shared TTY
+//    - distributed heap (one heap per cluster)
+//    - distributed rx_barrier (between all "load" threads)
+//    - distributed tx_barrier (between all "store" threads)
+//    - RX kernel chbufs (used by "load" threads)
+//    - TX kernel chbufs (used by "store" threads)
+//    Then the main thread exit, after launching the "load, "store", and "analyse"
+//    threads in all clusters.
+//
+// 2) Each "load" thread allocates containers[x][y][n] from local heap,
+//    and register containers pointers in the local stack.
+//    Each "load" thread allocates data buffers & mwmr fifo descriptors
+//    from local heap, and register pointers in global arrays.
+//    Each "load" thread initialises the containers as empty in fifo_s2l.
+//    Then each "load" thread signals mwmr fifos initialisation completion
+//    to other threads in same cluster, using the local_sync[x][y] variables.
+//
+// When initialisation is completed, all threads are running in parallel:
+//
+// 1) The "load" thread get an empty working container from the fifo_s2l,
+//    transfer one container from the kernel rx_chbuf to this user container,
+//    and transfer ownership of this container to one "analysis" thread by writing
+//    into the fifo_l2a.    
+// 2) The "analyse" thread get one working container from the fifo_l2a, analyse
+//    each packet header, compute the packet type (depending on the SRC MAC address),
+//    increment the correspondint classification counter, and transpose the SRC
+//    and the DST MAC addresses fot TX tranmission.
+// 3) The "store" thread transfer get a full working container from the fifo_a2s,
+//    transfer this user container content to the the kernel tx_chbuf,
+//    and transfer ownership of this empty container to the "load" thread by writing
+//    into the fifo_s2l.   
+//     
+// Instrumentation results display is done by the "store" thread in cluster[0][0]
+// when all "store" threads completed the number of clusters specified by the
+// CONTAINERS_MAX parameter.
+///////////////////////////////////////////////////////////////////////////////////////
+
+#include "stdio.h"
+#include "user_barrier.h"
+#include "malloc.h"
+#include "user_lock.h"
+#include "mwmr_channel.h"
+
+#define X_SIZE_MAX        16
+#define Y_SIZE_MAX        16
+#define NPROCS_MAX        8
+#define CONTAINERS_MAX    5000
+#define VERBOSE_ANALYSE   0
+
+// macro to use a shared TTY
+#define printf(...);    { lock_acquire( &tty_lock ); \
+                          giet_tty_printf(__VA_ARGS__);  \
+                          lock_release( &tty_lock ); }
+
+
+///////////////////////////////////////////////////////////////////////////////////////
+//    Global variables
+// The MWMR channels (descriptors and buffers), as well as the working containers 
+// used by the "analysis" threads are distributed in clusters.
+// But the pointers on these distributed structures are stored in cluster[0][0].
+///////////////////////////////////////////////////////////////////////////////////////
+
+// pointers on distributed containers
+unsigned int*         container[X_SIZE_MAX][Y_SIZE_MAX][NPROCS_MAX-2];  
+
+// pointers on distributed mwmr fifos containing container descriptors
+mwmr_channel_t*       mwmr_l2a[X_SIZE_MAX][Y_SIZE_MAX];  
+mwmr_channel_t*       mwmr_a2s[X_SIZE_MAX][Y_SIZE_MAX];
+mwmr_channel_t*       mwmr_s2l[X_SIZE_MAX][Y_SIZE_MAX]; 
+
+// local synchros signaling local MWMR fifos initialisation completion
+volatile unsigned int local_sync[X_SIZE_MAX][Y_SIZE_MAX];  
+
+// lock protecting shared TTY
+user_lock_t           tty_lock;
+
+// distributed barrier between "load" threads
+giet_sqt_barrier_t    rx_barrier;
+
+// distributed barrier between "store" threads
+giet_sqt_barrier_t    tx_barrier;
+
+// instrumentation counters
+unsigned int          counter[16];
+
+// threads arguments array
+unsigned int thread_arg[16][16][4];
+
+////////////////////////////////////////////////////////////
+__attribute__ ((constructor)) void load( unsigned int* arg )
+////////////////////////////////////////////////////////////
+{
+    // get plat-form parameters
+    unsigned int x_size;                       // number of clusters in a row
+    unsigned int y_size;                       // number of clusters in a column
+    unsigned int nprocs;                       // number of processors per cluster
+    giet_procs_number( &x_size , &y_size , &nprocs );
+
+    // each "load" thread get processor identifiers
+    unsigned int    x;
+    unsigned int    y;
+    unsigned int    p;
+    giet_proc_xyp( &x, &y, &p );
+
+    // each "load" thread allocates containers[x][y][n] (from local heap)
+    // and register pointers in the local stack
+    unsigned int   n;
+    unsigned int*  cont[NPROCS_MAX-2]; 
+
+    for ( n = 0 ; n < (nprocs - 2) ; n++ )
+    {
+        container[x][y][n] = malloc( 4096 );
+        cont[n]            = container[x][y][n];
+    }
+    
+    // each "load" thread allocates data buffers for mwmr fifos (from local heap)
+    unsigned int*  data_l2a = malloc( (nprocs - 2)<<2 );
+    unsigned int*  data_a2s = malloc( (nprocs - 2)<<2 );
+    unsigned int*  data_s2l = malloc( (nprocs - 2)<<2 );
+
+    // each "load" thread allocates mwmr fifos descriptors (from local heap)
+    mwmr_l2a[x][y] = malloc( sizeof(mwmr_channel_t) );
+    mwmr_a2s[x][y] = malloc( sizeof(mwmr_channel_t) );
+    mwmr_s2l[x][y] = malloc( sizeof(mwmr_channel_t) );
+
+    // each "load" thread registers local pointers on mwmr fifos in local stack
+    mwmr_channel_t* fifo_l2a = mwmr_l2a[x][y];
+    mwmr_channel_t* fifo_a2s = mwmr_a2s[x][y];
+    mwmr_channel_t* fifo_s2l = mwmr_s2l[x][y];
+
+    // each "load" thread initialises local mwmr fifos descriptors
+    // ( width = 4 bytes / depth = number of analysis threads )
+    mwmr_init( fifo_l2a , data_l2a , 1 , (nprocs - 2) );
+    mwmr_init( fifo_a2s , data_a2s , 1 , (nprocs - 2) );
+    mwmr_init( fifo_s2l , data_s2l , 1 , (nprocs - 2) );
+
+    
+    // each "load" thread initialises local containers as empty in fifo_s2l
+    for ( n = 0 ; n < (nprocs - 2) ; n++ ) mwmr_write( fifo_s2l , &n , 1 );
+
+    // each "load" thread signals mwmr fifos initialisation completion
+    // to other threads in same cluster.
+    local_sync[x][y] = 1;
+
+    // only "load" thread[0][0] displays status
+    if ( (x==0) && (y==0) )
+    {
+        printf("\n[CLASSIF] load on P[%d,%d,%d] enters main loop at cycle %d\n"
+               "      &mwmr_l2a  = %x / &data_l2a  = %x\n"
+               "      &mwmr_a2s  = %x / &data_a2s  = %x\n"
+               "      &mwmr_s2l  = %x / &data_s2l  = %x\n"
+               "      &cont[0]   = %x\n"
+               "      x_size = %d / y_size = %d / nprocs = %d\n",
+               x , y , p , giet_proctime(), 
+               (unsigned int)fifo_l2a, (unsigned int)data_l2a,
+               (unsigned int)fifo_a2s, (unsigned int)data_a2s,
+               (unsigned int)fifo_s2l, (unsigned int)data_s2l,
+               (unsigned int)cont[0],
+               x_size, y_size, nprocs );
+    }
+
+    /////////////////////////////////////////////////////////////
+    // "load" thread enters the main loop (on containers)
+    unsigned int  count = 0;     // loaded containers count
+    unsigned int  index;         // available container index
+    unsigned int* temp;          // pointer on available container
+
+    while ( count < CONTAINERS_MAX ) 
+    { 
+        // get one empty container index from fifo_s2l
+        mwmr_read( fifo_s2l , &index , 1 );
+        temp = cont[index];
+
+        // get one container from  kernel rx_chbuf
+        giet_nic_rx_move( temp );
+
+        // get packets number
+        unsigned int npackets = temp[0] & 0x0000FFFF;
+        unsigned int nwords   = temp[0] >> 16;
+
+        if ( (x==0) && (y==0) )
+        {
+            printf("\n[CLASSIF] load on P[%d,%d,%d] get container %d at cycle %d"
+                   " : %d packets / %d words\n",
+                   x, y, p, index, giet_proctime(), npackets, nwords );
+        }
+
+        // put the full container index to fifo_l2a
+        mwmr_write( fifo_l2a, &index , 1 );
+
+        count++;
+    }
+
+    // all "load" threads synchronise before stats
+    sqt_barrier_wait( &rx_barrier );
+
+    // "load" thread[0][0] displays stats
+    if ( (x==0) && (y==0) ) giet_nic_rx_stats();
+
+    // all "load" thread exit
+    giet_pthread_exit("completed");
+ 
+} // end load()
+
+
+//////////////////////////////////////////////////////////////
+__attribute__ ((constructor)) void store( unsigned int * arg )
+//////////////////////////////////////////////////////////////
+{
+    // get plat-form parameters
+    unsigned int x_size;                       // number of clusters in a row
+    unsigned int y_size;                       // number of clusters in a column
+    unsigned int nprocs;                       // number of processors per cluster
+    giet_procs_number( &x_size , &y_size , &nprocs );
+
+    // get processor identifiers
+    unsigned int    x;
+    unsigned int    y;
+    unsigned int    p;
+    giet_proc_xyp( &x, &y, &p );
+
+    // each "store" thread wait mwmr channels initialisation
+    while ( local_sync[x][y] == 0 ) asm volatile ("nop");
+
+    // each "store" thread registers pointers on working containers in local stack
+    unsigned int   n;
+    unsigned int*  cont[NPROCS_MAX-2]; 
+
+    for ( n = 0 ; n < (nprocs - 2) ; n++ )
+    {
+        cont[n] = container[x][y][n];
+    }
+    
+    // each "store" thread registers pointers on mwmr fifos in local stack
+    mwmr_channel_t* fifo_l2a = mwmr_l2a[x][y];
+    mwmr_channel_t* fifo_a2s = mwmr_a2s[x][y];
+    mwmr_channel_t* fifo_s2l = mwmr_s2l[x][y];
+
+    // only "store" thread[0][0] displays status
+    if ( (x==0) && (y==0) )
+    {
+        printf("\n[CLASSIF] store on P[%d,%d,%d] enters main loop at cycle %d\n"
+               "      &mwmr_l2a  = %x\n"
+               "      &mwmr_a2s  = %x\n"
+               "      &mwmr_s2l  = %x\n"
+               "      &cont[0]   = %x\n",
+               x , y , p , giet_proctime(), 
+               (unsigned int)fifo_l2a,
+               (unsigned int)fifo_a2s,
+               (unsigned int)fifo_s2l,
+               (unsigned int)cont[0] );
+    }
+
+    /////////////////////////////////////////////////////////////
+    // "store" thread enter the main loop (on containers)
+    unsigned int count = 0;     // stored containers count
+    unsigned int index;         // empty container index
+    unsigned int* temp;         // pointer on empty container
+
+    while ( count < CONTAINERS_MAX ) 
+    { 
+        // get one working container index from fifo_a2s
+        mwmr_read( fifo_a2s , &index , 1 );
+        temp = cont[index];
+
+        // put one container to kernel tx_chbuf
+        giet_nic_tx_move( temp );
+ 
+        // get packets number
+        unsigned int npackets = temp[0] & 0x0000FFFF;
+        unsigned int nwords   = temp[0] >> 16;
+
+        if ( (x==0) && (y==0) )
+        {
+            printf("\n[CLASSIF] store on P[%d,%d,%d] get container %d at cycle %d"
+                   " : %d packets / %d words\n",
+                   x, y, p, index, giet_proctime(), npackets, nwords );
+        }
+
+        // put the working container index to fifo_s2l
+        mwmr_write( fifo_s2l, &index , 1 );
+
+        count++;
+    }
+
+    // all "store" threads synchronise before result display
+    sqt_barrier_wait( &tx_barrier );
+
+    // "store" thread[0,0] and displays results 
+    if ( (x==0) && (y==0) )
+    {
+        printf("\nClassification Results\n"
+               " - TYPE 0 : %d packets\n"
+               " - TYPE 1 : %d packets\n"
+               " - TYPE 2 : %d packets\n"
+               " - TYPE 3 : %d packets\n"
+               " - TYPE 4 : %d packets\n"
+               " - TYPE 5 : %d packets\n"
+               " - TYPE 6 : %d packets\n"
+               " - TYPE 7 : %d packets\n"
+               " - TYPE 8 : %d packets\n"
+               " - TYPE 9 : %d packets\n"
+               " - TYPE A : %d packets\n"
+               " - TYPE B : %d packets\n"
+               " - TYPE C : %d packets\n"
+               " - TYPE D : %d packets\n"
+               " - TYPE E : %d packets\n"
+               " - TYPE F : %d packets\n"
+               "    TOTAL = %d packets\n",
+               counter[0x0], counter[0x1], counter[0x2], counter[0x3],
+               counter[0x4], counter[0x5], counter[0x6], counter[0x7],
+               counter[0x8], counter[0x9], counter[0xA], counter[0xB],
+               counter[0xC], counter[0xD], counter[0xE], counter[0xF],
+               counter[0x0]+ counter[0x1]+ counter[0x2]+ counter[0x3]+
+               counter[0x4]+ counter[0x5]+ counter[0x6]+ counter[0x7]+
+               counter[0x8]+ counter[0x9]+ counter[0xA]+ counter[0xB]+
+               counter[0xC]+ counter[0xD]+ counter[0xE]+ counter[0xF] );
+
+        giet_nic_tx_stats();
+    }
+
+    // all "store" thread exit
+    giet_pthread_exit("Thread completed");
+
+} // end store()
+
+
+///////////////////////////////////////////////////////////////
+__attribute__ ((constructor)) void analyse( unsigned int* arg )
+///////////////////////////////////////////////////////////////
+{
+    // get platform parameters
+    unsigned int    x_size;						// number of clusters in row
+    unsigned int    y_size;                     // number of clusters in a column
+    unsigned int    nprocs;                     // number of processors per cluster
+    giet_procs_number( &x_size, &y_size, &nprocs );
+
+    // get processor identifiers
+    unsigned int    x;
+    unsigned int    y;
+    unsigned int    p;
+    giet_proc_xyp( &x, &y, &p );
+
+    // each "analyse" thread wait mwmr channels initialisation
+    while ( local_sync[x][y] == 0 ) asm volatile ("nop");
+
+    // each "analyse" threads register pointers on working containers in local stack
+    unsigned int   n;
+    unsigned int*  cont[NPROCS_MAX-2]; 
+    for ( n = 0 ; n < (nprocs - 2) ; n++ )
+    {
+        cont[n] = container[x][y][n];
+    }
+
+    // each "analyse" threads register pointers on mwmr fifos in local stack
+    mwmr_channel_t* fifo_l2a = mwmr_l2a[x][y];
+    mwmr_channel_t* fifo_a2s = mwmr_a2s[x][y];
+
+    // only "analyse" thread[0][0] display status
+    if ( (x==0) && (y==0) )
+    {
+        printf("\n[CLASSIF] analyse on P[%d,%d,%d] enters main loop at cycle %d\n"
+               "       &mwmr_l2a = %x\n"
+               "       &mwmr_a2s = %x\n"
+               "       &cont[0]  = %x\n",
+               x, y, p, giet_proctime(), 
+               (unsigned int)fifo_l2a,
+               (unsigned int)fifo_a2s,
+               (unsigned int)cont[0] );
+    }
+      
+    //////////////////////////////////////////////////////////////////////
+    // all "analyse" threads enter the main infinite loop (on containers)
+    unsigned int  index;           // available container index
+    unsigned int* temp;            // pointer on available container
+    unsigned int  nwords;          // number of words in container
+    unsigned int  npackets;        // number of packets in container
+    unsigned int  length;          // number of bytes in current packet
+    unsigned int  first;           // current packet first word in container
+    unsigned int  type;            // current packet type
+    unsigned int  pid;             // current packet index
+
+#if VERBOSE_ANALYSE
+    unsigned int       verbose_len[10]; // save length for 10 packets in one container
+    unsigned long long verbose_dst[10]; // save dest   for 10 packets in one container
+    unsigned long long verbose_src[10]; // save source for 10 packets in one container
+#endif
+
+    while ( 1 )
+    { 
+
+#if VERBOSE_ANALYSE 
+            for( pid = 0 ; pid < 10 ; pid++ )
+            {
+                verbose_len[pid] = 0;
+                verbose_dst[pid] = 0;
+                verbose_src[pid] = 0;
+            }
+#endif
+        // get one working container index from fifo_l2a
+        mwmr_read( fifo_l2a , &index , 1 );
+        temp = cont[index];
+
+        // get packets number and words number
+        npackets = temp[0] & 0x0000FFFF;
+        nwords   = temp[0] >> 16;
+
+        if ( (x==0) && (y==0) )
+        {
+            printf("\n[CLASSIF] analyse on P[%d,%d,%d] get container at cycle %d"
+                   " : %d packets / %d words\n",
+                   x, y, p, giet_proctime(), npackets, nwords );
+        }
+
+        // initialize word index in container
+        first = 34;
+
+        // loop on packets
+        for( pid = 0 ; pid < npackets ; pid++ )
+        {
+            // get packet length from container header
+            if ( (pid & 0x1) == 0 )  length = temp[1+(pid>>1)] >> 16;
+            else                     length = temp[1+(pid>>1)] & 0x0000FFFF;
+
+            // compute packet DST and SRC MAC addresses
+            unsigned int word0 = temp[first];
+            unsigned int word1 = temp[first + 1];
+            unsigned int word2 = temp[first + 2];
+
+#if VERBOSE_ANALYSE 
+            unsigned long long dst = ((unsigned long long)(word1 & 0xFFFF0000)>>16) |
+                                     (((unsigned long long)word0)<<16);
+            unsigned long long src = ((unsigned long long)(word1 & 0x0000FFFF)<<32) |
+                                     ((unsigned long long)word2);
+            if ( pid < 10 )
+            {
+                verbose_len[pid] = length;
+                verbose_dst[pid] = dst;
+                verbose_src[pid] = src;
+            }
+#endif
+            // compute type from SRC MAC address and increment counter
+            type = word1 & 0x0000000F;
+            atomic_increment( &counter[type], 1 );
+
+            // exchange SRC & DST MAC addresses for TX
+            temp[first]     = ((word1 & 0x0000FFFF)<<16) | ((word2 & 0xFFFF0000)>>16);
+            temp[first + 1] = ((word2 & 0x0000FFFF)<<16) | ((word0 & 0xFFFF0000)>>16);
+            temp[first + 2] = ((word0 & 0x0000FFFF)<<16) | ((word1 & 0xFFFF0000)>>16);
+
+            // update first word index 
+            if ( length & 0x3 ) first += (length>>2)+1;
+            else                first += (length>>2);
+        }
+        
+#if VERBOSE_ANALYSE 
+        if ( (x==0) && (y==0) )
+        {
+            printf("\n*** Thread analyse on P[%d,%d,%d] / container %d at cycle %d\n"
+                   "   - Packet 0 : plen = %d / dst_mac = %l / src_mac = %l\n"
+                   "   - Packet 1 : plen = %d / dst_mac = %l / src_mac = %l\n"
+                   "   - Packet 2 : plen = %d / dst_mac = %l / src_mac = %l\n"
+                   "   - Packet 3 : plen = %d / dst_mac = %l / src_mac = %l\n"
+                   "   - Packet 4 : plen = %d / dst_mac = %l / src_mac = %l\n"
+                   "   - Packet 5 : plen = %d / dst_mac = %l / src_mac = %l\n"
+                   "   - Packet 6 : plen = %d / dst_mac = %l / src_mac = %l\n"
+                   "   - Packet 7 : plen = %d / dst_mac = %l / src_mac = %l\n"
+                   "   - Packet 8 : plen = %d / dst_mac = %l / src_mac = %l\n"
+                   "   - Packet 9 : plen = %d / dst_mac = %l / src_mac = %l\n",
+                   x , y , p , index , giet_proctime() ,  
+                   verbose_len[0] , verbose_dst[0] , verbose_src[0] ,
+                   verbose_len[1] , verbose_dst[1] , verbose_src[1] ,
+                   verbose_len[2] , verbose_dst[2] , verbose_src[2] ,
+                   verbose_len[3] , verbose_dst[3] , verbose_src[3] ,
+                   verbose_len[4] , verbose_dst[4] , verbose_src[4] ,
+                   verbose_len[5] , verbose_dst[5] , verbose_src[5] ,
+                   verbose_len[6] , verbose_dst[6] , verbose_src[6] ,
+                   verbose_len[7] , verbose_dst[7] , verbose_src[7] ,
+                   verbose_len[8] , verbose_dst[8] , verbose_src[8] ,
+                   verbose_len[9] , verbose_dst[9] , verbose_src[9] );
+        }
+#endif
+            
+        // pseudo-random delay
+        unsigned int delay = giet_rand()>>3;
+        unsigned int time;
+        for( time = 0 ; time < delay ; time++ ) asm volatile ("nop");
+
+        // put the working container index to fifo_a2s
+        mwmr_write( fifo_a2s , &index , 1 );
+    }
+} // end analyse()
+
+//////////////////////////////////////////
+__attribute__ ((constructor)) void main()
+//////////////////////////////////////////
+{
+    // indexes for loops
+    unsigned int x , y , n;
+
+    // get identifiers for proc executing main
+    unsigned int x_id;                          // x cluster coordinate
+    unsigned int y_id;                          // y cluster coordinate
+    unsigned int p_id;                          // local processor index
+    giet_proc_xyp( &x_id , &y_id , &p_id );
+
+    // get & check plat-form parameters
+    unsigned int x_size;                       // number of clusters in a row
+    unsigned int y_size;                       // number of clusters in a column
+    unsigned int nprocs;                       // number of processors per cluster
+    giet_procs_number( &x_size , &y_size , &nprocs );
+
+    giet_pthread_assert( ((nprocs >= 3) && (nprocs <= 8)),
+                         "[CLASSIF ERROR] number of procs per cluster must in [3...8]");
+
+    giet_pthread_assert( ((x_size >= 1) && (x_size <= 16)),
+                         "[CLASSIF ERROR] x_size must be in [1...16]");
+
+    giet_pthread_assert( ((y_size >= 1) && (y_size <= 16)),
+                         "[CLASSIF ERROR] y_size must be in [1...16]");
+
+    // distributed heap initialisation
+    for ( x = 0 ; x < x_size ; x++ ) 
+    {
+        for ( y = 0 ; y < y_size ; y++ ) 
+        {
+            heap_init( x , y );
+        }
+    }
+
+    // shared TTY allocation
+    giet_tty_alloc( 1 );     
+    lock_init( &tty_lock);
+
+    printf("\n[CLASSIF] start at cycle %d on %d cores\n", 
+           giet_proctime(), (x_size * y_size * nprocs) );
+
+    // thread index 
+    // required bt pthread_create()
+    // unused in this appli because no pthread_join()
+    pthread_t   trdid; 
+
+    // rx_barrier initialisation
+    sqt_barrier_init( &rx_barrier, x_size , y_size , 1 );
+
+    // tx_barrier initialisation
+    sqt_barrier_init( &tx_barrier, x_size , y_size , 1 );
+
+    // allocate and start RX NIC and CMA channels
+    giet_nic_rx_alloc( x_size , y_size );
+    giet_nic_rx_start();
+
+    // allocate and start TX NIC and CMA channels
+    giet_nic_tx_alloc( x_size , y_size );
+    giet_nic_tx_start();
+
+    // Initialisation completed
+    printf("\n[CLASSIF] initialisation completed at cycle %d\n", giet_proctime() );
+    
+    // launch load, store and analyse threads
+    for ( x = 0 ; x < x_size ; x++ )
+    {
+        for ( y = 0 ; y < y_size ; y++ )
+        {
+            for ( n = 0 ; n < nprocs ; n++ )
+            {
+                // compute argument value 
+                thread_arg[x][y][n] = (x<<8) | (y<<4) | n;
+
+                if ( n == 0 )       // "load" thread
+                {
+                    if ( giet_pthread_create( &trdid,
+                                              NULL,                  // no attribute
+                                              &load,
+                                              &thread_arg[x][y][n] ) )
+                    {
+                        printf("\n[CLASSIF ERROR] launching thread load\n" );
+                        giet_pthread_exit( NULL );
+                    }
+                    else
+                    { 
+                        printf("\n[CLASSIF] thread load activated : trdid = %x\n", trdid );
+                    }
+                }
+                else if ( n == 1 )  // "store" thread
+                {
+                    if ( giet_pthread_create( &trdid,
+                                              NULL,                  // no attribute
+                                              &store,
+                                              &thread_arg[x][y][n] ) )
+                    {
+                        printf("\n[CLASSIF ERROR] launching thread store\n" );
+                        giet_pthread_exit( NULL );
+                    }
+                    else
+                    { 
+                        printf("\n[CLASSIF] thread store activated : trdid = %x\n", trdid );
+                    }
+                }
+                else              // "analyse" threads            
+                {
+                    if ( giet_pthread_create( &trdid,
+                                              NULL,              // no attribute
+                                              &analyse,
+                                              &thread_arg[x][y][n] ) )
+                    {
+                        printf("\n[CLASSIF ERROR] launching thread analyse\n" );
+                        giet_pthread_exit( NULL );
+                    }
+                    else
+                    { 
+                        printf("\n[CLASSIF] thread analyse activated : trdid = %x\n", trdid );
+                    }  
+                }
+            }
+        }
+    }
+
+    giet_pthread_exit( "completed" );
+    
+} // end main()
+
Index: soft/giet_vm/applications/classif/classif.py
===================================================================
--- soft/giet_vm/applications/classif/classif.py	(revision 707)
+++ soft/giet_vm/applications/classif/classif.py	(revision 708)
@@ -10,8 +10,9 @@
 #  This file describes the mapping of the multi-threaded "classif" 
 #  application on a multi-clusters, multi-processors architecture.
-#  The mapping of tasks on processors is the following:
-#    - one "load" task per cluster containing processors, 
-#    - one "store" task per cluster containing processors, 
-#    - (nprocs-2) "analyse" task per cluster containing processors.
+#  The mapping of threads on processors is the following:
+#    - the "main" on cluster[0][0]  
+#    - one "load" thread per cluster containing processors, 
+#    - one "store" thread per cluster containing processors, 
+#    - (nprocs-2) "analyse" thread per cluster containing processors.
 #  The mapping of virtual segments is the following:
 #    - There is one shared data vseg in cluster[0][0]
@@ -39,21 +40,21 @@
     y_width   = mapping.y_width
 
-    assert (nprocs >= 3)
+    assert (nprocs >= 3) and (nprocs <= 8)
 
     # define vsegs base & size
-    code_base  = 0x10000000
-    code_size  = 0x00010000     # 64 Kbytes (replicated in each cluster)
+    code_base  = 0x10000000     
+    code_size  = 0x00010000     # 64 Kbytes (per cluster)
     
     data_base  = 0x20000000
-    data_size  = 0x00010000     # 64 Kbytes 
+    data_size  = 0x00010000     # 64 Kbytes (non replicated) 
 
     heap_base  = 0x30000000
-    heap_size  = 0x00040000     # 256 Kbytes (per cluster)      
+    heap_size  = 0x00200000     # 2M bytes (per cluster)      
 
     stack_base = 0x40000000 
-    stack_size = 0x00200000     # 2 Mbytes (per cluster)
+    stack_size = 0x00010000     # 64 Kbytes (per thread)
 
     # create vspace
-    vspace = mapping.addVspace( name = 'classif', startname = 'classif_data' )
+    vspace = mapping.addVspace( name = 'classif', startname = 'classif_data', active = False )
     
     # data vseg : shared / cluster[0][0]
@@ -73,7 +74,7 @@
                 mapping.addVseg( vspace, 'classif_heap_%d_%d' %(x,y), base , size, 
                                  'C_WU', vtype = 'HEAP', x = x, y = y, pseg = 'RAM', 
-                                 local = False )
+                                 local = False, big = True )
 
-    # code vsegs : local (one copy in each cluster)
+    # code vsegs : local (one copy per cluster)
     for x in xrange (x_size):
         for y in xrange (y_size):
@@ -88,4 +89,10 @@
 
     # stacks vsegs: local (one stack per processor => nprocs stacks per cluster)
+    # ... plus main_stack in cluster[0][0]
+    mapping.addVseg( vspace, 'main_stack',
+                     stack_base, stack_size, 'C_WU', vtype = 'BUFFER', 
+                     x = 0 , y = 0 , pseg = 'RAM',
+                     local = True )
+
     for x in xrange (x_size):
         for y in xrange (y_size):
@@ -94,13 +101,18 @@
                 for p in xrange( nprocs ):
                     proc_id = (((x * y_size) + y) * nprocs) + p
-                    size    = (stack_size / nprocs) & 0xFFFFF000
-                    base    = stack_base + (proc_id * size)
+                    base    = stack_base + (proc_id * stack_size) + stack_size 
 
                     mapping.addVseg( vspace, 'classif_stack_%d_%d_%d' % (x,y,p), 
-                                     base, size, 'C_WU', vtype = 'BUFFER', 
+                                     base, stack_size, 'C_WU', vtype = 'BUFFER', 
                                      x = x , y = y , pseg = 'RAM',
-                                     local = True, big = True )
+                                     local = True )
 
-    # distributed tasks / one task per processor
+    # distributed threads / one thread per processor
+    # ... plus main on P[0][0][0]
+    mapping.addThread( vspace, 'main', True, 0, 0, 1,
+                       'main_stack',
+                       'classif_heap_0_0',
+                       0 )                      # index in start_vector
+
     for x in xrange (x_size):
         for y in xrange (y_size):
@@ -108,19 +120,18 @@
             if ( mapping.clusters[cluster_id].procs ):
                 for p in xrange( nprocs ):
-                    trdid = (((x * y_size) + y) * nprocs) + p
-                    if  ( p== 0 ):                              # task load
-                        task_index = 2
-                        task_name  = 'load_%d_%d_%d' %(x,y,p)            
-                    elif  ( p== 1 ):                            # task store
-                        task_index = 1
-                        task_name  = 'store_%d_%d_%d' %(x,y,p)            
-                    else :                                      # task analyse
-                        task_index = 0
-                        task_name  = 'analyse_%d_%d_%d' % (x,y,p)
+                    if  ( p== 0 ):                              # thread load
+                        start_index = 3
+                        thread_name = 'load_%d_%d_%d' %(x,y,p)            
+                    elif  ( p== 1 ):                            # thread store
+                        start_index = 2
+                        thread_name = 'store_%d_%d_%d' %(x,y,p)            
+                    else :                                      # thread analyse
+                        start_index = 1
+                        thread_name = 'analyse_%d_%d_%d' % (x,y,p)
 
-                    mapping.addTask( vspace, task_name, trdid, x, y, p,
-                                     'classif_stack_%d_%d_%d' % (x,y,p), 
-                                     'classif_heap_%d_%d' % (x,y),
-                                     task_index )
+                    mapping.addThread( vspace, thread_name, False , x, y, p,
+                                       'classif_stack_%d_%d_%d' % (x,y,p), 
+                                       'classif_heap_%d_%d' % (x,y),
+                                       start_index )   # index in start_vector
 
     # extend mapping name
Index: soft/giet_vm/applications/classif/main.c
===================================================================
--- soft/giet_vm/applications/classif/main.c	(revision 707)
+++ 	(revision )
@@ -1,596 +1,0 @@
-///////////////////////////////////////////////////////////////////////////////////////
-// File   : main.c   (for classif application)
-// Date   : november 2014
-// author : Alain Greiner
-///////////////////////////////////////////////////////////////////////////////////////
-// This multi-threaded application takes a stream of Gigabit Ethernet packets,
-// and makes packet analysis and classification, based on the source MAC address.
-// It uses the NIC peripheral, and the distributed kernel chbufs accessed by the CMA 
-// component to receive and send packets on the Gigabit Ethernet port. 
-//
-// It can run on architectures containing up to 16 * 16 clusters,
-// and up to 8 processors per cluster.
-//
-// It requires N+2 TTY terminals, as each task in cluster[0][0] displays messages.
-//
-// This application is described as a TCG (Task and Communication Graph) 
-// containing (N+2) tasks per cluster:
-// - one "load" task
-// - one "store" task
-// - N "analyse" tasks
-// The containers are distributed (N+2 containers per cluster):
-// - one RX container (part of the kernel rx_chbuf), in the kernel heap.
-// - one TX container (part of the kernel tx-chbuf), in the kernel heap.
-// - N working containers (one per analysis task), in the user heap.
-// In each cluster, the "load", analysis" and "store" tasks communicates through
-// three local MWMR fifos: 
-// - fifo_l2a : tranfer a full container from "load" to "analyse" task.
-// - fifo_a2s : transfer a full container from "analyse" to "store" task.
-// - fifo_s2l : transfer an empty container from "store" to "load" task.
-// For each fifo, one item is a 32 bits word defining the index of an
-// available working container.
-// The pointers on the working containers, and the pointers on the MWMR fifos
-// are global arrays stored in cluster[0][0].
-// a local MWMR fifo containing NB_PROCS_MAX containers (one item = one container).
-// The MWMR fifo descriptors array is defined as a global variable in cluster[0][0]. 
-//
-// Initialisation is done in three steps by the "load" & "store" tasks:
-// 1) Task "load" in cluster[0][0] initialises the heaps in all clusters. Other tasks
-//    are waiting on the global_sync synchronisation variable.
-// 2) Then task "load" in cluster[0][0] initialises the barrier between all "load"
-//    tasks, allocates NIC & CMA RX channel, and starts the NIC_CMA RX transfer.
-//    Other "load" tasks are waiting on the load_sync synchronisation variable.
-//    Task "store" in cluster[0][0] initialises the barrier between all "store" tasks,
-//    allocates NIC & CMA TX channels, and starts the NIC_CMA TX transfer.
-//    Other "store" tasks are waiting on the store_sync synchronisation variable.
-// 3) When this global initialisation is completed, the "load" task in all clusters
-//    allocates the working containers and the MWMR fifos descriptors from the
-//    user local heap. In each cluster, the "analyse" and "store" tasks are waiting
-//    the local initialisation completion on the local_sync[x][y] variables.
-//
-// When initialisation is completed, all tasks loop on containers:
-// 1) The "load" task get an empty working container from the fifo_s2l,
-//    transfer one container from the kernel rx_chbuf to this user container,
-//    and transfer ownership of this container to one "analysis" task by writing
-//    into the fifo_l2a.    
-// 2) The "analyse" task get one working container from the fifo_l2a, analyse
-//    each packet header, compute the packet type (depending on the SRC MAC address),
-//    increment the correspondint classification counter, and transpose the SRC
-//    and the DST MAC addresses fot TX tranmission.
-// 3) The "store" task transfer get a full working container from the fifo_a2s,
-//    transfer this user container content to the the kernel tx_chbuf,
-//    and transfer ownership of this empty container to the "load" task by writing
-//    into the fifo_s2l.   
-//     
-// Instrumentation results display is done by the "store" task in cluster[0][0]
-// when all "store" tasks completed the number of clusters specified by the
-// CONTAINERS_MAX parameter.
-///////////////////////////////////////////////////////////////////////////////////////
-
-#include "stdio.h"
-#include "user_barrier.h"
-#include "malloc.h"
-#include "user_lock.h"
-#include "mwmr_channel.h"
-
-#define X_SIZE_MAX      16
-#define Y_SIZE_MAX      16
-#define NPROCS_MAX      8
-#define CONTAINERS_MAX  50
-#define VERBOSE_ANALYSE 0
-
-///////////////////////////////////////////////////////////////////////////////////////
-//    Global variables
-// The MWMR channels (descriptors and buffers), as well as the working containers 
-// used by the "analysis" tasks are distributed in clusters.
-// But the pointers on these distributed structures are stored in cluster[0][0].
-///////////////////////////////////////////////////////////////////////////////////////
-
-// pointers on distributed containers
-unsigned int*       container[X_SIZE_MAX][Y_SIZE_MAX][NPROCS_MAX-2];  
-
-// pointers on distributed mwmr fifos containing : temp[x][y][l] container descriptors
-mwmr_channel_t*     mwmr_l2a[X_SIZE_MAX][Y_SIZE_MAX];  
-mwmr_channel_t*     mwmr_a2s[X_SIZE_MAX][Y_SIZE_MAX];
-mwmr_channel_t*     mwmr_s2l[X_SIZE_MAX][Y_SIZE_MAX]; 
-
-// local synchros signaling local MWMR fifos initialisation completion
-volatile unsigned int        local_sync[X_SIZE_MAX][Y_SIZE_MAX];  
-
-// global synchro signaling global initialisation completion
-volatile unsigned int        global_sync = 0;
-volatile unsigned int        load_sync   = 0; 
-volatile unsigned int        store_sync  = 0; 
-
-// instrumentation counters
-unsigned int        counter[16];
-
-// distributed barrier between "load" tasks
-giet_sqt_barrier_t  rx_barrier;
-
-// distributed barrier between "store" tasks
-giet_sqt_barrier_t  tx_barrier;
-
-// NIC_RX and NIC_TX channel index
-unsigned int        nic_rx_channel;
-unsigned int        nic_tx_channel;
-
-/////////////////////////////////////////
-__attribute__ ((constructor)) void load()
-/////////////////////////////////////////
-{
-    // each "load" task get platform parameters
-    unsigned int    x_size;			// number of clusters in a row
-    unsigned int    y_size;                     // number of clusters in a column
-    unsigned int    nprocs;                     // number of processors per cluster
-    giet_procs_number( &x_size, &y_size, &nprocs );
-
-    giet_assert( (x_size <= X_SIZE_MAX) && 
-                 (y_size <= Y_SIZE_MAX) && 
-                 (nprocs <= NPROCS_MAX) , 
-                 "[CLASSIF ERROR] illegal platform parameters" );
-
-    // each "load" task get processor identifiers
-    unsigned int    x;
-    unsigned int    y;
-    unsigned int    l;
-    giet_proc_xyp( &x, &y, &l );
-
-    // "load" task[0][0]
-    // - initialises the heap in every cluster
-    // - initialises barrier between all load tasks,
-    // - allocates the NIC & CMA RX channels, and start the NIC_CMA RX transfer.
-    // Other "load" tasks wait completion
-    if ( (x==0) && (y==0) )
-    {
-        // allocate a private TTY
-        giet_tty_alloc(0);
-
-        giet_tty_printf("\n*** Task load on P[%d][%d][%d] starts at cycle %d\n"
-                        "  x_size = %d / y_size = %d / nprocs = %d\n",
-                        x , y , l , giet_proctime() , x_size, y_size, nprocs );
-
-        unsigned int xid;  // x cluster coordinate index
-        unsigned int yid;  // y cluster coordinate index
-
-        for ( xid = 0 ; xid < x_size ; xid++ )
-        {
-	        for ( yid = 0 ; yid < y_size ; yid++ )
-	        {
-	            heap_init( xid, yid );
-	        }
-        }
-    
-	    global_sync = 1;
-
-        sqt_barrier_init( &rx_barrier, x_size , y_size , 1 );
-        nic_rx_channel = giet_nic_rx_alloc( x_size , y_size );
-        giet_nic_rx_start( nic_rx_channel );
-        load_sync = 1;
-    }
-    else
-    {
-        while ( load_sync == 0 ) asm volatile ("nop");
-    }    
-
-    // each load tasks allocates containers[x][y][n] (from local heap)
-    // and register pointers in the local stack
-    unsigned int   n;
-    unsigned int*  cont[NPROCS_MAX-2]; 
-    unsigned int   analysis_tasks = nprocs-2;
-
-    for ( n = 0 ; n < analysis_tasks ; n++ )
-    {
-        container[x][y][n] = malloc( 4096 );
-        cont[n]            = container[x][y][n];
-    }
-    
-    // each load task allocates data buffers for mwmr fifos (from local heap)
-    unsigned int*  data_l2a = malloc( analysis_tasks<<2 );
-    unsigned int*  data_a2s = malloc( analysis_tasks<<2 );
-    unsigned int*  data_s2l = malloc( analysis_tasks<<2 );
-
-    // each load task allocates mwmr fifos descriptors (from local heap)
-    mwmr_l2a[x][y] = malloc( sizeof(mwmr_channel_t) );
-    mwmr_a2s[x][y] = malloc( sizeof(mwmr_channel_t) );
-    mwmr_s2l[x][y] = malloc( sizeof(mwmr_channel_t) );
-
-    // each load task registers local pointers on mwmr fifos in local stack
-    mwmr_channel_t* fifo_l2a = mwmr_l2a[x][y];
-    mwmr_channel_t* fifo_a2s = mwmr_a2s[x][y];
-    mwmr_channel_t* fifo_s2l = mwmr_s2l[x][y];
-
-    // each load task initialises local mwmr fifos descriptors
-    // ( width = 4 bytes / depth = number of analysis tasks )
-    mwmr_init( fifo_l2a , data_l2a , 1 , analysis_tasks );
-    mwmr_init( fifo_a2s , data_a2s , 1 , analysis_tasks );
-    mwmr_init( fifo_s2l , data_s2l , 1 , analysis_tasks );
-
-    
-    // each load task initialises local containers as empty in fifo_s2l
-    for ( n = 0 ; n < analysis_tasks ; n++ ) mwmr_write( fifo_s2l , &n , 1 );
-
-    // each load task[x][y] signals mwmr fifos initialisation completion
-    // to other tasks in same cluster[x][y]
-    local_sync[x][y] = 1;
-
-    // load task[0][0] displays status
-    if ( (x==0) && (y==0) )
-    giet_tty_printf("\n*** Task load on P[%d,%d,%d] enters main loop at cycle %d\n"
-                    "      nic_rx_channel = %d / nic_tx_channel = %d\n"
-                    "      &mwmr_l2a  = %x / &data_l2a  = %x\n"
-                    "      &mwmr_a2s  = %x / &data_a2s  = %x\n"
-                    "      &mwmr_s2l  = %x / &data_s2l  = %x\n"
-                    "      &cont[0]   = %x\n"
-                    "      x_size = %d / y_size = %d / nprocs = %d\n",
-                    x , y , l , giet_proctime(), 
-                    nic_rx_channel , nic_tx_channel,
-                    (unsigned int)fifo_l2a, (unsigned int)data_l2a,
-                    (unsigned int)fifo_a2s, (unsigned int)data_a2s,
-                    (unsigned int)fifo_s2l, (unsigned int)data_s2l,
-                    (unsigned int)cont[0],
-                    x_size, y_size, nprocs );
- 
-    /////////////////////////////////////////////////////////////
-    // All load tasks enter the main loop (on containers)
-    unsigned int  count = 0;     // loaded containers count
-    unsigned int  index;         // available container index
-    unsigned int* temp;          // pointer on available container
-
-    while ( count < CONTAINERS_MAX ) 
-    { 
-        // get one empty container index from fifo_s2l
-        mwmr_read( fifo_s2l , &index , 1 );
-        temp = cont[index];
-
-        // get one container from  kernel rx_chbuf
-        giet_nic_rx_move( nic_rx_channel, temp );
-
-        // get packets number
-        unsigned int npackets = temp[0] & 0x0000FFFF;
-        unsigned int nwords   = temp[0] >> 16;
-
-        if ( (x==0) && (y==0) )
-        giet_tty_printf("\n*** Task load on P[%d,%d,%d] get container %d at cycle %d"
-                        " : %d packets / %d words\n",
-                        x, y, l, count, giet_proctime(), npackets, nwords );
-
-        // put the full container index to fifo_l2a
-        mwmr_write( fifo_l2a, &index , 1 );
-
-        count++;
-    }
-
-    // all "load" tasks synchronise before stats
-    sqt_barrier_wait( &rx_barrier );
-
-    // "load" task[0][0] stops the NIC_CMA RX transfer and displays stats
-    if ( (x==0) && (y==0) ) 
-    {
-        giet_nic_rx_stop( nic_rx_channel );
-        giet_nic_rx_stats( nic_rx_channel );
-    }
-
-    // all "load" task exit
-    giet_exit("Task completed");
- 
-} // end load()
-
-
-//////////////////////////////////////////
-__attribute__ ((constructor)) void store()
-//////////////////////////////////////////
-{
-    // each "load" task get platform parameters
-    unsigned int    x_size;						// number of clusters in row
-    unsigned int    y_size;                     // number of clusters in a column
-    unsigned int    nprocs;                     // number of processors per cluster
-    giet_procs_number( &x_size, &y_size, &nprocs );
-
-    // get processor identifiers
-    unsigned int    x;
-    unsigned int    y;
-    unsigned int    l;
-    giet_proc_xyp( &x, &y, &l );
-
-    // "Store" tasks wait completion of heaps initialization
-    while ( global_sync == 0 ) asm volatile ("nop");
-
-    // "store" task[0][0] initialises the barrier between all "store" tasks,
-    // allocates NIC & CMA TX channels, and starts the NIC_CMA TX transfer.
-    // Other "store" tasks wait completion.
-    if ( (x==0) && (y==0) )
-    {
-        // allocate a private TTY
-        giet_tty_alloc(0);
-
-        giet_tty_printf("\n*** Task store on P[%d][%d][%d] starts at cycle %d\n"
-                        "  x_size = %d / y_size = %d / nprocs = %d\n",
-                        x , y , l , giet_proctime() , x_size, y_size, nprocs );
- 
-        sqt_barrier_init( &tx_barrier , x_size , y_size , 1 );
-        nic_tx_channel = giet_nic_tx_alloc( x_size , y_size );
-        giet_nic_tx_start( nic_tx_channel );
-        store_sync = 1;
-    }
-    else
-    {
-        while ( store_sync == 0 ) asm volatile ("nop");
-    }    
-
-    // all "store" tasks wait mwmr channels initialisation
-    while ( local_sync[x][y] == 0 ) asm volatile ("nop");
-
-    // each "store" tasks register pointers on working containers in local stack
-    unsigned int   n;
-    unsigned int   analysis_tasks = nprocs-2;
-    unsigned int*  cont[NPROCS_MAX-2]; 
-
-    for ( n = 0 ; n < analysis_tasks ; n++ )
-    {
-        cont[n] = container[x][y][n];
-    }
-    
-    // all "store" tasks register pointers on mwmr fifos in local stack
-    mwmr_channel_t* fifo_l2a = mwmr_l2a[x][y];
-    mwmr_channel_t* fifo_a2s = mwmr_a2s[x][y];
-    mwmr_channel_t* fifo_s2l = mwmr_s2l[x][y];
-
-    // "store" task[0][0] displays status
-    if ( (x==0) && (y==0) )
-    giet_tty_printf("\n*** Task store on P[%d,%d,%d] enters main loop at cycle %d\n"
-                    "      &mwmr_l2a  = %x\n"
-                    "      &mwmr_a2s  = %x\n"
-                    "      &mwmr_s2l  = %x\n"
-                    "      &cont[0]   = %x\n",
-                    x , y , l , giet_proctime(), 
-                    (unsigned int)fifo_l2a,
-                    (unsigned int)fifo_a2s,
-                    (unsigned int)fifo_s2l,
-                    (unsigned int)cont[0] );
-
-
-    /////////////////////////////////////////////////////////////
-    // all "store" tasks enter the main loop (on containers)
-    unsigned int count = 0;     // stored containers count
-    unsigned int index;         // empty container index
-    unsigned int* temp;         // pointer on empty container
-
-    while ( count < CONTAINERS_MAX ) 
-    { 
-        // get one working container index from fifo_a2s
-        mwmr_read( fifo_a2s , &index , 1 );
-        temp = cont[index];
-
-        // put one container to  kernel tx_chbuf
-        giet_nic_tx_move( nic_tx_channel, temp );
-
-        // get packets number
-        unsigned int npackets = temp[0] & 0x0000FFFF;
-        unsigned int nwords   = temp[0] >> 16;
-
-        if ( (x==0) && (y==0) )
-        giet_tty_printf("\n*** Task store on P[%d,%d,%d] get container %d at cycle %d"
-                        " : %d packets / %d words\n",
-                        x, y, l, count, giet_proctime(), npackets, nwords );
-
-        // put the working container index to fifo_s2l
-        mwmr_write( fifo_s2l, &index , 1 );
-
-        count++;
-    }
-
-    // all "store" tasks synchronise before result display
-    sqt_barrier_wait( &tx_barrier );
-
-    // "store" task[0,0] stops NIC_CMA TX transfer and displays results 
-    if ( (x==0) && (y==0) )
-    {
-        giet_nic_tx_stop( nic_tx_channel );
-
-        giet_tty_printf("\n@@@@ Classification Results @@@\n"
-                        " - TYPE 0 : %d packets\n"
-                        " - TYPE 1 : %d packets\n"
-                        " - TYPE 2 : %d packets\n"
-                        " - TYPE 3 : %d packets\n"
-                        " - TYPE 4 : %d packets\n"
-                        " - TYPE 5 : %d packets\n"
-                        " - TYPE 6 : %d packets\n"
-                        " - TYPE 7 : %d packets\n"
-                        " - TYPE 8 : %d packets\n"
-                        " - TYPE 9 : %d packets\n"
-                        " - TYPE A : %d packets\n"
-                        " - TYPE B : %d packets\n"
-                        " - TYPE C : %d packets\n"
-                        " - TYPE D : %d packets\n"
-                        " - TYPE E : %d packets\n"
-                        " - TYPE F : %d packets\n"
-                        "    TOTAL = %d packets\n",
-                        counter[0x0], counter[0x1], counter[0x2], counter[0x3],
-                        counter[0x4], counter[0x5], counter[0x6], counter[0x7],
-                        counter[0x8], counter[0x9], counter[0xA], counter[0xB],
-                        counter[0xC], counter[0xD], counter[0xE], counter[0xF],
-                        counter[0x0]+ counter[0x1]+ counter[0x2]+ counter[0x3]+
-                        counter[0x4]+ counter[0x5]+ counter[0x6]+ counter[0x7]+
-                        counter[0x8]+ counter[0x9]+ counter[0xA]+ counter[0xB]+
-                        counter[0xC]+ counter[0xD]+ counter[0xE]+ counter[0xF] );
-
-        giet_nic_tx_stats( nic_tx_channel );
-    }
-
-    // all "store" task exit
-    giet_exit("Task completed");
-
-} // end store()
-
-
-////////////////////////////////////////////
-__attribute__ ((constructor)) void analyse()
-////////////////////////////////////////////
-{
-    // each "load" task get platform parameters
-    unsigned int    x_size;						// number of clusters in row
-    unsigned int    y_size;                     // number of clusters in a column
-    unsigned int    nprocs;                     // number of processors per cluster
-    giet_procs_number( &x_size, &y_size, &nprocs );
-
-    // get processor identifiers
-    unsigned int    x;
-    unsigned int    y;
-    unsigned int    l;
-    giet_proc_xyp( &x, &y, &l );
-
-    if ( (x==0) && (y==0) )
-    {
-        // allocate a private TTY
-        giet_tty_alloc(0);
-
-        giet_tty_printf("\n*** Task analyse on P[%d][%d][%d] starts at cycle %d\n"
-                        "  x_size = %d / y_size = %d / nprocs = %d\n",
-                        x , y , l , giet_proctime() , x_size, y_size, nprocs );
-    }
-
-    // all "analyse" tasks wait heaps and mwmr channels initialisation
-    while ( local_sync[x][y] == 0 ) asm volatile ("nop");
-
-    // all "analyse" tasks register pointers on working containers in local stack
-    unsigned int   n;
-    unsigned int   analysis_tasks = nprocs-2;
-    unsigned int*  cont[NPROCS_MAX-2]; 
-    for ( n = 0 ; n < analysis_tasks ; n++ )
-    {
-        cont[n] = container[x][y][n];
-    }
-
-    // all "analyse" tasks register pointers on mwmr fifos in local stack
-    mwmr_channel_t* fifo_l2a = mwmr_l2a[x][y];
-    mwmr_channel_t* fifo_a2s = mwmr_a2s[x][y];
-
-    // "analyse" task[0][0] display status
-    if ( (x==0) && (y==0) )
-    giet_tty_printf("\n*** Task analyse on P[%d,%d,%d] enters main loop at cycle %d\n"
-                    "       &mwmr_l2a = %x\n"
-                    "       &mwmr_a2s = %x\n"
-                    "       &cont[0]  = %x\n",
-                    x, y, l, giet_proctime(), 
-                    (unsigned int)fifo_l2a,
-                    (unsigned int)fifo_a2s,
-                    (unsigned int)cont[0] );
-      
-    /////////////////////////////////////////////////////////////
-    // all "analyse" tasks enter the main loop (on containers)
-    unsigned int  index;           // available container index
-    unsigned int* temp;            // pointer on available container
-    unsigned int  nwords;          // number of words in container
-    unsigned int  npackets;        // number of packets in container
-    unsigned int  length;          // number of bytes in current packet
-    unsigned int  first;           // current packet first word in container
-    unsigned int  type;            // current packet type
-    unsigned int  p;               // current packet index
-
-#if VERBOSE_ANALYSE
-    unsigned int       verbose_len[10]; // save length for all packets in one container
-    unsigned long long verbose_dst[10]; // save length for all packets in one container
-    unsigned long long verbose_src[10]; // save length for all packets in one container
-#endif
-
-    while ( 1 )
-    { 
-
-#if VERBOSE_ANALYSE 
-            for( p = 0 ; p < 10 ; p++ )
-            {
-                verbose_len[p] = 0;
-                verbose_dst[p] = 0;
-                verbose_src[p] = 0;
-            }
-#endif
-        // get one working container index from fifo_l2a
-        mwmr_read( fifo_l2a , &index , 1 );
-        temp = cont[index];
-
-        // get packets number and words number
-        npackets = temp[0] & 0x0000FFFF;
-        nwords   = temp[0] >> 16;
-
-        if ( (x==0) && (y==0) )
-        giet_tty_printf("\n*** Task analyse on P[%d,%d,%d] get container at cycle %d"
-                        " : %d packets / %d words\n",
-						x, y, l, giet_proctime(), npackets, nwords );
-
-        // initialize word index in container
-        first = 34;
-
-        // loop on packets
-        for( p = 0 ; p < npackets ; p++ )
-        {
-            // get packet length from container header
-            if ( (p & 0x1) == 0 )  length = temp[1+(p>>1)] >> 16;
-            else                   length = temp[1+(p>>1)] & 0x0000FFFF;
-
-            // compute packet DST and SRC MAC addresses
-            unsigned int word0 = temp[first];
-            unsigned int word1 = temp[first + 1];
-            unsigned int word2 = temp[first + 2];
-
-#if VERBOSE_ANALYSE 
-            unsigned long long dst = ((unsigned long long)(word1 & 0xFFFF0000)>>16) |
-                                     (((unsigned long long)word0)<<16);
-            unsigned long long src = ((unsigned long long)(word1 & 0x0000FFFF)<<32) |
-                                     ((unsigned long long)word2);
-            if ( p < 10 )
-            {
-                verbose_len[p] = length;
-                verbose_dst[p] = dst;
-                verbose_src[p] = src;
-            }
-#endif
-            // compute type from SRC MAC address and increment counter
-            type = word1 & 0x0000000F;
-            atomic_increment( &counter[type], 1 );
-
-            // exchange SRC & DST MAC addresses for TX
-            temp[first]     = ((word1 & 0x0000FFFF)<<16) | ((word2 & 0xFFFF0000)>>16);
-            temp[first + 1] = ((word2 & 0x0000FFFF)<<16) | ((word0 & 0xFFFF0000)>>16);
-            temp[first + 2] = ((word0 & 0x0000FFFF)<<16) | ((word1 & 0xFFFF0000)>>16);
-
-            // update first word index 
-            if ( length & 0x3 ) first += (length>>2)+1;
-            else                first += (length>>2);
-        }
-        
-#if VERBOSE_ANALYSE 
-        if ( (x==0) && (y==0) )
-        giet_tty_printf("\n*** Task analyse on P[%d,%d,%d] completes at cycle %d\n"
-                        "   - Packet 0 : plen = %d / dst_mac = %l / src_mac = %l\n"
-                        "   - Packet 1 : plen = %d / dst_mac = %l / src_mac = %l\n"
-                        "   - Packet 2 : plen = %d / dst_mac = %l / src_mac = %l\n"
-                        "   - Packet 3 : plen = %d / dst_mac = %l / src_mac = %l\n"
-                        "   - Packet 4 : plen = %d / dst_mac = %l / src_mac = %l\n"
-                        "   - Packet 5 : plen = %d / dst_mac = %l / src_mac = %l\n"
-                        "   - Packet 6 : plen = %d / dst_mac = %l / src_mac = %l\n"
-                        "   - Packet 7 : plen = %d / dst_mac = %l / src_mac = %l\n"
-                        "   - Packet 8 : plen = %d / dst_mac = %l / src_mac = %l\n"
-                        "   - Packet 9 : plen = %d / dst_mac = %l / src_mac = %l\n",
-                        x , y , l , giet_proctime() , 
-                        verbose_len[0] , verbose_dst[0] , verbose_src[0] ,
-                        verbose_len[1] , verbose_dst[1] , verbose_src[1] ,
-                        verbose_len[2] , verbose_dst[2] , verbose_src[2] ,
-                        verbose_len[3] , verbose_dst[3] , verbose_src[3] ,
-                        verbose_len[4] , verbose_dst[4] , verbose_src[4] ,
-                        verbose_len[5] , verbose_dst[5] , verbose_src[5] ,
-                        verbose_len[6] , verbose_dst[6] , verbose_src[6] ,
-                        verbose_len[7] , verbose_dst[7] , verbose_src[7] ,
-                        verbose_len[8] , verbose_dst[8] , verbose_src[8] ,
-                        verbose_len[9] , verbose_dst[9] , verbose_src[9] );
-#endif
-            
-        // pseudo-random delay
-        unsigned int delay = giet_rand()>>3;
-        for( p = 0 ; p < delay ; p++ ) asm volatile ("nop");
-
-        // put the working container index to fifo_a2s
-        mwmr_write( fifo_a2s , &index , 1 );
-    }
-} // end analyse()
-
Index: soft/giet_vm/applications/convol/Makefile
===================================================================
--- soft/giet_vm/applications/convol/Makefile	(revision 707)
+++ soft/giet_vm/applications/convol/Makefile	(revision 708)
@@ -1,10 +1,16 @@
+
+CC = mipsel-unknown-elf-gcc
+AS = mipsel-unknown-elf-as
+LD = mipsel-unknown-elf-ld
+DU = mipsel-unknown-elf-objdump
+AR = mipsel-unknown-elf-ar
 
 APP_NAME = convol
 
-OBJS= main.o 
+OBJS= convol.o 
 
 LIBS= -L../../build/libs -luser
 
-INCLUDES = -I../../giet_libs -I. -I../..
+INCLUDES = -I../../giet_libs -I. -I../.. -I../../giet_xml
 
 LIB_DEPS = ../../build/libs/libuser.a
Index: soft/giet_vm/applications/convol/convol.c
===================================================================
--- soft/giet_vm/applications/convol/convol.c	(revision 708)
+++ soft/giet_vm/applications/convol/convol.c	(revision 708)
@@ -0,0 +1,823 @@
+///////////////////////////////////////////////////////////////////////////////////////
+// File   : convol.c  
+// Date   : june 2014
+// author : Alain Greiner
+///////////////////////////////////////////////////////////////////////////////////////
+// This multi-threaded application implements a 2D convolution product.  
+// It can run on a multi-processors, multi-clusters architecture, with one thread
+// per processor, and uses the POSIX threads API.
+// 
+// The main() function can be launched on any processor P[x,y,l].
+// It makes the initialisations, launch (N-1) threads to run the execute() function
+// on the (N-1) other processors than P[x,y,l], call himself the execute() function, 
+// and finally call the instrument() function to display instrumentation results 
+// when the parallel execution is completed.
+//
+// The convolution kernel is [201]*[35] pixels, but it can be factored in two
+// independant line and column convolution products.
+// The five buffers containing the image are distributed in clusters.
+// 
+// The (1024 * 1024) pixels image is read from a file (2 bytes per pixel).
+//
+// - number of clusters containing processors must be power of 2 no larger than 256.
+// - number of processors per cluster must be power of 2 no larger than 8.
+///////////////////////////////////////////////////////////////////////////////////////
+
+#include "stdio.h"
+#include "stdlib.h"
+#include "user_barrier.h"
+#include "malloc.h"
+
+#define USE_SQT_BARRIER            1
+#define VERBOSE                    1
+#define SUPER_VERBOSE              0
+
+#define X_SIZE_MAX                 16
+#define Y_SIZE_MAX                 16
+#define PROCS_MAX                  8
+#define CLUSTERS_MAX               (X_SIZE_MAX * Y_SIZE_MAX)
+
+#define INITIAL_DISPLAY_ENABLE     0
+#define FINAL_DISPLAY_ENABLE       1
+
+#define PIXEL_SIZE                 2
+#define NL                         1024
+#define NP                         1024
+#define NB_PIXELS                  (NP * NL)
+#define FRAME_SIZE                 (NB_PIXELS * PIXEL_SIZE)
+
+#define SEEK_SET                   0
+
+#define TA(c,l,p)  (A[c][((NP) * (l)) + (p)])
+#define TB(c,p,l)  (B[c][((NL) * (p)) + (l)])
+#define TC(c,l,p)  (C[c][((NP) * (l)) + (p)])
+#define TD(c,l,p)  (D[c][((NP) * (l)) + (p)])
+#define TZ(c,l,p)  (Z[c][((NP) * (l)) + (p)])
+
+#define max(x,y) ((x) > (y) ? (x) : (y))
+#define min(x,y) ((x) < (y) ? (x) : (y))
+
+// macro to use a shared TTY
+#define printf(...);  { lock_acquire( &tty_lock ); \
+                        giet_tty_printf(__VA_ARGS__);  \
+                        lock_release( &tty_lock ); }
+
+//////////////////////////////////////////////////////////
+//   global variables stored in seg_data in cluster[0,0]
+//////////////////////////////////////////////////////////
+
+// Instrumentation counters (cluster_id, lpid]
+
+unsigned int START[CLUSTERS_MAX][PROCS_MAX];
+unsigned int H_BEG[CLUSTERS_MAX][PROCS_MAX];
+unsigned int H_END[CLUSTERS_MAX][PROCS_MAX];
+unsigned int V_BEG[CLUSTERS_MAX][PROCS_MAX];
+unsigned int V_END[CLUSTERS_MAX][PROCS_MAX];
+unsigned int D_BEG[CLUSTERS_MAX][PROCS_MAX];
+unsigned int D_END[CLUSTERS_MAX][PROCS_MAX];
+
+// global synchronization barrier
+
+#if USE_SQT_BARRIER
+giet_sqt_barrier_t  barrier;
+#else
+giet_barrier_t      barrier;
+#endif
+
+volatile unsigned int barrier_init_ok    = 0;
+volatile unsigned int load_image_ok      = 0;
+volatile unsigned int instrumentation_ok = 0;
+
+// lock protecting access to shared TTY
+user_lock_t         tty_lock;
+
+// global pointers on distributed buffers in all clusters
+unsigned short * GA[CLUSTERS_MAX];
+int *            GB[CLUSTERS_MAX];
+int *            GC[CLUSTERS_MAX];
+int *            GD[CLUSTERS_MAX];
+unsigned char *  GZ[CLUSTERS_MAX];
+
+
+
+////////////////////////////////////////////
+__attribute__ ((constructor)) void execute()
+////////////////////////////////////////////
+{
+    /////////////////////////////////////////////////////////////////////////////////////
+    // Each thread[x,y,p] initialises the convolution kernel parameters in local stack.
+    // The values defined in the next 12 lines are Philips proprietary information.
+    /////////////////////////////////////////////////////////////////////////////////////
+
+    int   vnorm  = 115;
+    int   vf[35] = { 1, 1, 2, 2, 2,
+                     2, 3, 3, 3, 4,
+                     4, 4, 4, 5, 5,
+                     5, 5, 5, 5, 5,
+                     5, 5, 4, 4, 4,
+                     4, 3, 3, 3, 2,
+                     2, 2, 2, 1, 1 };
+
+    int hrange = 100;
+    int hnorm  = 201;
+
+    // get plat-form parameters
+    unsigned int x_size;             // number of clusters in a row
+    unsigned int y_size;             // number of clusters in a column
+    unsigned int nprocs;             // number of processors per cluster
+    giet_procs_number( &x_size , &y_size , &nprocs );
+
+    // get processor identifiers
+    unsigned int x;                  // x coordinate
+    unsigned int y;                  // y coordinate
+    unsigned int lpid;               // local proc index 
+    giet_proc_xyp( &x, &y, &lpid );
+
+    // indexes for loops
+    int c; // cluster index 
+    int l; // line index 
+    int p; // pixel index 
+    int z; // vertical filter index 
+
+    int          file       = 0;                            // file descriptor
+    unsigned int nclusters  = x_size * y_size;              // number of clusters
+    unsigned int cluster_id = (x * y_size) + y;             // continuous cluster index
+    unsigned int thread_id  = (cluster_id * nprocs) + lpid; // continuous thread index
+    unsigned int nthreads   = nclusters * nprocs;           // number of threads
+    unsigned int frame_size = FRAME_SIZE;                   // total size (bytes)
+    unsigned int lines_per_thread   = NL / nthreads;        // lines per thread
+    unsigned int lines_per_cluster  = NL / nclusters;       // lines per cluster
+    unsigned int pixels_per_thread  = NP / nthreads;        // columns per thread
+    unsigned int pixels_per_cluster = NP / nclusters;       // columns per cluster
+
+    int first, last;
+
+    unsigned int date = giet_proctime();
+    START[cluster_id][lpid] = date;
+
+    /////////////////////////////////////////////////////////////////////
+    // Each thread[x][y][0] allocate the global buffers in cluster(x,y)
+    /////////////////////////////////////////////////////////////////////
+    if ( lpid == 0 )
+    {
+
+#if VERBOSE
+printf( "\n[CONVOL] thread[%d,%d,%d] enters malloc at cycle %d\n", 
+                 x,y,lpid, date );
+#endif
+
+        GA[cluster_id] = remote_malloc( (FRAME_SIZE/nclusters)   , x , y );
+        GB[cluster_id] = remote_malloc( (FRAME_SIZE/nclusters)*2 , x , y );
+        GC[cluster_id] = remote_malloc( (FRAME_SIZE/nclusters)*2 , x , y );
+        GD[cluster_id] = remote_malloc( (FRAME_SIZE/nclusters)*2 , x , y );
+        GZ[cluster_id] = remote_malloc( (FRAME_SIZE/nclusters)/2 , x , y );
+        
+#if VERBOSE
+printf( "\n[CONVOL]  Shared Buffer Virtual Addresses in cluster(%d,%d)\n"
+        "### GA = %x\n"
+        "### GB = %x\n"               
+        "### GC = %x\n"               
+        "### GD = %x\n"               
+        "### GZ = %x\n",
+        x, y,
+        GA[cluster_id],
+        GB[cluster_id],
+        GC[cluster_id],
+        GD[cluster_id],
+        GZ[cluster_id] );
+#endif
+    }
+
+    ///////////////////////////////
+    #if USE_SQT_BARRIER
+    sqt_barrier_wait( &barrier );
+    #else
+    barrier_wait( &barrier );
+    #endif
+
+    //////////////////////////////////////////////////////////////////////
+    // Each thread[x,y,p] initialise in its private stack a copy of the
+    // arrays of pointers on the shared, distributed buffers.
+    //////////////////////////////////////////////////////////////////////
+
+    unsigned short * A[CLUSTERS_MAX];
+    int            * B[CLUSTERS_MAX];
+    int            * C[CLUSTERS_MAX];
+    int            * D[CLUSTERS_MAX];
+    unsigned char  * Z[CLUSTERS_MAX];
+
+    for (c = 0; c < nclusters; c++)
+    {
+        A[c] = GA[c];
+        B[c] = GB[c];
+        C[c] = GC[c];
+        D[c] = GD[c];
+        Z[c] = GZ[c];
+    }
+
+    /////////////////////////////////////////////////////////////////////////////
+    // Ech thread[x,y,0] open the file containing image, and load it from disk 
+    // to the local A[c] buffer (frame_size / nclusters loaded in each cluster).
+    // Other threads are waiting on the init_ok condition.
+    ////////////////////////////////////////////////////////////////////////////
+    if ( lpid==0 )
+    {
+        // open file
+        file = giet_fat_open( "/misc/philips_1024.raw" , 0 );
+        if (file < 0 )
+        {
+            printf("\n[CONVOL ERROR] thread[%d,%d,%d] "
+                   "cannot open file /misc/philips_1024.raw",
+                   x, y, lpid );
+            giet_pthread_exit( NULL );
+        } 
+ 
+#if VERBOSE
+printf( "\n[CONVOL] thread[%d,%d,%d] open file /misc/philips_1024.raw at cycle %d\n", 
+        x, y, lpid, giet_proctime() );
+#endif
+
+        unsigned int offset = (frame_size/nclusters)*cluster_id;
+        unsigned int size   = frame_size/nclusters;
+
+        if ( giet_fat_lseek( file,
+                             offset,
+                             SEEK_SET ) )
+        {
+            printf("\n[CONVOL ERROR] thread[%d,%d,%d] "
+                   "cannot seek file /misc/philips_1024.raw",
+                   x, y, lpid );
+            giet_pthread_exit( NULL );
+        } 
+
+        if ( giet_fat_read( file,
+                            A[cluster_id],
+                            size ) != size )
+        {
+            printf("\n[CONVOL ERROR] thread[%d,%d,%d] "
+                   "cannot read file /misc/philips_1024.raw",
+                   x, y, lpid );
+            giet_pthread_exit( NULL );
+        }
+ 
+#if VERBOSE
+printf( "\n[CONVOL] thread[%d,%d,%d] load file /misc/philips_1024.raw at cycle %d\n", 
+        x, y, lpid, giet_proctime() );
+#endif
+
+    }
+
+    ///////////////////////////////
+    #if USE_SQT_BARRIER
+    sqt_barrier_wait( &barrier );
+    #else
+    barrier_wait( &barrier );
+    #endif
+
+    /////////////////////////////////////////////////////////////////////////////
+    // Optionnal parallel display of the initial image stored in A[c] buffers.
+    // Eah thread[x,y,p] displays (NL/nthreads) lines. (one byte per pixel).
+    /////////////////////////////////////////////////////////////////////////////
+
+    if ( INITIAL_DISPLAY_ENABLE )
+    {
+
+#if VERBOSE
+printf( "\n[CONVOL] thread[%d,%d,%d] starts initial display at cycle %d\n",
+        x, y, lpid, giet_proctime() );
+#endif
+
+        unsigned int line;
+        unsigned int offset = lines_per_thread * lpid;
+
+        for ( l = 0 ; l < lines_per_thread ; l++ )
+        {
+            line = offset + l;
+
+            for ( p = 0 ; p < NP ; p++ )
+            {
+                TZ(cluster_id, line, p) = (unsigned char)(TA(cluster_id, line, p) >> 8);
+            }
+
+            giet_fbf_sync_write( NP*(l + (thread_id * lines_per_thread) ), 
+                                 &TZ(cluster_id, line, 0), 
+                                 NP);
+        }
+
+#if VERBOSE 
+printf( "\n[CONVOL] thread[%d,%d,%d] completes initial display at cycle %d\n",
+        x, y, lpid, giet_proctime() );
+#endif
+
+        ////////////////////////////
+        #if USE_SQT_BARRIER
+        sqt_barrier_wait( &barrier );
+        #else
+        barrier_wait( &barrier );
+        #endif
+
+    }
+
+    ////////////////////////////////////////////////////////
+    // parallel horizontal filter : 
+    // B <= transpose(FH(A))
+    // D <= A - FH(A)
+    // Each thread computes (NL/nthreads) lines 
+    // The image must be extended :
+    // if (z<0)    TA(cluster_id,l,z) == TA(cluster_id,l,0)
+    // if (z>NP-1) TA(cluster_id,l,z) == TA(cluster_id,l,NP-1)
+    ////////////////////////////////////////////////////////
+
+    date  = giet_proctime();
+    H_BEG[cluster_id][lpid] = date;
+
+#if VERBOSE 
+printf( "\n[CONVOL] thread[%d,%d,%d] starts horizontal filter"
+        " at cycle %d\n",
+        x, y, lpid, date );
+#else
+if ( (x==0) && (y==0) && (lpid==0) ) 
+printf( "\n[CONVOL] thread[0,0,0] starts horizontal filter"
+        " at cycle %d\n", date );
+#endif
+
+    // l = absolute line index / p = absolute pixel index  
+    // first & last define which lines are handled by a given thread
+
+    first = thread_id * lines_per_thread;
+    last  = first + lines_per_thread;
+
+    for (l = first; l < last; l++)
+    {
+        // src_c and src_l are the cluster index and the line index for A & D
+        int src_c = l / lines_per_cluster;
+        int src_l = l % lines_per_cluster;
+
+        // We use the specific values of the horizontal ep-filter for optimisation:
+        // sum(p) = sum(p-1) + TA[p+hrange] - TA[p-hrange-1]
+        // To minimize the number of tests, the loop on pixels is split in three domains 
+
+        int sum_p = (hrange + 2) * TA(src_c, src_l, 0);
+        for (z = 1; z < hrange; z++)
+        {
+            sum_p = sum_p + TA(src_c, src_l, z);
+        }
+
+        // first domain : from 0 to hrange
+        for (p = 0; p < hrange + 1; p++)
+        {
+            // dst_c and dst_p are the cluster index and the pixel index for B
+            int dst_c = p / pixels_per_cluster;
+            int dst_p = p % pixels_per_cluster;
+            sum_p = sum_p + (int) TA(src_c, src_l, p + hrange) - (int) TA(src_c, src_l, 0);
+            TB(dst_c, dst_p, l) = sum_p / hnorm;
+            TD(src_c, src_l, p) = (int) TA(src_c, src_l, p) - sum_p / hnorm;
+        }
+        // second domain : from (hrange+1) to (NP-hrange-1)
+        for (p = hrange + 1; p < NP - hrange; p++)
+        {
+            // dst_c and dst_p are the cluster index and the pixel index for B
+            int dst_c = p / pixels_per_cluster;
+            int dst_p = p % pixels_per_cluster;
+            sum_p = sum_p + (int) TA(src_c, src_l, p + hrange) 
+                          - (int) TA(src_c, src_l, p - hrange - 1);
+            TB(dst_c, dst_p, l) = sum_p / hnorm;
+            TD(src_c, src_l, p) = (int) TA(src_c, src_l, p) - sum_p / hnorm;
+        }
+        // third domain : from (NP-hrange) to (NP-1)
+        for (p = NP - hrange; p < NP; p++)
+        {
+            // dst_c and dst_p are the cluster index and the pixel index for B
+            int dst_c = p / pixels_per_cluster;
+            int dst_p = p % pixels_per_cluster;
+            sum_p = sum_p + (int) TA(src_c, src_l, NP - 1) 
+                          - (int) TA(src_c, src_l, p - hrange - 1);
+            TB(dst_c, dst_p, l) = sum_p / hnorm;
+            TD(src_c, src_l, p) = (int) TA(src_c, src_l, p) - sum_p / hnorm;
+        }
+
+#if SUPER_VERBOSE
+printf(" - line %d computed at cycle %d\n", l, giet_proctime() );
+#endif    
+
+    }
+
+    date  = giet_proctime();
+    H_END[cluster_id][lpid] = date;
+
+#if VERBOSE 
+printf( "\n[CONVOL] thread[%d,%d,%d] completes horizontal filter"
+        " at cycle %d\n",
+        x, y, lpid, date );
+#else
+if ( (x==0) && (y==0) && (lpid==0) ) 
+printf( "\n[CONVOL] thread[0,0,0] completes horizontal filter"
+        " at cycle %d\n", date );
+#endif
+
+    /////////////////////////////
+    #if USE_SQT_BARRIER
+    sqt_barrier_wait( &barrier );
+    #else
+    barrier_wait( &barrier );
+    #endif
+
+
+    ///////////////////////////////////////////////////////////////
+    // parallel vertical filter : 
+    // C <= transpose(FV(B))
+    // Each thread computes (NP/nthreads) columns
+    // The image must be extended :
+    // if (l<0)    TB(cluster_id,p,l) == TB(cluster_id,p,0)
+    // if (l>NL-1)   TB(cluster_id,p,l) == TB(cluster_id,p,NL-1)
+    ///////////////////////////////////////////////////////////////
+
+    date  = giet_proctime();
+    V_BEG[cluster_id][lpid] = date;
+
+#if VERBOSE 
+printf( "\n[CONVOL] thread[%d,%d,%d] starts vertical filter"
+        " at cycle %d\n",
+        x, y, lpid, date );
+#else
+if ( (x==0) && (y==0) && (lpid==0) ) 
+printf( "\n[CONVOL] thread[0,0,0] starts vertical filter"
+        " at cycle %d\n", date );
+#endif
+
+    // l = absolute line index / p = absolute pixel index
+    // first & last define which pixels are handled by a given thread
+
+    first = thread_id * pixels_per_thread;
+    last  = first + pixels_per_thread;
+
+    for (p = first; p < last; p++)
+    {
+        // src_c and src_p are the cluster index and the pixel index for B
+        int src_c = p / pixels_per_cluster;
+        int src_p = p % pixels_per_cluster;
+
+        int sum_l;
+
+        // We use the specific values of the vertical ep-filter
+        // To minimize the number of tests, the NL lines are split in three domains 
+
+        // first domain : explicit computation for the first 18 values
+        for (l = 0; l < 18; l++)
+        {
+            // dst_c and dst_l are the cluster index and the line index for C
+            int dst_c = l / lines_per_cluster;
+            int dst_l = l % lines_per_cluster;
+
+            for (z = 0, sum_l = 0; z < 35; z++)
+            {
+                sum_l = sum_l + vf[z] * TB(src_c, src_p, max(l - 17 + z,0) );
+            }
+            TC(dst_c, dst_l, p) = sum_l / vnorm;
+        }
+        // second domain
+        for (l = 18; l < NL - 17; l++)
+        {
+            // dst_c and dst_l are the cluster index and the line index for C
+            int dst_c = l / lines_per_cluster;
+            int dst_l = l % lines_per_cluster;
+
+            sum_l = sum_l + TB(src_c, src_p, l + 4)
+                  + TB(src_c, src_p, l + 8)
+                  + TB(src_c, src_p, l + 11)
+                  + TB(src_c, src_p, l + 15)
+                  + TB(src_c, src_p, l + 17)
+                  - TB(src_c, src_p, l - 5)
+                  - TB(src_c, src_p, l - 9)
+                  - TB(src_c, src_p, l - 12)
+                  - TB(src_c, src_p, l - 16)
+                  - TB(src_c, src_p, l - 18);
+
+            TC(dst_c, dst_l, p) = sum_l / vnorm;
+        }
+        // third domain
+        for (l = NL - 17; l < NL; l++)
+        {
+            // dst_c and dst_l are the cluster index and the line index for C
+            int dst_c = l / lines_per_cluster;
+            int dst_l = l % lines_per_cluster;
+
+            sum_l = sum_l + TB(src_c, src_p, min(l + 4, NL - 1))
+                  + TB(src_c, src_p, min(l + 8, NL - 1))
+                  + TB(src_c, src_p, min(l + 11, NL - 1))
+                  + TB(src_c, src_p, min(l + 15, NL - 1))
+                  + TB(src_c, src_p, min(l + 17, NL - 1))
+                  - TB(src_c, src_p, l - 5)
+                  - TB(src_c, src_p, l - 9)
+                  - TB(src_c, src_p, l - 12)
+                  - TB(src_c, src_p, l - 16)
+                  - TB(src_c, src_p, l - 18);
+
+            TC(dst_c, dst_l, p) = sum_l / vnorm;
+        }
+
+#if SUPER_VERBOSE
+printf(" - column %d computed at cycle %d\n", p, giet_proctime());
+#endif 
+
+    }
+
+    date  = giet_proctime();
+    V_END[cluster_id][lpid] = date;
+
+#if VERBOSE 
+printf( "\n[CONVOL] thread[%d,%d,%d] completes vertical filter"
+        " at cycle %d\n",
+        x, y, lpid, date );
+#else
+if ( (x==0) && (y==0) && (lpid==0) ) 
+printf( "\n[CONVOL] thread[0,0,0] completes vertical filter"
+        " at cycle %d\n", date );
+#endif
+
+    ////////////////////////////
+    #if USE_SQT_BARRIER
+    sqt_barrier_wait( &barrier );
+    #else
+    barrier_wait( &barrier );
+    #endif
+
+    ////////////////////////////////////////////////////////////////////////
+    // Optional parallel display of the final image Z <= D + C
+    // Eah thread[x,y,p] displays (NL/nthreads) lines. (one byte per pixel).
+    ////////////////////////////////////////////////////////////////////////
+
+    if ( FINAL_DISPLAY_ENABLE )
+    {
+        date  = giet_proctime();
+        D_BEG[cluster_id][lpid] = date;
+
+#if VERBOSE
+printf( "\n[CONVOL] thread[%d,%d,%d] starts final display"
+        " at cycle %d\n",
+        x, y, lpid, date);
+#else
+if ( (x==0) && (y==0) && (lpid==0) ) 
+printf( "\n[CONVOL] thread[0,0,0] starts final display"
+        " at cycle %d\n", date );
+#endif
+
+        unsigned int line;
+        unsigned int offset = lines_per_thread * lpid;
+
+        for ( l = 0 ; l < lines_per_thread ; l++ )
+        {
+            line = offset + l;
+
+            for ( p = 0 ; p < NP ; p++ )
+            {
+                TZ(cluster_id, line, p) = 
+                   (unsigned char)( (TD(cluster_id, line, p) + 
+                                     TC(cluster_id, line, p) ) >> 8 );
+            }
+
+            giet_fbf_sync_write( NP*(l + (thread_id * lines_per_thread) ), 
+                                 &TZ(cluster_id, line, 0), 
+                                 NP);
+        }
+
+        date  = giet_proctime();
+        D_END[cluster_id][lpid] = date;
+
+#if VERBOSE
+printf( "\n[CONVOL] thread[%d,%d,%d] completes final display"
+        " at cycle %d\n",
+        x, y, lpid, date);
+#else
+if ( (x==0) && (y==0) && (lpid==0) ) 
+printf( "\n[CONVOL] thread[0,0,0] completes final display"
+        " at cycle %d\n", date );
+#endif
+     
+    //////////////////////////////
+    #if USE_SQT_BARRIER
+    sqt_barrier_wait( &barrier );
+    #else
+    barrier_wait( &barrier );
+    #endif
+
+    }
+
+    // all threads (but the one executing main) exit
+    if ( (x!=0) || (y!=0) || (lpid!=0) )
+    {
+        giet_pthread_exit( "completed");
+    }
+
+} // end execute()
+
+
+
+/////////////////////////////////////////
+void instrument( unsigned int nclusters,
+                 unsigned int nprocs )
+/////////////////////////////////////////
+{
+        int cc, pp;
+
+        unsigned int min_start = 0xFFFFFFFF;
+        unsigned int max_start = 0;
+
+        unsigned int min_h_beg = 0xFFFFFFFF;
+        unsigned int max_h_beg = 0;
+
+        unsigned int min_h_end = 0xFFFFFFFF;
+        unsigned int max_h_end = 0;
+
+        unsigned int min_v_beg = 0xFFFFFFFF;
+        unsigned int max_v_beg = 0;
+
+        unsigned int min_v_end = 0xFFFFFFFF;
+        unsigned int max_v_end = 0;
+
+        unsigned int min_d_beg = 0xFFFFFFFF;
+        unsigned int max_d_beg = 0;
+
+        unsigned int min_d_end = 0xFFFFFFFF;
+        unsigned int max_d_end = 0;
+
+        for (cc = 0; cc < nclusters; cc++)
+        {
+            for (pp = 0; pp < nprocs; pp++ )
+            {
+                if (START[cc][pp] < min_start) min_start = START[cc][pp];
+                if (START[cc][pp] > max_start) max_start = START[cc][pp];
+
+                if (H_BEG[cc][pp] < min_h_beg) min_h_beg = H_BEG[cc][pp];
+                if (H_BEG[cc][pp] > max_h_beg) max_h_beg = H_BEG[cc][pp];
+
+                if (H_END[cc][pp] < min_h_end) min_h_end = H_END[cc][pp];
+                if (H_END[cc][pp] > max_h_end) max_h_end = H_END[cc][pp];
+
+                if (V_BEG[cc][pp] < min_v_beg) min_v_beg = V_BEG[cc][pp];
+                if (V_BEG[cc][pp] > max_v_beg) max_v_beg = V_BEG[cc][pp];
+
+                if (V_END[cc][pp] < min_v_end) min_v_end = V_END[cc][pp];
+                if (V_END[cc][pp] > max_v_end) max_v_end = V_END[cc][pp];
+
+                if (D_BEG[cc][pp] < min_d_beg) min_d_beg = D_BEG[cc][pp];
+                if (D_BEG[cc][pp] > max_d_beg) max_d_beg = D_BEG[cc][pp];
+
+                if (D_END[cc][pp] < min_d_end) min_d_end = D_END[cc][pp];
+                if (D_END[cc][pp] > max_d_end) max_d_end = D_END[cc][pp];
+            }
+        }
+
+        printf(" - START : min = %d / max = %d / med = %d / delta = %d\n",
+               min_start, max_start, (min_start+max_start)/2, max_start-min_start);
+
+        printf(" - H_BEG : min = %d / max = %d / med = %d / delta = %d\n",
+               min_h_beg, max_h_beg, (min_h_beg+max_h_beg)/2, max_h_beg-min_h_beg);
+
+        printf(" - H_END : min = %d / max = %d / med = %d / delta = %d\n",
+               min_h_end, max_h_end, (min_h_end+max_h_end)/2, max_h_end-min_h_end);
+
+        printf(" - V_BEG : min = %d / max = %d / med = %d / delta = %d\n",
+               min_v_beg, max_v_beg, (min_v_beg+max_v_beg)/2, max_v_beg-min_v_beg);
+
+        printf(" - V_END : min = %d / max = %d / med = %d / delta = %d\n",
+               min_v_end, max_v_end, (min_v_end+max_v_end)/2, max_v_end-min_v_end);
+
+        printf(" - D_BEG : min = %d / max = %d / med = %d / delta = %d\n",
+               min_d_beg, max_d_beg, (min_d_beg+max_d_beg)/2, max_d_beg-min_d_beg);
+
+        printf(" - D_END : min = %d / max = %d / med = %d / delta = %d\n",
+               min_d_end, max_d_end, (min_d_end+max_d_end)/2, max_d_end-min_d_end);
+
+        printf( "\n General Scenario (Kcycles for each step)\n" );
+        printf( " - BOOT OS           = %d\n", (min_start            )/1000 );
+        printf( " - LOAD IMAGE        = %d\n", (min_h_beg - min_start)/1000 );
+        printf( " - H_FILTER          = %d\n", (max_h_end - min_h_beg)/1000 );
+        printf( " - BARRIER HORI/VERT = %d\n", (min_v_beg - max_h_end)/1000 );
+        printf( " - V_FILTER          = %d\n", (max_v_end - min_v_beg)/1000 );
+        printf( " - BARRIER VERT/DISP = %d\n", (min_d_beg - max_v_end)/1000 );
+        printf( " - DISPLAY           = %d\n", (max_d_end - min_d_beg)/1000 );
+
+} // end instrument()
+
+
+
+///////////////////////////////////////////
+__attribute__ ((constructor)) void main()
+///////////////////////////////////////////
+{
+    // get plat-form parameters
+    unsigned int x_size;                 // number of clusters in a row
+    unsigned int y_size;                 // number of clusters in a column
+    unsigned int nprocs;                 // number of processors per cluster
+    giet_procs_number( &x_size , &y_size , &nprocs );
+
+    // processor identifiers
+    unsigned int x;                      // x coordinate
+    unsigned int y;                      // y coordinate
+    unsigned int lpid;                   // local proc index
+    giet_proc_xyp( &x, &y, &lpid );
+
+    // indexes for loops
+    unsigned int cx;
+    unsigned int cy;
+    unsigned int n;
+
+    unsigned int nclusters  = x_size * y_size;          
+    unsigned int nthreads   = nclusters * nprocs;     
+
+    // parameters checking 
+    if ((nprocs != 1) && (nprocs != 2) && (nprocs != 4) && (nprocs != 8))
+        giet_pthread_exit( "[CONVOL ERROR] NB_PROCS_MAX must be 1, 2, 4 or 8\n");
+
+    if ((x_size!=1) && (x_size!=2) && (x_size!=4) && (x_size!=8) && (x_size!=16))
+        giet_pthread_exit( "[CONVOL ERROR] X_SIZE must be 1, 2, 4, 8, 16\n");
+        
+    if ((y_size!=1) && (y_size!=2) && (y_size!=4) && (y_size!=8) && (y_size!=16))
+        giet_pthread_exit( "[CONVOL ERROR] Y_SIZE must be 1, 2, 4, 8, 16\n");
+
+    if ( NL % nclusters != 0 )
+        giet_pthread_exit( "[CONVOL ERROR] X_SIZE*Y_SIZE must be a divider of NL");
+
+    if ( NP % nclusters != 0 )
+        giet_pthread_exit( "[CONVOL ERROR] X_SIZE*Y_SIZE must be a divider of NP");
+
+    // get a shared TTY
+    giet_tty_alloc( 1 );
+    lock_init( &tty_lock );
+
+    // initializes the distributed heap[x,y]
+    for ( cx = 0 ; cx < x_size ; cx++ )
+    {
+        for ( cy = 0 ; cy < y_size ; cy++ )
+        {
+            heap_init( cx , cy );
+        }
+    }
+
+    // allocate trdid[] array
+    pthread_t* trdid = malloc( nthreads * sizeof(pthread_t) );
+
+    // barrier initialisation
+#if USE_SQT_BARRIER
+    sqt_barrier_init( &barrier, x_size , y_size , nprocs );
+#else
+    barrier_init( &barrier, nthreads );
+#endif
+
+    printf("\n[CONVOL] thread[0,0,0] completes initialisation at cycle %d\n" 
+           "- CLUSTERS     = %d\n"
+           "- PROCS        = %d\n" 
+           "- THREADS      = %d\n",
+           giet_proctime(), nclusters, nprocs, nthreads );
+
+    // launch other threads to run execute() function
+    for ( n = 1 ; n < nthreads ; n++ )
+    {
+        if ( giet_pthread_create( &trdid[n],
+                                  NULL,                  // no attribute
+                                  &execute,
+                                  NULL ) )               // no argument
+        {
+            printf("\n[TRANSPOSE ERROR] creating thread %x\n", trdid[n] );
+            giet_pthread_exit( NULL );
+        }
+    }
+
+    // run the execute() function
+    execute();
+
+    // wait other threads completion
+    for ( n = 1 ; n < nthreads ; n++ )
+    {
+        if ( giet_pthread_join( trdid[n], NULL ) )
+        {
+            printf("\n[TRANSPOSE ERROR] joining thread %x\n", trdid[n] );
+            giet_pthread_exit( NULL );
+        }
+        else
+        {
+            printf("\n[TRANSPOSE] thread %x joined at cycle %d\n",
+                   trdid[n] , giet_proctime() );
+        }
+    }
+
+    // call the instrument() function
+    instrument( nclusters , nprocs );
+
+    giet_pthread_exit( "completed" );
+    
+} // end main() 
+
+
+
+// Local Variables:
+// tab-width: 3
+// c-basic-offset: 3
+// c-file-offsets:((innamespace . 0)(inline-open . 0))
+// indent-tabs-mode: nil
+// End:
+
+// vim: filetype=cpp:expandtab:shiftwidth=3:tabstop=3:softtabstop=3
+
+
Index: soft/giet_vm/applications/convol/convol.py
===================================================================
--- soft/giet_vm/applications/convol/convol.py	(revision 707)
+++ soft/giet_vm/applications/convol/convol.py	(revision 708)
@@ -11,6 +11,6 @@
 #  application on a multi-clusters, multi-processors architecture.
 #  This include both the mapping of virtual segments on the clusters,
-#  and the mapping of tasks on processors.
-#  There is one task per processor.
+#  and the mapping of threads on processors.
+#  There is one thread per processor.
 #  The mapping of virtual segments is the following:
 #    - There is one shared data vseg in cluster[0][0]
@@ -85,5 +85,5 @@
                                      local = True, big = True )
             
-    # heap vsegs : distributed but non local (any heap can be accessed by any task)
+    # heap vsegs : distributed but non local (any heap can be accessed by any thread)
     for x in xrange (x_size):
         for y in xrange (y_size):
@@ -97,5 +97,5 @@
                                  local = False, big = True )
 
-    # distributed tasks : one task per processor 
+    # distributed threads : one thread per processor 
     for x in xrange (x_size):
         for y in xrange (y_size):
@@ -103,10 +103,18 @@
             if ( mapping.clusters[cluster_id].procs ):
                 for p in xrange( nprocs ):
-                    trdid = (((x * y_size) + y) * nprocs) + p
+                    if (x == 0) and (y == 0) and (p == 0) :   # main thread
+                        startid = 1
+                        is_main = True
+                    else :                                    # trsp thread
+                        startid = 0
+                        is_main = False
 
-                    mapping.addTask( vspace, 'conv_%d_%d_%d' % (x,y,p),
-                                     trdid, x, y, p,
-                                     'conv_stack_%d_%d_%d' % (x,y,p),
-                                     'conv_heap_%d_%d' % (x,y), 0 )
+                    mapping.addThread( vspace,
+                                       'conv_%d_%d_%d' % (x,y,p),
+                                       is_main,
+                                       x, y, p,
+                                       'conv_stack_%d_%d_%d' % (x,y,p),
+                                       'conv_heap_%d_%d' % (x,y),
+                                       startid )
 
     # extend mapping name
Index: soft/giet_vm/applications/convol/main.c
===================================================================
--- soft/giet_vm/applications/convol/main.c	(revision 707)
+++ 	(revision )
@@ -1,749 +1,0 @@
-///////////////////////////////////////////////////////////////////////////////////////
-// File   : main.c   (convol application)
-// Date   : june 2014
-// author : Alain Greiner
-///////////////////////////////////////////////////////////////////////////////////////
-// This multi-threaded application application implements a 2D convolution product.  
-// The convolution kernel is [201]*[35] pixels, but it can be factored in two
-// independant line and column convolution products.
-// It can run on a multi-processors, multi-clusters architecture, with one thread
-// per processor. 
-// 
-// The (1024 * 1024) pixels image is read from a file (2 bytes per pixel).
-//
-// - number of clusters containing processors must be power of 2 no larger than 256.
-// - number of processors per cluster must be power of 2 no larger than 8.
-///////////////////////////////////////////////////////////////////////////////////////
-
-#include "stdio.h"
-#include "stdlib.h"
-#include "user_barrier.h"
-#include "malloc.h"
-
-#define USE_SQT_BARRIER            1
-#define VERBOSE                    1
-#define SUPER_VERBOSE              0
-
-#define X_SIZE_MAX                 16
-#define Y_SIZE_MAX                 16
-#define PROCS_MAX                  8
-#define CLUSTERS_MAX               (X_SIZE_MAX * Y_SIZE_MAX)
-
-#define INITIAL_DISPLAY_ENABLE     0
-#define FINAL_DISPLAY_ENABLE       1
-
-#define PIXEL_SIZE                 2
-#define NL                         1024
-#define NP                         1024
-#define NB_PIXELS                  (NP * NL)
-#define FRAME_SIZE                 (NB_PIXELS * PIXEL_SIZE)
-
-#define SEEK_SET                   0
-
-#define TA(c,l,p)  (A[c][((NP) * (l)) + (p)])
-#define TB(c,p,l)  (B[c][((NL) * (p)) + (l)])
-#define TC(c,l,p)  (C[c][((NP) * (l)) + (p)])
-#define TD(c,l,p)  (D[c][((NP) * (l)) + (p)])
-#define TZ(c,l,p)  (Z[c][((NP) * (l)) + (p)])
-
-#define max(x,y) ((x) > (y) ? (x) : (y))
-#define min(x,y) ((x) < (y) ? (x) : (y))
-
-// macro to use a shared TTY
-#define printf(...)     lock_acquire( &tty_lock ); \
-                        giet_tty_printf(__VA_ARGS__);  \
-                        lock_release( &tty_lock )
-
-// global instrumentation counters (cluster_id, lpid]
-
-unsigned int START[CLUSTERS_MAX][PROCS_MAX];
-unsigned int H_BEG[CLUSTERS_MAX][PROCS_MAX];
-unsigned int H_END[CLUSTERS_MAX][PROCS_MAX];
-unsigned int V_BEG[CLUSTERS_MAX][PROCS_MAX];
-unsigned int V_END[CLUSTERS_MAX][PROCS_MAX];
-unsigned int D_BEG[CLUSTERS_MAX][PROCS_MAX];
-unsigned int D_END[CLUSTERS_MAX][PROCS_MAX];
-
-// global synchronization barriers
-
-#if USE_SQT_BARRIER
-giet_sqt_barrier_t  barrier;
-#else
-giet_barrier_t      barrier;
-#endif
-
-volatile unsigned int barrier_init_ok    = 0;
-volatile unsigned int load_image_ok      = 0;
-volatile unsigned int instrumentation_ok = 0;
-
-// lock protecting access to shared TTY
-user_lock_t         tty_lock;
-
-// global pointers on distributed buffers in all clusters
-unsigned short * GA[CLUSTERS_MAX];
-int *            GB[CLUSTERS_MAX];
-int *            GC[CLUSTERS_MAX];
-int *            GD[CLUSTERS_MAX];
-unsigned char *  GZ[CLUSTERS_MAX];
-
-///////////////////////////////////////////
-__attribute__ ((constructor)) void main()
-///////////////////////////////////////////
-{
-    //////////////////////////////////
-    // convolution kernel parameters
-    // The content of this section is
-    // Philips proprietary information.
-    ///////////////////////////////////
-
-    int   vnorm  = 115;
-    int   vf[35] = { 1, 1, 2, 2, 2,
-                     2, 3, 3, 3, 4,
-                     4, 4, 4, 5, 5,
-                     5, 5, 5, 5, 5,
-                     5, 5, 4, 4, 4,
-                     4, 3, 3, 3, 2,
-                     2, 2, 2, 1, 1 };
-
-    int hrange = 100;
-    int hnorm  = 201;
-
-    unsigned int date = 0;
-
-    int c; // cluster index for loops
-    int l; // line index for loops
-    int p; // pixel index for loops
-    int z; // vertical filter index for loops
-
-    // plat-form parameters
-    unsigned int x_size;             // number of clusters in a row
-    unsigned int y_size;             // number of clusters in a column
-    unsigned int nprocs;             // number of processors per cluster
-    
-    giet_procs_number( &x_size , &y_size , &nprocs );
-
-    // processor identifiers
-    unsigned int x;                                         // x coordinate
-    unsigned int y;                                         // y coordinate
-    unsigned int lpid;                                      // local proc/task id
-    giet_proc_xyp( &x, &y, &lpid );
-
-    int          file       = 0;                            // file descriptor
-    unsigned int nclusters  = x_size * y_size;              // number of clusters
-    unsigned int cluster_id = (x * y_size) + y;             // continuous cluster index
-    unsigned int task_id    = (cluster_id * nprocs) + lpid; // continuous task index
-    unsigned int ntasks     = nclusters * nprocs;           // number of tasks
-    unsigned int frame_size = FRAME_SIZE;                   // total size (bytes)
-
-    unsigned int lines_per_task     = NL / ntasks;          // lines per task
-    unsigned int lines_per_cluster  = NL / nclusters;       // lines per cluster
-    unsigned int pixels_per_task    = NP / ntasks;          // columns per task
-    unsigned int pixels_per_cluster = NP / nclusters;       // columns per cluster
-
-    int first, last;
-
-    date = giet_proctime();
-    START[cluster_id][lpid] = date;
-
-     // parameters checking 
-   
-    if ((nprocs != 1) && (nprocs != 2) && (nprocs != 4) && (nprocs != 8))
-        giet_exit( "[CONVOL ERROR] NB_PROCS_MAX must be 1, 2, 4 or 8\n");
-
-    if ((x_size!=1) && (x_size!=2) && (x_size!=4) && (x_size!=8) && (x_size!=16))
-        giet_exit( "[CONVOL ERROR] x_size must be 1, 2, 4, 8, 16\n");
-        
-    if ((y_size!=1) && (y_size!=2) && (y_size!=4) && (y_size!=8) && (y_size!=16))
-        giet_exit( "[CONVOL ERROR] y_size must be 1, 2, 4, 8, 16\n");
-
-    if ( NL % nclusters != 0 )
-        giet_exit( "[CONVOL ERROR] CLUSTERS_MAX must be a divider of NL");
-
-    if ( NP % nclusters != 0 )
-        giet_exit( "[CONVOL ERROR] CLUSTERS_MAX must be a divider of NP");
-
-    
-    ///////////////////////////////////////////////////////////////////
-    // task[0][0][0] makes various initialisations
-    ///////////////////////////////////////////////////////////////////
-    
-    if ( (x==0) && (y==0) && (lpid==0) )
-    {
-        // get a shared TTY
-        giet_tty_alloc( 1 );
-
-        // initializes TTY lock
-        lock_init( &tty_lock );
-
-        // initializes the distributed heap[x,y]
-        unsigned int cx;
-        unsigned int cy;
-        for ( cx = 0 ; cx < x_size ; cx++ )
-        {
-            for ( cy = 0 ; cy < y_size ; cy++ )
-            {
-                heap_init( cx , cy );
-            }
-        }
-
-#if USE_SQT_BARRIER
-        sqt_barrier_init( &barrier, x_size , y_size , nprocs );
-#else
-        barrier_init( &barrier, ntasks );
-#endif
-
-        printf("\n[CONVOL] task[0,0,0] completes initialisation at cycle %d\n" 
-               "- CLUSTERS   = %d\n"
-               "- PROCS      = %d\n" 
-               "- TASKS      = %d\n" 
-               "- LINES/TASK = %d\n",
-               giet_proctime(), nclusters, nprocs, ntasks, lines_per_task );
-
-        barrier_init_ok = 1;
-    }
-    else
-    {
-        while ( barrier_init_ok == 0 );
-    }
-
-    ///////////////////////////////////////////////////////////////////
-    // All task[x][y][0] allocate the global buffers in cluster(x,y)
-    // These buffers mut be sector-aligned.
-    ///////////////////////////////////////////////////////////////////
-    if ( lpid == 0 )
-    {
-
-#if VERBOSE
-printf( "\n[CONVOL] task[%d,%d,%d] enters malloc at cycle %d\n", 
-                 x,y,lpid, date );
-#endif
-
-        GA[cluster_id] = remote_malloc( (FRAME_SIZE/nclusters)   , x , y );
-        GB[cluster_id] = remote_malloc( (FRAME_SIZE/nclusters)*2 , x , y );
-        GC[cluster_id] = remote_malloc( (FRAME_SIZE/nclusters)*2 , x , y );
-        GD[cluster_id] = remote_malloc( (FRAME_SIZE/nclusters)*2 , x , y );
-        GZ[cluster_id] = remote_malloc( (FRAME_SIZE/nclusters)/2 , x , y );
-        
-#if VERBOSE
-printf( "\n[CONVOL]  Shared Buffer Virtual Addresses in cluster(%d,%d)\n"
-        "### GA = %x\n"
-        "### GB = %x\n"               
-        "### GC = %x\n"               
-        "### GD = %x\n"               
-        "### GZ = %x\n",
-        x, y,
-        GA[cluster_id],
-        GB[cluster_id],
-        GC[cluster_id],
-        GD[cluster_id],
-        GZ[cluster_id] );
-#endif
-    }
-
-    ///////////////////////////////
-    #if USE_SQT_BARRIER
-    sqt_barrier_wait( &barrier );
-    #else
-    barrier_wait( &barrier );
-    #endif
-
-    ///////////////////////////////////////////////////////////////////
-    // All tasks initialise in their private stack a copy of the
-    // arrays of pointers on the shared, distributed buffers.
-    ///////////////////////////////////////////////////////////////////
-
-    unsigned short * A[CLUSTERS_MAX];
-    int            * B[CLUSTERS_MAX];
-    int            * C[CLUSTERS_MAX];
-    int            * D[CLUSTERS_MAX];
-    unsigned char  * Z[CLUSTERS_MAX];
-
-    for (c = 0; c < nclusters; c++)
-    {
-        A[c] = GA[c];
-        B[c] = GB[c];
-        C[c] = GC[c];
-        D[c] = GD[c];
-        Z[c] = GZ[c];
-    }
-
-    ///////////////////////////////////////////////////////////////////////////
-    // task[0,0,0] open the file containing image, and load it from disk 
-    // to all A[c] buffers (frame_size / nclusters loaded in each cluster).
-    // Other tasks are waiting on the init_ok condition.
-    //////////////////////////////////////////////////////////////////////////
-    if ( (x==0) && (y==0) && (lpid==0) )
-    {
-        // open file
-        file = giet_fat_open( "/misc/philips_1024.raw" , 0 );
-        if ( file < 0 ) giet_exit( "[CONVOL ERROR] task[0,0,0] cannot open"
-                                   " file /misc/philips_1024.raw" );
- 
-        printf( "\n[CONVOL] task[0,0,0] open file /misc/philips_1024.raw"
-                " at cycle %d\n", giet_proctime() );
-
-        for ( c = 0 ; c < nclusters ; c++ )
-        {
-            printf( "\n[CONVOL] task[0,0,0] starts load "
-                    "for cluster %d at cycle %d\n", c, giet_proctime() );
-
-            giet_fat_lseek( file,
-                            (frame_size/nclusters)*c,
-                            SEEK_SET );
-
-            giet_fat_read( file,
-                           A[c],
-                           frame_size/nclusters );
-
-            printf( "\n[CONVOL] task[0,0,0] completes load "
-                    "for cluster %d at cycle %d\n", c, giet_proctime() );
-        }
-        load_image_ok = 1;
-    }
-    else
-    {
-        while ( load_image_ok == 0 );
-    }
-
-    /////////////////////////////////////////////////////////////////////////////
-    // Optionnal parallel display of the initial image stored in A[c] buffers.
-    // Eah task displays (NL/ntasks) lines. (one byte per pixel).
-    /////////////////////////////////////////////////////////////////////////////
-
-    if ( INITIAL_DISPLAY_ENABLE )
-    {
-
-#if VERBOSE
-printf( "\n[CONVOL] task[%d,%d,%d] starts initial display"
-        " at cycle %d\n",
-        x, y, lpid, giet_proctime() );
-#endif
-
-        unsigned int line;
-        unsigned int offset = lines_per_task * lpid;
-
-        for ( l = 0 ; l < lines_per_task ; l++ )
-        {
-            line = offset + l;
-
-            for ( p = 0 ; p < NP ; p++ )
-            {
-                TZ(cluster_id, line, p) = (unsigned char)(TA(cluster_id, line, p) >> 8);
-            }
-
-            giet_fbf_sync_write( NP*(l + (task_id * lines_per_task) ), 
-                                 &TZ(cluster_id, line, 0), 
-                                 NP);
-        }
-
-#if VERBOSE 
-printf( "\n[CONVOL] task[%d,%d,%d] completes initial display"
-        " at cycle %d\n",
-        x, y, lpid, giet_proctime() );
-#endif
-
-        ////////////////////////////
-        #if USE_SQT_BARRIER
-        sqt_barrier_wait( &barrier );
-        #else
-        barrier_wait( &barrier );
-        #endif
-
-    }
-
-    ////////////////////////////////////////////////////////
-    // parallel horizontal filter : 
-    // B <= transpose(FH(A))
-    // D <= A - FH(A)
-    // Each task computes (NL/ntasks) lines 
-    // The image must be extended :
-    // if (z<0)    TA(cluster_id,l,z) == TA(cluster_id,l,0)
-    // if (z>NP-1) TA(cluster_id,l,z) == TA(cluster_id,l,NP-1)
-    ////////////////////////////////////////////////////////
-
-    date  = giet_proctime();
-    H_BEG[cluster_id][lpid] = date;
-
-#if VERBOSE 
-printf( "\n[CONVOL] task[%d,%d,%d] starts horizontal filter"
-        " at cycle %d\n",
-        x, y, lpid, date );
-#else
-if ( (x==0) && (y==0) && (lpid==0) ) 
-printf( "\n[CONVOL] task[0,0,0] starts horizontal filter"
-        " at cycle %d\n", date );
-#endif
-
-    // l = absolute line index / p = absolute pixel index  
-    // first & last define which lines are handled by a given task
-
-    first = task_id * lines_per_task;
-    last  = first + lines_per_task;
-
-    for (l = first; l < last; l++)
-    {
-        // src_c and src_l are the cluster index and the line index for A & D
-        int src_c = l / lines_per_cluster;
-        int src_l = l % lines_per_cluster;
-
-        // We use the specific values of the horizontal ep-filter for optimisation:
-        // sum(p) = sum(p-1) + TA[p+hrange] - TA[p-hrange-1]
-        // To minimize the number of tests, the loop on pixels is split in three domains 
-
-        int sum_p = (hrange + 2) * TA(src_c, src_l, 0);
-        for (z = 1; z < hrange; z++)
-        {
-            sum_p = sum_p + TA(src_c, src_l, z);
-        }
-
-        // first domain : from 0 to hrange
-        for (p = 0; p < hrange + 1; p++)
-        {
-            // dst_c and dst_p are the cluster index and the pixel index for B
-            int dst_c = p / pixels_per_cluster;
-            int dst_p = p % pixels_per_cluster;
-            sum_p = sum_p + (int) TA(src_c, src_l, p + hrange) - (int) TA(src_c, src_l, 0);
-            TB(dst_c, dst_p, l) = sum_p / hnorm;
-            TD(src_c, src_l, p) = (int) TA(src_c, src_l, p) - sum_p / hnorm;
-        }
-        // second domain : from (hrange+1) to (NP-hrange-1)
-        for (p = hrange + 1; p < NP - hrange; p++)
-        {
-            // dst_c and dst_p are the cluster index and the pixel index for B
-            int dst_c = p / pixels_per_cluster;
-            int dst_p = p % pixels_per_cluster;
-            sum_p = sum_p + (int) TA(src_c, src_l, p + hrange) 
-                          - (int) TA(src_c, src_l, p - hrange - 1);
-            TB(dst_c, dst_p, l) = sum_p / hnorm;
-            TD(src_c, src_l, p) = (int) TA(src_c, src_l, p) - sum_p / hnorm;
-        }
-        // third domain : from (NP-hrange) to (NP-1)
-        for (p = NP - hrange; p < NP; p++)
-        {
-            // dst_c and dst_p are the cluster index and the pixel index for B
-            int dst_c = p / pixels_per_cluster;
-            int dst_p = p % pixels_per_cluster;
-            sum_p = sum_p + (int) TA(src_c, src_l, NP - 1) 
-                          - (int) TA(src_c, src_l, p - hrange - 1);
-            TB(dst_c, dst_p, l) = sum_p / hnorm;
-            TD(src_c, src_l, p) = (int) TA(src_c, src_l, p) - sum_p / hnorm;
-        }
-
-#if SUPER_VERBOSE
-printf(" - line %d computed at cycle %d\n", l, giet_proctime() );
-#endif    
-
-    }
-
-    date  = giet_proctime();
-    H_END[cluster_id][lpid] = date;
-
-#if VERBOSE 
-printf( "\n[CONVOL] task[%d,%d,%d] completes horizontal filter"
-        " at cycle %d\n",
-        x, y, lpid, date );
-#else
-if ( (x==0) && (y==0) && (lpid==0) ) 
-printf( "\n[CONVOL] task[0,0,0] completes horizontal filter"
-        " at cycle %d\n", date );
-#endif
-
-    /////////////////////////////
-    #if USE_SQT_BARRIER
-    sqt_barrier_wait( &barrier );
-    #else
-    barrier_wait( &barrier );
-    #endif
-
-
-    ///////////////////////////////////////////////////////////////
-    // parallel vertical filter : 
-    // C <= transpose(FV(B))
-    // Each task computes (NP/ntasks) columns
-    // The image must be extended :
-    // if (l<0)    TB(cluster_id,p,l) == TB(cluster_id,p,0)
-    // if (l>NL-1)   TB(cluster_id,p,l) == TB(cluster_id,p,NL-1)
-    ///////////////////////////////////////////////////////////////
-
-    date  = giet_proctime();
-    V_BEG[cluster_id][lpid] = date;
-
-#if VERBOSE 
-printf( "\n[CONVOL] task[%d,%d,%d] starts vertical filter"
-        " at cycle %d\n",
-        x, y, lpid, date );
-#else
-if ( (x==0) && (y==0) && (lpid==0) ) 
-printf( "\n[CONVOL] task[0,0,0] starts vertical filter"
-        " at cycle %d\n", date );
-#endif
-
-    // l = absolute line index / p = absolute pixel index
-    // first & last define which pixels are handled by a given task
-
-    first = task_id * pixels_per_task;
-    last  = first + pixels_per_task;
-
-    for (p = first; p < last; p++)
-    {
-        // src_c and src_p are the cluster index and the pixel index for B
-        int src_c = p / pixels_per_cluster;
-        int src_p = p % pixels_per_cluster;
-
-        int sum_l;
-
-        // We use the specific values of the vertical ep-filter
-        // To minimize the number of tests, the NL lines are split in three domains 
-
-        // first domain : explicit computation for the first 18 values
-        for (l = 0; l < 18; l++)
-        {
-            // dst_c and dst_l are the cluster index and the line index for C
-            int dst_c = l / lines_per_cluster;
-            int dst_l = l % lines_per_cluster;
-
-            for (z = 0, sum_l = 0; z < 35; z++)
-            {
-                sum_l = sum_l + vf[z] * TB(src_c, src_p, max(l - 17 + z,0) );
-            }
-            TC(dst_c, dst_l, p) = sum_l / vnorm;
-        }
-        // second domain
-        for (l = 18; l < NL - 17; l++)
-        {
-            // dst_c and dst_l are the cluster index and the line index for C
-            int dst_c = l / lines_per_cluster;
-            int dst_l = l % lines_per_cluster;
-
-            sum_l = sum_l + TB(src_c, src_p, l + 4)
-                  + TB(src_c, src_p, l + 8)
-                  + TB(src_c, src_p, l + 11)
-                  + TB(src_c, src_p, l + 15)
-                  + TB(src_c, src_p, l + 17)
-                  - TB(src_c, src_p, l - 5)
-                  - TB(src_c, src_p, l - 9)
-                  - TB(src_c, src_p, l - 12)
-                  - TB(src_c, src_p, l - 16)
-                  - TB(src_c, src_p, l - 18);
-
-            TC(dst_c, dst_l, p) = sum_l / vnorm;
-        }
-        // third domain
-        for (l = NL - 17; l < NL; l++)
-        {
-            // dst_c and dst_l are the cluster index and the line index for C
-            int dst_c = l / lines_per_cluster;
-            int dst_l = l % lines_per_cluster;
-
-            sum_l = sum_l + TB(src_c, src_p, min(l + 4, NL - 1))
-                  + TB(src_c, src_p, min(l + 8, NL - 1))
-                  + TB(src_c, src_p, min(l + 11, NL - 1))
-                  + TB(src_c, src_p, min(l + 15, NL - 1))
-                  + TB(src_c, src_p, min(l + 17, NL - 1))
-                  - TB(src_c, src_p, l - 5)
-                  - TB(src_c, src_p, l - 9)
-                  - TB(src_c, src_p, l - 12)
-                  - TB(src_c, src_p, l - 16)
-                  - TB(src_c, src_p, l - 18);
-
-            TC(dst_c, dst_l, p) = sum_l / vnorm;
-        }
-
-#if SUPER_VERBOSE
-printf(" - column %d computed at cycle %d\n", p, giet_proctime());
-#endif 
-
-    }
-
-    date  = giet_proctime();
-    V_END[cluster_id][lpid] = date;
-
-#if VERBOSE 
-printf( "\n[CONVOL] task[%d,%d,%d] completes vertical filter"
-        " at cycle %d\n",
-        x, y, lpid, date );
-#else
-if ( (x==0) && (y==0) && (lpid==0) ) 
-printf( "\n[CONVOL] task[0,0,0] completes vertical filter"
-        " at cycle %d\n", date );
-#endif
-
-    ////////////////////////////
-    #if USE_SQT_BARRIER
-    sqt_barrier_wait( &barrier );
-    #else
-    barrier_wait( &barrier );
-    #endif
-
-    ////////////////////////////////////////////////////////////////
-    // Optional parallel display of the final image Z <= D + C
-    // Eah task displays (NL/ntasks) lines. (one byte per pixel).
-    ////////////////////////////////////////////////////////////////
-
-    if ( FINAL_DISPLAY_ENABLE )
-    {
-        date  = giet_proctime();
-        D_BEG[cluster_id][lpid] = date;
-
-#if VERBOSE
-printf( "\n[CONVOL] task[%d,%d,%d] starts final display"
-        " at cycle %d\n",
-        x, y, lpid, date);
-#else
-if ( (x==0) && (y==0) && (lpid==0) ) 
-printf( "\n[CONVOL] task[0,0,0] starts final display"
-        " at cycle %d\n", date );
-#endif
-
-        unsigned int line;
-        unsigned int offset = lines_per_task * lpid;
-
-        for ( l = 0 ; l < lines_per_task ; l++ )
-        {
-            line = offset + l;
-
-            for ( p = 0 ; p < NP ; p++ )
-            {
-                TZ(cluster_id, line, p) = 
-                   (unsigned char)( (TD(cluster_id, line, p) + 
-                                     TC(cluster_id, line, p) ) >> 8 );
-            }
-
-            giet_fbf_sync_write( NP*(l + (task_id * lines_per_task) ), 
-                                 &TZ(cluster_id, line, 0), 
-                                 NP);
-        }
-
-        date  = giet_proctime();
-        D_END[cluster_id][lpid] = date;
-
-#if VERBOSE
-printf( "\n[CONVOL] task[%d,%d,%d] completes final display"
-        " at cycle %d\n",
-        x, y, lpid, date);
-#else
-if ( (x==0) && (y==0) && (lpid==0) ) 
-printf( "\n[CONVOL] task[0,0,0] completes final display"
-        " at cycle %d\n", date );
-#endif
-     
-    //////////////////////////////
-    #if USE_SQT_BARRIER
-    sqt_barrier_wait( &barrier );
-    #else
-    barrier_wait( &barrier );
-    #endif
-
-    }
-
-    /////////////////////////////////////////////////////////
-    // Task[0,0,0] makes the instrumentation 
-    /////////////////////////////////////////////////////////
-
-    if ( (x==0) && (y==0) && (lpid==0) )
-    {
-        date  = giet_proctime();
-        printf("\n[CONVOL] task[0,0,0] starts instrumentation"
-               " at cycle %d\n\n", date );
-
-        int cc, pp;
-
-        unsigned int min_start = 0xFFFFFFFF;
-        unsigned int max_start = 0;
-
-        unsigned int min_h_beg = 0xFFFFFFFF;
-        unsigned int max_h_beg = 0;
-
-        unsigned int min_h_end = 0xFFFFFFFF;
-        unsigned int max_h_end = 0;
-
-        unsigned int min_v_beg = 0xFFFFFFFF;
-        unsigned int max_v_beg = 0;
-
-        unsigned int min_v_end = 0xFFFFFFFF;
-        unsigned int max_v_end = 0;
-
-        unsigned int min_d_beg = 0xFFFFFFFF;
-        unsigned int max_d_beg = 0;
-
-        unsigned int min_d_end = 0xFFFFFFFF;
-        unsigned int max_d_end = 0;
-
-        for (cc = 0; cc < nclusters; cc++)
-        {
-            for (pp = 0; pp < nprocs; pp++ )
-            {
-                if (START[cc][pp] < min_start) min_start = START[cc][pp];
-                if (START[cc][pp] > max_start) max_start = START[cc][pp];
-
-                if (H_BEG[cc][pp] < min_h_beg) min_h_beg = H_BEG[cc][pp];
-                if (H_BEG[cc][pp] > max_h_beg) max_h_beg = H_BEG[cc][pp];
-
-                if (H_END[cc][pp] < min_h_end) min_h_end = H_END[cc][pp];
-                if (H_END[cc][pp] > max_h_end) max_h_end = H_END[cc][pp];
-
-                if (V_BEG[cc][pp] < min_v_beg) min_v_beg = V_BEG[cc][pp];
-                if (V_BEG[cc][pp] > max_v_beg) max_v_beg = V_BEG[cc][pp];
-
-                if (V_END[cc][pp] < min_v_end) min_v_end = V_END[cc][pp];
-                if (V_END[cc][pp] > max_v_end) max_v_end = V_END[cc][pp];
-
-                if (D_BEG[cc][pp] < min_d_beg) min_d_beg = D_BEG[cc][pp];
-                if (D_BEG[cc][pp] > max_d_beg) max_d_beg = D_BEG[cc][pp];
-
-                if (D_END[cc][pp] < min_d_end) min_d_end = D_END[cc][pp];
-                if (D_END[cc][pp] > max_d_end) max_d_end = D_END[cc][pp];
-            }
-        }
-
-        printf(" - START : min = %d / max = %d / med = %d / delta = %d\n",
-               min_start, max_start, (min_start+max_start)/2, max_start-min_start);
-
-        printf(" - H_BEG : min = %d / max = %d / med = %d / delta = %d\n",
-               min_h_beg, max_h_beg, (min_h_beg+max_h_beg)/2, max_h_beg-min_h_beg);
-
-        printf(" - H_END : min = %d / max = %d / med = %d / delta = %d\n",
-               min_h_end, max_h_end, (min_h_end+max_h_end)/2, max_h_end-min_h_end);
-
-        printf(" - V_BEG : min = %d / max = %d / med = %d / delta = %d\n",
-               min_v_beg, max_v_beg, (min_v_beg+max_v_beg)/2, max_v_beg-min_v_beg);
-
-        printf(" - V_END : min = %d / max = %d / med = %d / delta = %d\n",
-               min_v_end, max_v_end, (min_v_end+max_v_end)/2, max_v_end-min_v_end);
-
-        printf(" - D_BEG : min = %d / max = %d / med = %d / delta = %d\n",
-               min_d_beg, max_d_beg, (min_d_beg+max_d_beg)/2, max_d_beg-min_d_beg);
-
-        printf(" - D_END : min = %d / max = %d / med = %d / delta = %d\n",
-               min_d_end, max_d_end, (min_d_end+max_d_end)/2, max_d_end-min_d_end);
-
-        printf( "\n General Scenario (Kcycles for each step)\n" );
-        printf( " - BOOT OS           = %d\n", (min_start            )/1000 );
-        printf( " - LOAD IMAGE        = %d\n", (min_h_beg - min_start)/1000 );
-        printf( " - H_FILTER          = %d\n", (max_h_end - min_h_beg)/1000 );
-        printf( " - BARRIER HORI/VERT = %d\n", (min_v_beg - max_h_end)/1000 );
-        printf( " - V_FILTER          = %d\n", (max_v_end - min_v_beg)/1000 );
-        printf( " - BARRIER VERT/DISP = %d\n", (min_d_beg - max_v_end)/1000 );
-        printf( " - DISPLAY           = %d\n", (max_d_end - min_d_beg)/1000 );
-
-        instrumentation_ok = 1;
-    }
-    else
-    {
-        while ( instrumentation_ok == 0 );
-    }
-
-    giet_exit( "completed");
-
-} // end main()
-
-// Local Variables:
-// tab-width: 3
-// c-basic-offset: 3
-// c-file-offsets:((innamespace . 0)(inline-open . 0))
-// indent-tabs-mode: nil
-// End:
-
-// vim: filetype=cpp:expandtab:shiftwidth=3:tabstop=3:softtabstop=3
-
-
Index: soft/giet_vm/applications/coproc/Makefile
===================================================================
--- soft/giet_vm/applications/coproc/Makefile	(revision 707)
+++ soft/giet_vm/applications/coproc/Makefile	(revision 708)
@@ -1,6 +1,12 @@
+
+CC = mipsel-unknown-elf-gcc
+AS = mipsel-unknown-elf-as
+LD = mipsel-unknown-elf-ld
+DU = mipsel-unknown-elf-objdump
+AR = mipsel-unknown-elf-ar
 
 APP_NAME = coproc
 
-OBJS= main.o 
+OBJS= coproc.o 
 
 LIBS= -L../../build/libs -luser
Index: soft/giet_vm/applications/coproc/coproc.c
===================================================================
--- soft/giet_vm/applications/coproc/coproc.c	(revision 708)
+++ soft/giet_vm/applications/coproc/coproc.c	(revision 708)
@@ -0,0 +1,126 @@
+///////////////////////////////////////////////////////////////////////////////////////
+//  file   : coproc.c
+//  date   : avril 2015
+//  author : Alain Greiner
+///////////////////////////////////////////////////////////////////////////////////////
+//  This file describes the single thread "coproc" application.
+//  It uses the GCD (Greater Common Divider) hardware coprocessor
+//  to make the GCD computation between two vectors of 32 bits integers.
+//  The vectors size is defined by the VECTOR_SIZE parameter.
+///////////////////////////////////////////////////////////////////////////////////////
+
+
+#include "stdio.h"
+#include "mapping_info.h"       // for coprocessors types an modes
+
+#define  VECTOR_SIZE 128   
+
+#define  DMA_MODE    MODE_DMA_IRQ
+
+#define  VERBOSE     1
+
+// Memory buffers for coprocessor
+unsigned int opa[VECTOR_SIZE] __attribute__((aligned(64)));
+unsigned int opb[VECTOR_SIZE] __attribute__((aligned(64)));
+unsigned int res[VECTOR_SIZE] __attribute__((aligned(64)));
+
+/////////////////////////////////////////
+__attribute__ ((constructor)) void main()
+{
+    // get processor identifiers
+    unsigned int    x;
+    unsigned int    y;
+    unsigned int    lpid;
+    giet_proc_xyp( &x, &y, &lpid );
+
+    // get a private TTY terminal
+    giet_tty_alloc( 0 );
+
+    giet_tty_printf("\n*** Starting coproc application on processor"
+                    "[%d,%d,%d] at cycle %d\n", 
+                    x, y, lpid, giet_proctime() );
+
+    // initializes opa & opb buffers
+    unsigned int word;
+    for ( word = 0 ; word < VECTOR_SIZE ; word++ )
+    {
+        opa[word] = giet_rand() + 1;
+        opb[word] = giet_rand() + 1;
+    }
+
+    unsigned int coproc_info;
+
+    /////////////////////// request a GCD coprocessor
+    giet_coproc_alloc( MWR_SUBTYPE_GCD, &coproc_info );
+
+    // check coprocessor ports
+    unsigned int nb_to_coproc   = (coproc_info    ) & 0xFF;
+    unsigned int nb_from_coproc = (coproc_info>> 8) & 0xFF;
+    unsigned int nb_config      = (coproc_info>>16) & 0xFF;
+    unsigned int nb_status      = (coproc_info>>24) & 0xFF;
+    giet_pthread_assert( ((nb_to_coproc   == 2) &&
+                         (nb_from_coproc == 1) &&
+                         (nb_config      == 1) &&
+                         (nb_status      == 0) ) ,
+                         "wrong GCD coprocessor interface" );
+
+#if  VERBOSE
+giet_tty_printf("\n*** get GCD coprocessor at cycle %d\n", giet_proctime() );
+#endif
+
+    //////////////////////// initializes channel for OPA
+    giet_coproc_channel_t opa_desc;
+    opa_desc.channel_mode = DMA_MODE;
+    opa_desc.buffer_size  = VECTOR_SIZE<<2;
+    opa_desc.buffer_vaddr = (unsigned int)opa;
+    giet_coproc_channel_init( 0 , &opa_desc );
+    
+    //////////////////////// initializes channel for OPB
+    giet_coproc_channel_t opb_desc;
+    opb_desc.channel_mode = DMA_MODE;
+    opb_desc.buffer_size  = VECTOR_SIZE<<2;
+    opb_desc.buffer_vaddr = (unsigned int)opb;
+    giet_coproc_channel_init( 1 , &opb_desc );
+    
+    //////////////////////// initializes channel for RES
+    giet_coproc_channel_t res_desc;
+    res_desc.channel_mode = DMA_MODE;
+    res_desc.buffer_size  = VECTOR_SIZE<<2;
+    res_desc.buffer_vaddr = (unsigned int)res;
+    giet_coproc_channel_init( 2 , &res_desc );
+    
+#if  VERBOSE
+giet_tty_printf("\n*** channels initialized at cycle %d\n", giet_proctime() );
+#endif
+
+    /////////////////////// starts communication channels
+    giet_coproc_run( 0 );
+
+#if  VERBOSE
+giet_tty_printf("\n*** start GCD coprocessor at cycle %d\n", giet_proctime() );
+#endif
+
+    /////////////////////// wait coprocessor completion
+    if ( DMA_MODE == MODE_DMA_NO_IRQ )
+    {
+        giet_coproc_completed( );
+    }
+
+#if  VERBOSE
+giet_tty_printf("\n*** GCD computation completed at cycle %d\n", giet_proctime() );
+#endif
+
+    // display result
+    for ( word = 0 ; word < VECTOR_SIZE ; word++ )
+    {
+        giet_tty_printf("pgcd( %d , %d ) = %d\n",
+        opa[word] , opb[word] , res[word] );
+    }
+
+    ////////////////////// release GCD coprocessor
+    giet_coproc_release( 0 );
+
+    giet_pthread_exit("completed");
+
+} // end main
+
Index: soft/giet_vm/applications/coproc/coproc.py
===================================================================
--- soft/giet_vm/applications/coproc/coproc.py	(revision 707)
+++ soft/giet_vm/applications/coproc/coproc.py	(revision 708)
@@ -33,5 +33,5 @@
     x = 0
     y = 0
-    p = 0
+    p = 1
 
     assert( (x < x_size) and (y < y_size) )
@@ -50,5 +50,5 @@
 
     # create vspace
-    vspace = mapping.addVspace( name = 'coproc', startname = 'coproc_data' )
+    vspace = mapping.addVspace( name = 'coproc', startname = 'coproc_data', active = False )
     
     # data vseg in cluster[x,y]
@@ -69,7 +69,12 @@
                      local = False, big = True )
 
-    # one task on processor[x,y,0]
-    mapping.addTask( vspace, 'coproc', 0 , x , y , 0 ,
-                     'coproc_stack' , '' , 0 )
+    # one thread on processor[x,y,p]
+    mapping.addThread( vspace,
+                       'coproc',
+                       True,                    # is_main
+                       x, y, p,
+                       'coproc_stack',
+                       '',                      # no heap
+                       0 )                      # start_id
 
     # extend mapping name
Index: soft/giet_vm/applications/coproc/main.c
===================================================================
--- soft/giet_vm/applications/coproc/main.c	(revision 707)
+++ 	(revision )
@@ -1,122 +1,0 @@
-///////////////////////////////////////////////////////////////////////////////////////
-//  file   : main.c  (for coproc application)
-//  date   : avril 2015
-//  author : Alain Greiner
-///////////////////////////////////////////////////////////////////////////////////////
-//  This file describes the single thread "coproc" application.
-//  It uses the embedded GCD (Greater Common Divider) coprocessor to make
-//  the GCD computation between two vectors of 32 bits integers.
-//  The vectors size is defined by the VECTOR_SIZE parameter.
-///////////////////////////////////////////////////////////////////////////////////////
-
-
-#include "stdio.h"
-#include "mapping_info.h"       // for coprocessors types an modes
-
-#define  VECTOR_SIZE 128   
-
-#define  DMA_MODE    MODE_DMA_IRQ
-
-#define VERBOSE      1
-
-// Memory buffers for coprocessor
-unsigned int opa[VECTOR_SIZE] __attribute__((aligned(64)));
-unsigned int opb[VECTOR_SIZE] __attribute__((aligned(64)));
-unsigned int res[VECTOR_SIZE] __attribute__((aligned(64)));
-
-/////////////////////////////////////////
-__attribute__ ((constructor)) void main()
-{
-    // get processor identifiers
-    unsigned int    x;
-    unsigned int    y;
-    unsigned int    lpid;
-    giet_proc_xyp( &x, &y, &lpid );
-
-    // get a private TTY terminal
-    giet_tty_alloc( 0 );
-
-    giet_tty_printf("\n*** Starting coproc application on processor"
-                    "[%d,%d,%d] at cycle %d\n", 
-                    x, y, lpid, giet_proctime() );
-
-    // initializes opa & opb buffers
-    unsigned int word;
-    for ( word = 0 ; word < VECTOR_SIZE ; word++ )
-    {
-        opa[word] = giet_rand() + 1;
-        opb[word] = giet_rand() + 1;
-    }
-
-    unsigned int coproc_info;
-
-    /////////////////////// request a GCD coprocessor
-    giet_coproc_alloc( MWR_SUBTYPE_GCD, &coproc_info );
-
-    // check coprocessor ports
-    unsigned int nb_to_coproc   = (coproc_info    ) & 0xFF;
-    unsigned int nb_from_coproc = (coproc_info>> 8) & 0xFF;
-    unsigned int nb_config      = (coproc_info>>16) & 0xFF;
-    unsigned int nb_status      = (coproc_info>>24) & 0xFF;
-    giet_assert( ((nb_to_coproc   == 2) &&
-                  (nb_from_coproc == 1) &&
-                  (nb_config      == 1) &&
-                  (nb_status      == 0) ) ,
-                  "wrong GCD coprocessor interface" );
-
-if ( VERBOSE )
-giet_tty_printf("\n*** get GCD coprocessor at cycle %d\n", giet_proctime() );
-
-    //////////////////////// initializes channel for OPA
-    giet_coproc_channel_t opa_desc;
-    opa_desc.channel_mode = DMA_MODE;
-    opa_desc.buffer_size  = VECTOR_SIZE<<2;
-    opa_desc.buffer_vaddr = (unsigned int)opa;
-    giet_coproc_channel_init( 0 , &opa_desc );
-    
-    //////////////////////// initializes channel for OPB
-    giet_coproc_channel_t opb_desc;
-    opb_desc.channel_mode = DMA_MODE;
-    opb_desc.buffer_size  = VECTOR_SIZE<<2;
-    opb_desc.buffer_vaddr = (unsigned int)opb;
-    giet_coproc_channel_init( 1 , &opb_desc );
-    
-    //////////////////////// initializes channel for RES
-    giet_coproc_channel_t res_desc;
-    res_desc.channel_mode = DMA_MODE;
-    res_desc.buffer_size  = VECTOR_SIZE<<2;
-    res_desc.buffer_vaddr = (unsigned int)res;
-    giet_coproc_channel_init( 2 , &res_desc );
-    
-if ( VERBOSE )
-giet_tty_printf("\n*** channels initialized at cycle %d\n", giet_proctime() );
-
-    /////////////////////// starts communication channels
-    giet_coproc_run( 0 );
-
-if ( VERBOSE )
-giet_tty_printf("\n*** start GCD coprocessor at cycle %d\n", giet_proctime() );
-
-    /////////////////////// wait coprocessor completion
-    if ( DMA_MODE == MODE_DMA_NO_IRQ )
-    {
-        giet_coproc_completed( );
-    }
-
-if ( VERBOSE )
-giet_tty_printf("\n*** GCD computation completed at cycle %d\n", giet_proctime() );
-
-    // display result
-    for ( word = 0 ; word < VECTOR_SIZE ; word++ )
-    {
-        giet_tty_printf("pgcd( %d , %d ) = %d\n",
-        opa[word] , opb[word] , res[word] );
-    }
-
-    ////////////////////// release GCD coprocessor
-    giet_coproc_release( 0 );
-
-    giet_exit("completed");
-
-} // end main
-
Index: soft/giet_vm/applications/display/Makefile
===================================================================
--- soft/giet_vm/applications/display/Makefile	(revision 707)
+++ soft/giet_vm/applications/display/Makefile	(revision 708)
@@ -1,6 +1,12 @@
+
+CC = mipsel-unknown-elf-gcc
+AS = mipsel-unknown-elf-as
+LD = mipsel-unknown-elf-ld
+DU = mipsel-unknown-elf-objdump
+AR = mipsel-unknown-elf-ar
 
 APP_NAME = display
 
-OBJS= main.o 
+OBJS= display.o 
 
 LIBS= -L../../build/libs -luser
Index: soft/giet_vm/applications/display/display.c
===================================================================
--- soft/giet_vm/applications/display/display.c	(revision 708)
+++ soft/giet_vm/applications/display/display.c	(revision 708)
@@ -0,0 +1,129 @@
+///////////////////////////////////////////////////////////////////////////////////////
+//  file   : display.c  
+//  date   : may 2014
+//  author : Alain Greiner
+///////////////////////////////////////////////////////////////////////////////////////
+//  This file describes the single thread "display" application.
+//  It uses the external chained buffer DMA to display a stream
+//  of images on the frame buffer.  
+///////////////////////////////////////////////////////////////////////////////////////
+
+#include <stdio.h>
+#include <hard_config.h>     // To check Frame Buffer size
+
+#define FILENAME    "misc/images_128.raw"
+#define NPIXELS     128
+#define NLINES      128
+#define NIMAGES     10                  
+
+#define INTERACTIVE 0
+
+unsigned char buf0[NPIXELS*NLINES] __attribute__((aligned(64)));
+unsigned char buf1[NPIXELS*NLINES] __attribute__((aligned(64)));
+
+unsigned int  sts0[16]  __attribute__((aligned(64)));
+unsigned int  sts1[16]  __attribute__((aligned(64)));
+
+////////////////////////////////////////////
+__attribute__((constructor)) void main()
+////////////////////////////////////////////
+{
+    // get processor identifiers
+    unsigned int    x;
+    unsigned int    y; 
+    unsigned int    p;
+    giet_proc_xyp( &x, &y, &p );
+
+    int             fd;
+    unsigned int    image = 0;
+
+    char            byte;
+
+    // parameters checking
+    if ( (NPIXELS != FBUF_X_SIZE) || (NLINES != FBUF_Y_SIZE) )
+    {
+        giet_pthread_exit("[DISPLAY ERROR] Frame buffer size does not fit image size");
+    }
+
+    // get a private TTY 
+    giet_tty_alloc(0);
+
+    giet_tty_printf("\n[DISPLAY] P[%d,%d,%d] starts at cycle %d\n"
+                    "  - buf0_vaddr = %x\n" 
+                    "  - buf1_vaddr = %x\n" 
+                    "  - sts0_vaddr = %x\n" 
+                    "  - sts1_vaddr = %x\n",
+                    x, y, p, giet_proctime(),
+                    buf0, buf1, sts0, sts1 );
+
+    // open file
+    fd = giet_fat_open( FILENAME , 0 );
+
+    giet_tty_printf("\n[DISPLAY] P[%d,%d,%d] open file %s at cycle %d\n", 
+                    x, y, p, FILENAME, giet_proctime() );
+
+    // get a Chained Buffer DMA channel
+    giet_fbf_cma_alloc();
+
+    // initialize the source and destination chbufs
+    giet_fbf_cma_init_buf( buf0 , buf1 , sts0 , sts1 );
+
+    // start Chained Buffer DMA channel
+    giet_fbf_cma_start( NPIXELS*NLINES );
+    
+    giet_tty_printf("\n[DISPLAY] Proc[%d,%d,%d] starts CMA at cycle %d\n", 
+                    x, y, p, giet_proctime() );
+
+    // Main loop on images
+    while ( 1 )
+    {
+        // load buf0
+        giet_fat_read( fd, buf0, NPIXELS*NLINES );
+
+        giet_tty_printf("\n[DISPLAY] Proc[%d,%d,%d] load image %d at cycle %d\n", 
+                        x, y, p, image, giet_proctime() );
+
+        // display buf0
+        giet_fbf_cma_display( 0 );
+
+        giet_tty_printf("\n[DISPLAY] Proc[%d,%d,%d] display image %d at cycle %d\n", 
+                        x, y, p, image, giet_proctime() );
+
+        image++;
+
+        if ( image == NIMAGES )
+        {
+            image = 0;
+            giet_fat_lseek( fd , 0 , 0 );
+        }
+
+        if ( INTERACTIVE ) giet_tty_getc( &byte );
+
+        // load buf1
+        giet_fat_read( fd, buf1, NPIXELS*NLINES );
+
+        giet_tty_printf("\n[DISPLAY] Proc[%d,%d,%d] load image %d at cycle %d\n", 
+                        x, y, p, image, giet_proctime() );
+
+        // display buf1
+        giet_fbf_cma_display( 1 );
+
+        giet_tty_printf("\n[DISPLAY] Proc[%d,%d,%d] display image %d at cycle %d\n", 
+                        x, y, p, image, giet_proctime() );
+
+        image++;
+
+        if ( image == NIMAGES )
+        {
+            image = 0;
+            giet_fat_lseek( fd , 0 , 0 );
+        }
+
+        if ( INTERACTIVE ) giet_tty_getc( &byte );
+    }
+
+    // stop Chained buffer DMA channel
+    giet_fbf_cma_stop();
+
+    giet_pthread_exit("completed");
+}
Index: soft/giet_vm/applications/display/display.py
===================================================================
--- soft/giet_vm/applications/display/display.py	(revision 707)
+++ soft/giet_vm/applications/display/display.py	(revision 708)
@@ -18,4 +18,9 @@
     x_width   = mapping.x_width
     y_width   = mapping.y_width
+
+    # define thread placement
+    x = 0
+    y = 0
+    p = 2
 
     # define vsegs base & size
@@ -58,5 +63,10 @@
 
     # task 
-    mapping.addTask( vspace, 'disp', 0, 0, 0, 1, 'disp_stack', 'disp_heap', 0 )
+    mapping.addThread( vspace, 'display', 
+                       True,               # is_main
+                       x, y, p,
+                       'disp_stack',
+                       'disp_heap',
+                       0 )                 # startid
 
     # extend mapping name
Index: soft/giet_vm/applications/display/main.c
===================================================================
--- soft/giet_vm/applications/display/main.c	(revision 707)
+++ 	(revision )
@@ -1,129 +1,0 @@
-///////////////////////////////////////////////////////////////////////////////////////
-//  file   : main.c     (display application)
-//  date   : may 2014
-//  author : Alain Greiner
-///////////////////////////////////////////////////////////////////////////////////////
-//  This file describes the single thread "display" application.
-//  It uses the external chained buffer DMA to display a stream
-//  of images on the frame buffer.  
-///////////////////////////////////////////////////////////////////////////////////////
-
-#include <stdio.h>
-#include <hard_config.h>     // To check Frame Buffer size
-
-#define FILENAME    "misc/images_128.raw"
-#define NPIXELS     128
-#define NLINES      128
-#define NIMAGES     10                  
-
-#define INTERACTIVE 0
-
-unsigned char buf0[NPIXELS*NLINES] __attribute__((aligned(64)));
-unsigned char buf1[NPIXELS*NLINES] __attribute__((aligned(64)));
-
-unsigned int  sts0[16]  __attribute__((aligned(64)));
-unsigned int  sts1[16]  __attribute__((aligned(64)));
-
-////////////////////////////////////////////
-__attribute__((constructor)) void main()
-////////////////////////////////////////////
-{
-    // get processor identifiers
-    unsigned int    x;
-    unsigned int    y; 
-    unsigned int    p;
-    giet_proc_xyp( &x, &y, &p );
-
-    int             fd;
-    unsigned int    image = 0;
-
-    char            byte;
-
-    // parameters checking
-    if ( (NPIXELS != FBUF_X_SIZE) || (NLINES != FBUF_Y_SIZE) )
-    {
-        giet_exit("[DISPLAY ERROR] Frame buffer size does not fit image size");
-    }
-
-    // get a private TTY 
-    giet_tty_alloc(0);
-
-    giet_tty_printf("\n[DISPLAY] P[%d,%d,%d] starts at cycle %d\n"
-                    "  - buf0_vaddr = %x\n" 
-                    "  - buf1_vaddr = %x\n" 
-                    "  - sts0_vaddr = %x\n" 
-                    "  - sts1_vaddr = %x\n",
-                    x, y, p, giet_proctime(),
-                    buf0, buf1, sts0, sts1 );
-
-    // open file
-    fd = giet_fat_open( FILENAME , 0 );
-
-    giet_tty_printf("\n[DISPLAY] P[%d,%d,%d] open file %s at cycle %d\n", 
-                    x, y, p, FILENAME, giet_proctime() );
-
-    // get a Chained Buffer DMA channel
-    giet_fbf_cma_alloc();
-
-    // initialize the source and destination chbufs
-    giet_fbf_cma_init_buf( buf0 , buf1 , sts0 , sts1 );
-
-    // start Chained Buffer DMA channel
-    giet_fbf_cma_start( NPIXELS*NLINES );
-    
-    giet_tty_printf("\n[DISPLAY] Proc[%d,%d,%d] starts CMA at cycle %d\n", 
-                    x, y, p, giet_proctime() );
-
-    // Main loop on images
-    while ( 1 )
-    {
-        // load buf0
-        giet_fat_read( fd, buf0, NPIXELS*NLINES );
-
-        giet_tty_printf("\n[DISPLAY] Proc[%d,%d,%d] load image %d at cycle %d\n", 
-                        x, y, p, image, giet_proctime() );
-
-        // display buf0
-        giet_fbf_cma_display( 0 );
-
-        giet_tty_printf("\n[DISPLAY] Proc[%d,%d,%d] display image %d at cycle %d\n", 
-                        x, y, p, image, giet_proctime() );
-
-        image++;
-
-        if ( image == NIMAGES )
-        {
-            image = 0;
-            giet_fat_lseek( fd , 0 , 0 );
-        }
-
-        if ( INTERACTIVE ) giet_tty_getc( &byte );
-
-        // load buf1
-        giet_fat_read( fd, buf1, NPIXELS*NLINES );
-
-        giet_tty_printf("\n[DISPLAY] Proc[%d,%d,%d] load image %d at cycle %d\n", 
-                        x, y, p, image, giet_proctime() );
-
-        // display buf1
-        giet_fbf_cma_display( 1 );
-
-        giet_tty_printf("\n[DISPLAY] Proc[%d,%d,%d] display image %d at cycle %d\n", 
-                        x, y, p, image, giet_proctime() );
-
-        image++;
-
-        if ( image == NIMAGES )
-        {
-            image = 0;
-            giet_fat_lseek( fd , 0 , 0 );
-        }
-
-        if ( INTERACTIVE ) giet_tty_getc( &byte );
-    }
-
-    // stop Chained buffer DMA channel
-    giet_fbf_cma_stop();
-
-    giet_exit("display completed");
-}
Index: soft/giet_vm/applications/gameoflife/Makefile
===================================================================
--- soft/giet_vm/applications/gameoflife/Makefile	(revision 707)
+++ soft/giet_vm/applications/gameoflife/Makefile	(revision 708)
@@ -1,6 +1,12 @@
+
+CC = mipsel-unknown-elf-gcc
+AS = mipsel-unknown-elf-as
+LD = mipsel-unknown-elf-ld
+DU = mipsel-unknown-elf-objdump
+AR = mipsel-unknown-elf-ar
 
 APP_NAME = gameoflife
 
-OBJS= main.o 
+OBJS= gameoflife.o 
 
 LIBS= -L../../build/libs -luser
Index: soft/giet_vm/applications/gameoflife/gameoflife.c
===================================================================
--- soft/giet_vm/applications/gameoflife/gameoflife.c	(revision 708)
+++ soft/giet_vm/applications/gameoflife/gameoflife.c	(revision 708)
@@ -0,0 +1,338 @@
+//////////////////////////////////////////////////////////////////////////////////
+// File    : main.c  (for gameoflife)
+// Date    : November 2013 / February 2015
+// Authors :  Alexandre Joannou <alexandre.joannou@lip6.fr> november 2013
+//            Alain Greiner <alain.greiner@lip6.fr> february 2015
+//////////////////////////////////////////////////////////////////////////////////
+// This multi-threaded application is an emulation of the Game of Life automaton.
+// The world size is defined by the HEIGHT and WIDTH parameters.
+//
+// There is at most one thread per processor in the platform.
+// - If the number of processors is larger than the number of lines,
+//   the number of threads is equal to the number of lines, and
+//   each thread process one single line. 
+// - if the number of processors is not larger than the number of lines,
+//   the number of threads is equal to the number of processors, and
+//   each thread process HEIGHT/nthreads (or HEIGHT/nthreads + 1) lines.
+// 
+// Thread running on processor P(0,0,0) execute the main() function,
+// that initialises the barrier, the TTY terminal, the CMA controler,
+// and launch the other threads, before calling the execute function.
+// Other threads are just running the execute() function.
+// 
+// The total number of processors cannot be larger than 1024 = 16 * 16 * 4
+//////////////////////////////////////////////////////////////////////////////////
+
+#include "stdio.h"
+#include "limits.h"
+#include "user_barrier.h"
+#include "mapping_info.h"
+#include "hard_config.h"
+#include "malloc.h"
+
+#define WIDTH           FBUF_X_SIZE
+#define HEIGHT          FBUF_Y_SIZE
+
+#define VERBOSE         1
+
+typedef unsigned char uint8_t;
+
+typedef struct
+{
+    unsigned int    index;    // index of first line to be processed
+    unsigned int    lines;    // number of lines to be processed 
+}   arguments_t;
+
+arguments_t   args[1024];     // at most 1024 threads 
+
+uint8_t world[2][HEIGHT][WIDTH] __attribute__((aligned(64)));
+
+uint8_t display[2][HEIGHT][WIDTH] __attribute__((aligned(64)));
+
+unsigned int status0[16];
+unsigned int status1[16];
+
+giet_sqt_barrier_t barrier;
+
+////////////////////////////////////
+void init_world( unsigned int phase,
+                 unsigned int base_line,
+                 unsigned int nb_line )
+{
+   unsigned int x,y;
+   for (y = base_line ; y < base_line + nb_line ; y++)
+   {
+      for(x = 0 ; x < WIDTH ; x++) 
+      {
+         world[phase][y][x] = (giet_rand() >> (x % 8)) & 0x1;
+      }
+   }
+}
+
+//////////////////////////////////////////////////////
+uint8_t number_of_alive_neighbour( unsigned int phase,
+                                   unsigned int x, 
+                                   unsigned int y )
+{
+   uint8_t nb = 0;
+
+   nb += world[phase][(y - 1) % HEIGHT][(x - 1) % WIDTH];
+   nb += world[phase][ y              ][(x - 1) % WIDTH];
+   nb += world[phase][(y + 1) % HEIGHT][(x - 1) % WIDTH];
+   nb += world[phase][(y - 1) % HEIGHT][ x             ];
+   nb += world[phase][(y + 1) % HEIGHT][ x             ];
+   nb += world[phase][(y - 1) % HEIGHT][(x + 1) % WIDTH];
+   nb += world[phase][ y              ][(x + 1) % WIDTH];
+   nb += world[phase][(y + 1) % HEIGHT][(x + 1) % WIDTH];
+
+   return nb;
+}
+
+/////////////////////////////////////////
+uint8_t compute_cell( unsigned int phase,
+                      unsigned int x, 
+                      unsigned int y )
+{
+   uint8_t nb_neighbours_alive = number_of_alive_neighbour( phase, x , y );
+
+   if (world[phase][y][x] == 1) 
+   {
+      if (nb_neighbours_alive == 2 || nb_neighbours_alive == 3)  return 1;
+   }
+   else 
+   {
+      if (nb_neighbours_alive == 3) return 1;
+      else                          return world[phase][y][x];
+   }
+   return 0;
+}
+
+/////////////////////////////////////////
+void compute_new_gen( unsigned int phase,
+                      unsigned int base_line, 
+                      unsigned int nb_line )
+{
+   unsigned int x,y;
+   for (y = base_line; y < base_line + nb_line; y++)
+   {
+      for(x = 0; x < WIDTH ; x++) 
+      {
+         world[phase][y][x] = compute_cell( 1 - phase , x , y );  
+      }
+   }
+}
+
+////////////////////////////////////
+void copy_world( unsigned int phase,
+                 unsigned int base_line,
+                 unsigned int nb_line )
+{
+   unsigned int x,y;
+   for (y = base_line; y < base_line + nb_line; y++)
+   {
+      for(x = 0; x < WIDTH ; x++) 
+      {
+         display[phase][y][x] = world[phase][y][x]*255;  
+      }
+   }
+}
+
+
+
+///////////////////////////////////////////////////////////////
+__attribute__((constructor)) void execute( arguments_t* pargs )
+///////////////////////////////////////////////////////////////
+{
+   unsigned int nb_lines      = pargs->lines;
+   unsigned int base_line     = pargs->index;
+
+   ///////////// parallel world  initialization 
+
+   // All processors initialize world[0]
+   init_world( 0 , base_line , nb_lines );
+
+   // copy world[0] to display[0]
+   copy_world( 0 , base_line , nb_lines );
+
+   // synchronise with other procs
+   sqt_barrier_wait( &barrier );
+
+   // main() makes display[0]
+   if ( base_line == 0 ) giet_fbf_cma_display ( 0 );
+
+   //////////// evolution : 2 steps per iteration 
+
+   unsigned int i = 0;
+   while( 1 )
+   {
+      // compute world[1] from world[0]
+      compute_new_gen( 1 , base_line , nb_lines );
+
+      // copy world[1] to display[1]
+      copy_world( 1 , base_line , nb_lines );
+
+      // synchronise with other procs
+      sqt_barrier_wait( &barrier );
+
+      // main makes display[1]
+      if ( base_line == 0 ) giet_fbf_cma_display ( 1 );
+   
+#if VERBOSE
+      if ( base_line == 0 ) giet_tty_printf(" - step %d\n", 2*i );
+#endif
+   
+      // compute world[0] from world[1]
+      compute_new_gen( 0 , base_line , nb_lines );
+
+      // copy world[0] to display[0]
+      copy_world( 0 , base_line , nb_lines );
+
+      // synchronise with other procs
+      sqt_barrier_wait( &barrier );
+
+      // main makes display[0]
+      if ( base_line == 0 ) giet_fbf_cma_display ( 0 );
+
+#if VERBOSE
+      if ( base_line == 0 ) giet_tty_printf(" - step %d\n", 2*i + 1 );
+#endif
+
+      i++;
+
+   } // end evolution loop
+
+   giet_pthread_exit("Completed");
+
+} // end main()
+
+
+
+////////////////////////////////////////
+__attribute__((constructor)) void main()
+////////////////////////////////////////
+{
+   // get processor identifier
+   unsigned int x;
+   unsigned int y;
+   unsigned int p;
+   giet_proc_xyp( &x, &y, &p );
+
+   // get platform parameters
+   unsigned int x_size;
+   unsigned int y_size;
+   unsigned int nprocs;
+   giet_procs_number( &x_size, &y_size, &nprocs );
+
+   giet_pthread_assert( (x_size <= 16) , "x_size no larger than 16" );
+   giet_pthread_assert( (y_size <= 16) , "y_size no larger than 16" );
+   giet_pthread_assert( (nprocs <=  4) , "nprocs no larger than 16" );
+
+   // compute number of threads and min number of lines per thread
+   // extra is the number of threads that must process one extra line
+   unsigned int total_procs = x_size * y_size * nprocs; 
+   unsigned int nthreads;
+   unsigned int nlines;
+   unsigned int extra;
+   if ( total_procs > HEIGHT )
+   {
+      nthreads = HEIGHT;
+      nlines   = 1;
+      extra    = 0;
+   }
+   else
+   {
+      nthreads = total_procs;
+      nlines   = HEIGHT / total_procs;
+      extra    = HEIGHT % total_procs;  
+   }
+
+   // get a shared TTY 
+   giet_tty_alloc( 1 );
+
+   // get a Chained Buffer DMA channel
+   giet_fbf_cma_alloc();
+
+   // initializes the source and destination buffers
+   giet_fbf_cma_init_buf( &display[0][0][0] , 
+                          &display[1][0][0] , 
+                          status0 ,
+                          status1 );
+
+   // activates CMA channel
+   giet_fbf_cma_start( HEIGHT * WIDTH );
+
+   // initializes distributed heap
+   unsigned int cx;
+   unsigned int cy;
+   for ( cx = 0 ; cx < x_size ; cx++ )
+   {
+      for ( cy = 0 ; cy < y_size ; cy++ )
+      {
+         heap_init( cx , cy );
+      }
+   }
+
+   // initialises barrier
+   sqt_barrier_init( &barrier , x_size , y_size , nprocs );
+
+   giet_tty_printf("\n[GAMEOFLIFE] P[%d,%d,%d] completes initialisation at cycle %d\n"
+                   " nprocs = %d / nlines = %d / nthreads = %d\n", 
+                   x, y, p, giet_proctime() , total_procs , HEIGHT , nthreads );
+
+   // compute arguments (index, nlines) for all threads
+   unsigned int n;                   // thread index
+   unsigned int index;               // first line index 
+   for ( n = 0 , index = 0 ; n < nthreads ; n++ )
+   {
+      if ( extra )
+      {
+         args[n].index = index;
+         args[n].lines = nlines + 1;
+         index         = index + nlines + 1;
+      }
+      else
+      {
+         args[n].index = index;
+         args[n].lines = nlines;
+         index         = index + nlines;
+      }
+#if VERBOSE      
+giet_tty_printf("[GAMEOFLIFE] Thread %d : first = %d / nlines = %d\n",
+                n , args[n].index , args[n].lines );
+#endif
+   }
+
+   // launch all other threads
+   pthread_t  trdid;                 // unused because no pthread_join()
+   for ( n = 1 ; n < nthreads ; n++ )
+   {
+      if ( giet_pthread_create( &trdid,
+                                NULL,                  // no attribute
+                                &execute,
+                                &args[n] ) )
+      {
+          giet_tty_printf("\n[TRANSPOSE ERROR] creating thread %x\n", n );
+          giet_pthread_exit( NULL );
+      }
+   }
+
+   // run execute function
+   execute( &args[0] );
+
+   giet_pthread_exit( "completed" );
+    
+} // end main()
+
+
+
+// Local Variables:
+// tab-width: 3
+// c-basic-offset: 3
+// c-file-offsets:((innamespace . 0)(inline-open . 0))
+// indent-tabs-mode: nil
+// End:
+
+// vim: filetype=cpp:expandtab:shiftwidth=3:tabstop=3:softtabstop=3
+
+
+
Index: soft/giet_vm/applications/gameoflife/gameoflife.py
===================================================================
--- soft/giet_vm/applications/gameoflife/gameoflife.py	(revision 707)
+++ soft/giet_vm/applications/gameoflife/gameoflife.py	(revision 708)
@@ -10,7 +10,5 @@
 #  This file describes the mapping of the multi-threaded "gameoflife" 
 #  application on a multi-clusters, multi-processors architecture.
-#  This include both the mapping of virtual segments on the clusters,
-#  and the mapping of tasks on processors.
-#  There is one task per processor.
+#  There is one thread per processor.
 #  The mapping of virtual segments is the following:
 #    - There is one shared data vseg in cluster[0][0]
@@ -102,10 +100,18 @@
             if ( mapping.clusters[cluster_id].procs ):
                 for p in xrange( nprocs ):
-                    trdid = (((x * y_size) + y) * nprocs) + p
+                    if (x == 0) and (y == 0) and (p == 0) :   # main thread
+                        startid = 1
+                        is_main = True
+                    else :                                    # other threads
+                        startid = 0
+                        is_main = False
 
-                    mapping.addTask( vspace, 'gol_%d_%d_%d' % (x,y,p),
-                                     trdid, x, y, p,
-                                     'gol_stack_%d_%d_%d' % (x,y,p),
-                                     'gol_heap_%d_%d' %(x,y) , 0 )  
+                    mapping.addThread( vspace,
+                                       'gol_%d_%d_%d' % (x,y,p),
+                                       is_main,
+                                       x, y, p,
+                                       'gol_stack_%d_%d_%d' % (x,y,p),
+                                       'gol_heap_%d_%d' % (x,y),
+                                       startid )
 
     # extend mapping name
Index: soft/giet_vm/applications/gameoflife/main.c
===================================================================
--- soft/giet_vm/applications/gameoflife/main.c	(revision 707)
+++ 	(revision )
@@ -1,276 +1,0 @@
-//////////////////////////////////////////////////////////////////////////////////
-// File    : main.c  (for gameoflife)
-// Date    : November 2013 / February 2015
-// Authors :  Alexandre Joannou <alexandre.joannou@lip6.fr> november 2013
-//            Alain Greiner <alain.greiner@lip6.fr> february 2015
-//
-// This multi-threaded application is an emulation of the Game of Life automaton.
-// The world size is defined by the HEIGHT and WIDTH parameters.
-// There is one task per processor.
-// Each task compute HEIGHT/nbprocs lines.
-// Task running on processor P(0,0,0) initialises the barrier, the TTY terminal,
-// and the chained buffer DMA controler.
-//
-// The number of processors must be a power of 2 not larger than HEIGHT.
-//////////////////////////////////////////////////////////////////////////////////
-
-#include "stdio.h"
-#include "limits.h"
-#include "user_barrier.h"
-#include "mapping_info.h"
-#include "hard_config.h"
-
-#define WIDTH           128
-#define HEIGHT          128
-#define NB_ITERATION    1000000000
-
-#define PRINTF(...) ({ if ( proc_id==0) { giet_tty_printf(__VA_ARGS__); } })
-
-typedef unsigned char uint8_t;
-
-uint8_t WORLD[2][HEIGHT][WIDTH] __attribute__((aligned(64)));
-
-uint8_t DISPLAY[2][HEIGHT][WIDTH] __attribute__((aligned(64)));
-
-unsigned int status0[16];
-unsigned int status1[16];
-
-giet_sqt_barrier_t barrier;
-
-volatile unsigned int init_ok;
-
-////////////////////////////////////
-void init_world( unsigned int phase,
-                 unsigned int base_line,
-                 unsigned int nb_line )
-{
-   unsigned int x,y;
-   for (y = base_line ; y < base_line + nb_line ; y++)
-   {
-      for(x = 0 ; x < WIDTH ; x++) 
-      {
-         WORLD[phase][y][x] = (giet_rand() >> (x % 8)) & 0x1;
-      }
-   }
-}
-
-//////////////////////////////////////////////////////
-uint8_t number_of_alive_neighbour( unsigned int phase,
-                                   unsigned int x, 
-                                   unsigned int y )
-{
-   uint8_t nb = 0;
-
-   nb += WORLD[phase][(y - 1) % HEIGHT][(x - 1) % WIDTH];
-   nb += WORLD[phase][ y              ][(x - 1) % WIDTH];
-   nb += WORLD[phase][(y + 1) % HEIGHT][(x - 1) % WIDTH];
-   nb += WORLD[phase][(y - 1) % HEIGHT][ x             ];
-   nb += WORLD[phase][(y + 1) % HEIGHT][ x             ];
-   nb += WORLD[phase][(y - 1) % HEIGHT][(x + 1) % WIDTH];
-   nb += WORLD[phase][ y              ][(x + 1) % WIDTH];
-   nb += WORLD[phase][(y + 1) % HEIGHT][(x + 1) % WIDTH];
-
-   return nb;
-}
-
-/////////////////////////////////////////
-uint8_t compute_cell( unsigned int phase,
-                      unsigned int x, 
-                      unsigned int y )
-{
-   uint8_t nb_neighbours_alive = number_of_alive_neighbour( phase, x , y );
-
-   if (WORLD[phase][y][x] == 1) 
-   {
-      if (nb_neighbours_alive == 2 || nb_neighbours_alive == 3)  return 1;
-   }
-   else 
-   {
-      if (nb_neighbours_alive == 3) return 1;
-      else                          return WORLD[phase][y][x];
-   }
-   return 0;
-}
-
-/////////////////////////////////////////
-void compute_new_gen( unsigned int phase,
-                      unsigned int base_line, 
-                      unsigned int nb_line )
-{
-   unsigned int x,y;
-   for (y = base_line; y < base_line + nb_line; y++)
-   {
-      for(x = 0; x < WIDTH ; x++) 
-      {
-         WORLD[phase][y][x] = compute_cell( 1 - phase , x , y );  
-      }
-   }
-}
-
-////////////////////////////////////
-void copy_world( unsigned int phase,
-                 unsigned int base_line,
-                 unsigned int nb_line )
-{
-   unsigned int x,y;
-   for (y = base_line; y < base_line + nb_line; y++)
-   {
-      for(x = 0; x < WIDTH ; x++) 
-      {
-         DISPLAY[phase][y][x] = WORLD[phase][y][x]*255;  
-      }
-   }
-}
-
-////////////////////////////////////////
-__attribute__((constructor)) void main()
-////////////////////////////////////////
-{
-   // get processor identifier
-   unsigned int x;
-   unsigned int y;
-   unsigned int p;
-   giet_proc_xyp( &x, &y, &p );
-
-   // get processors number
-   unsigned int x_size;
-   unsigned int y_size;
-   unsigned int nprocs;
-   giet_procs_number( &x_size, &y_size, &nprocs );
-
-   // compute continuous processor index & number of procs
-   unsigned int proc_id = (((x * y_size) + y) * nprocs) + p;  
-   unsigned int n_global_procs = x_size * y_size * nprocs; 
-
-   unsigned int i;
-
-   unsigned int nb_line       = HEIGHT / n_global_procs;
-   unsigned int base_line     = nb_line * proc_id; 
-   
-   // parameters checking
-   giet_assert( (n_global_procs <= HEIGHT),
-                " Number or processors larger than world height" );
-
-   giet_assert( ((WIDTH == FBUF_X_SIZE) && (HEIGHT == FBUF_Y_SIZE)),
-                "Frame Buffer size does not fit the world size" );
-   
-   giet_assert( ((x_size == 1) || (x_size == 2) || (x_size == 4) ||
-                 (x_size == 8) || (x_size == 16)),
-                "x_size must be a power of 2 no larger than 16" );
-
-   giet_assert( ((y_size == 1) || (y_size == 2) || (y_size == 4) ||
-                 (y_size == 8) || (y_size == 16)),
-                "y_size must be a power of 2 no larger than 16" );
-
-   giet_assert( ((nprocs == 1) || (nprocs == 2) || (nprocs == 4)), 
-                "nprocs must be a power of 2 no larger than 4" );
-
-   // P[0,0,0] makes initialisation
-   if ( proc_id == 0 )
-   {
-      // get a private TTY for P[0,0,0]
-      giet_tty_alloc( 0 );
-
-      // get a Chained Buffer DMA channel
-      giet_fbf_cma_alloc();
-
-      // initializes the source and destination buffers
-      giet_fbf_cma_init_buf( &DISPLAY[0][0][0] , 
-                             &DISPLAY[1][0][0] , 
-                             status0 ,
-                             status1 );
-
-      // activates CMA channel
-      giet_fbf_cma_start( HEIGHT * WIDTH );
-
-      // initializes distributed heap
-      unsigned int cx;
-      unsigned int cy;
-      for ( cx = 0 ; cx < x_size ; cx++ )
-      {
-         for ( cx = 0 ; cx < x_size ; cx++ )
-         {
-            heap_init( cx , cy );
-         }
-      }
-
-      // initialises barrier
-      sqt_barrier_init( &barrier , x_size , y_size , nprocs );
-
-      PRINTF("\n[GAMEOFLIFE] P[0,0,0] completes initialisation at cycle %d\n"
-             " nprocs = %d / nlines = %d\n", 
-             giet_proctime() , n_global_procs, HEIGHT );
-
-      // activates all other processors
-      init_ok = 1;
-   }
-   else
-   {
-      while ( init_ok == 0 ) asm volatile("nop\n nop\n nop");
-   }
-
-   ///////////// world  initialization ( All processors )
-
-   // All processors initialize WORLD[0]
-   init_world( 0 , base_line , nb_line );
-
-   // copy WORLD[0] to DISPLAY[0]
-   copy_world( 0 , base_line , nb_line );
-
-   // synchronise with other procs
-   sqt_barrier_wait( &barrier );
-
-   // P(0,0,0) displays DISPLAY[0]
-   if ( proc_id == 0 ) giet_fbf_cma_display ( 0 );
-
-   PRINTF("\n[GAMEOFLIFE] starts evolution at cycle %d\n", giet_proctime() );
-   
-   //////////// evolution : 2 steps per iteration 
-
-   for (i = 0 ; i < NB_ITERATION ; i++)
-   {
-      // compute WORLD[1] from WORLD[0]
-      compute_new_gen( 1 , base_line , nb_line );
-
-      // copy WORLD[1] to DISPLAY[1]
-      copy_world( 1 , base_line , nb_line );
-
-      // synchronise with other procs
-      sqt_barrier_wait( &barrier );
-
-      // P(0,0,0) displays DISPLAY[1]
-      if ( proc_id == 0 ) giet_fbf_cma_display ( 1 );
-   
-      PRINTF(" - step %d\n", 2*i );
-   
-      // compute WORLD[0] from WORLD[1]
-      compute_new_gen( 0 , base_line , nb_line );
-
-      // copy WORLD[0] to DISPLAY[0]
-      copy_world( 0 , base_line , nb_line );
-
-      // synchronise with other procs
-      sqt_barrier_wait( &barrier );
-
-      // P(0,0,0) displays DISPLAY[0]
-      if ( proc_id == 0 ) giet_fbf_cma_display ( 0 );
-
-      PRINTF(" - step %d\n", 2*i + 1 );
-   } // end main loop
-
-   PRINTF("\n*** End of main at cycle %d ***\n", giet_proctime());
-
-   giet_exit("Completed");
-} // end main()
-
-// Local Variables:
-// tab-width: 3
-// c-basic-offset: 3
-// c-file-offsets:((innamespace . 0)(inline-open . 0))
-// indent-tabs-mode: nil
-// End:
-
-// vim: filetype=cpp:expandtab:shiftwidth=3:tabstop=3:softtabstop=3
-
-
-
Index: soft/giet_vm/applications/raycast/Makefile
===================================================================
--- soft/giet_vm/applications/raycast/Makefile	(revision 707)
+++ soft/giet_vm/applications/raycast/Makefile	(revision 708)
@@ -1,6 +1,12 @@
+
+CC = mipsel-unknown-elf-gcc
+AS = mipsel-unknown-elf-as
+LD = mipsel-unknown-elf-ld
+DU = mipsel-unknown-elf-objdump
+AR = mipsel-unknown-elf-ar
 
 APP_NAME = raycast
 
-OBJS = ctrl.o disp.o game.o main.o
+OBJS = disp.o game.o raycast.o
 
 LIBS = -L../../build/libs -luser -lmath
Index: soft/giet_vm/applications/raycast/ctrl.c
===================================================================
--- soft/giet_vm/applications/raycast/ctrl.c	(revision 707)
+++ 	(revision )
@@ -1,32 +1,0 @@
-#include "ctrl.h"
-#include <math.h>
-
-// Exported functions
-
-void ctrlLogic(Game *game)
-{
-	game->player.dir += PLAYER_ROT;
-
-/*
-    if (g_tsTouch) {
-        if (g_tsX > 2 * SCREEN_WIDTH / 3) {
-            // Turn right
-            game->player.dir += PLAYER_ROT;
-        }
-        else if (g_tsX < 1 * SCREEN_WIDTH / 3) {
-            // Turn left
-            game->player.dir -= PLAYER_ROT;
-        }
-        if (g_tsY > 2 * SCREEN_HEIGHT / 3) {
-            // Move forward
-            game->player.x -= PLAYER_MOVE * cosf(game->player.dir);
-            game->player.y -= PLAYER_MOVE * sinf(game->player.dir);
-        }
-        else if (g_tsY < 1 * SCREEN_HEIGHT / 3) {
-            // Move backwards
-            game->player.x += PLAYER_MOVE * cosf(game->player.dir);
-            game->player.y += PLAYER_MOVE * sinf(game->player.dir);
-        }
-    }
-*/
-}
Index: soft/giet_vm/applications/raycast/ctrl.h
===================================================================
--- soft/giet_vm/applications/raycast/ctrl.h	(revision 707)
+++ 	(revision )
@@ -1,8 +1,0 @@
-#ifndef __CTRL_H
-#define __CTRL_H
-
-#include "game.h"
-
-void ctrlLogic(Game *game);
-
-#endif // __CTRL_H
Index: soft/giet_vm/applications/raycast/disp.c
===================================================================
--- soft/giet_vm/applications/raycast/disp.c	(revision 707)
+++ soft/giet_vm/applications/raycast/disp.c	(revision 708)
@@ -6,5 +6,5 @@
 #include <math.h>
 #include <hard_config.h>
-#include <user_sqt_lock.h>
+#include <user_lock.h>
 
 #define FIELD_OF_VIEW   (70.f * M_PI / 180.f)   // Camera field of view
@@ -13,25 +13,24 @@
 #define FLOOR_COLOR     (0x33)                  // dark gray
 
-// Globals
-
-static unsigned char*           buf[2];         // framebuffer
-static void *                   sts[2];         // for fbf_cma
-static unsigned int             cur_buf;        // current framebuffer
-static volatile unsigned int    slice_x;        // slice index (shared)
-static sqt_lock_t               slice_x_lock;   // slice index lock
-static volatile unsigned int    slice_cnt;      // slice count (shared)
-
-// Textures indexed by block number
-static unsigned char *g_tex[] =
-{
-    NULL, // 0
-    NULL, // rock
-    NULL, // door
-    NULL, // handle
-    NULL, // wood
-};
-
+
+////////////////////////////
+// Extern variables
+////////////////////////////
+
+extern unsigned char*  g_tex[5];
+extern unsigned char*  buf[2];
+extern void*           sts[2];
+extern unsigned int    cur_buf;
+extern unsigned int    slice_x;
+extern unsigned int    slice_count;
+extern sqt_lock_t      slice_get_lock;
+extern sqt_lock_t      slice_done_lock;
+extern Game            game;
+
+//////////////////////
 // Local functions
-
+////////////////////////
+
+/////////////////////////////////////////////////////////////////////////
 static void dispDrawColumnTex(int x, int y0, int y1, unsigned char *line)
 {
@@ -47,4 +46,5 @@
 }
 
+///////////////////////////////////////////////////////////////////////////
 static void dispDrawColumnSolid(int x, int y0, int y1, unsigned char color)
 {
@@ -57,5 +57,6 @@
 }
 
-static void dispDrawSlice(Game *game, int x, int height, int type, int tx)
+////////////////////////////////////////////////////////////////
+static void dispDrawSlice( int x, int height, int type, int tx )
 {
     // Ceiling
@@ -90,8 +91,9 @@
 }
 
-static float dispRaycast(Game *game, int *type, float *tx, float angle)
-{
-    float x = game->player.x;
-    float y = game->player.y;
+///////////////////////////////////////////////////////////////////////
+static float dispRaycast( int *type, float *tx, float angle )
+{
+    float x = game.player.x;
+    float y = game.player.y;
 
     // Camera is inside a block.
@@ -162,4 +164,5 @@
 }
 
+////////////////////////////////////////////////////////////////
 static void dispTranspose(unsigned char *buf, unsigned int size)
 {
@@ -179,5 +182,10 @@
 }
 
-static unsigned char *dispLoadTexture(char *path)
+////////////////////////
+// Exported functions
+////////////////////////
+
+//////////////////////////////////////////
+unsigned char* dispLoadTexture(char *path)
 {
     int fd;
@@ -186,5 +194,6 @@
     tex = malloc(TEX_SIZE * TEX_SIZE);
     fd = giet_fat_open(path, O_RDONLY);
-    if (fd < 0) {
+    if (fd < 0) 
+    {
         free(tex);
         return NULL;
@@ -196,97 +205,51 @@
     dispTranspose(tex, TEX_SIZE);
 
-    giet_tty_printf("[RAYCAST] loaded tex %s\n", path);
-
     return tex;
 }
 
-// Exported functions
-
-void dispInit()
-{
-    unsigned int w, h, p;
-
-    // Initialize lock
-    giet_procs_number(&w, &h, &p);
-    sqt_lock_init(&slice_x_lock, w, h, p);
-
-    // Allocate framebuffer
-    buf[0] = malloc(FBUF_X_SIZE * FBUF_Y_SIZE);
-    buf[1] = malloc(FBUF_X_SIZE * FBUF_Y_SIZE);
-    sts[0] = malloc(64);
-    sts[1] = malloc(64);
-
-    // Initialize framebuffer
-    giet_fbf_cma_alloc();
-    giet_fbf_cma_init_buf(buf[0], buf[1], sts[0], sts[1]);
-    giet_fbf_cma_start(FBUF_X_SIZE * FBUF_Y_SIZE);
-
-    // Load textures
-    g_tex[1] = dispLoadTexture("misc/rock_32.raw");
-    g_tex[2] = dispLoadTexture("misc/door_32.raw");
-    g_tex[3] = dispLoadTexture("misc/handle_32.raw");
-    g_tex[4] = dispLoadTexture("misc/wood_32.raw");
-
-    cur_buf = 0;
-    slice_cnt = 0;
-    slice_x = FBUF_X_SIZE;
-}
-
-int dispRenderSlice(Game *game)
-{
-    unsigned int x;
+///////////////////////////////////////////////////
+unsigned int dispRenderSlice( unsigned int* slice )
+{
+    // return 0 when there is no more slice to do
+
     int type;
     float angle, dist, tx;
-
-    sqt_lock_acquire(&slice_x_lock);
-
-    if (slice_x == FBUF_X_SIZE) {
-        // No more work to do for this frame
-        sqt_lock_release(&slice_x_lock);
+    unsigned int x;
+
+    // get a slice index 
+    sqt_lock_acquire( &slice_get_lock );
+
+    if (slice_x >= FBUF_X_SIZE)      // No more work to do for this frame
+    {
+        sqt_lock_release( &slice_get_lock );
         return 0;
     }
-    else {
-        // Keep slice coordinate
-        x = slice_x++;
-    }
-
-    sqt_lock_release(&slice_x_lock);
-
-    angle = game->player.dir - FIELD_OF_VIEW / 2.f +
+    else                             // Keep slice coordinate
+    {
+        x       = slice_x;
+        slice_x = x + 1;
+        sqt_lock_release( &slice_get_lock );
+        *slice  = x;
+    }
+
+    // Cast a ray to get wall distance
+    angle = game.player.dir - FIELD_OF_VIEW / 2.f +
             x * FIELD_OF_VIEW / FBUF_X_SIZE;
 
-    // Cast a ray to get wall distance
-    dist = dispRaycast(game, &type, &tx, angle);
-
-    // Perspective correction
-    dist *= cos(game->player.dir - angle);
+    dist = dispRaycast(&type, &tx, angle);
+
+    dist *= cos(game.player.dir - angle);
 
     // Draw ceiling, wall and floor
-    dispDrawSlice(game, x, FBUF_Y_SIZE / dist, type, tx * TEX_SIZE);
+    dispDrawSlice(x, FBUF_Y_SIZE / dist, type, tx * TEX_SIZE);
 
     // Signal this slice is done
-    atomic_increment((unsigned int*)&slice_cnt, 1);
+
+    sqt_lock_acquire( &slice_done_lock );
+    slice_count++;
+    sqt_lock_release( &slice_done_lock );
 
     return 1;
-}
-
-void dispRender(Game *game)
-{
-    int start = giet_proctime();
-
-    // Start rendering
-    slice_cnt = 0;
-    slice_x = 0;
-
-    // Render slices
-    while (dispRenderSlice(game));
-
-    // Wait for completion
-    while (slice_cnt != FBUF_X_SIZE);
-
-    // Flip framebuffer
-    giet_fbf_cma_display(cur_buf);
-    cur_buf = !cur_buf;
-    giet_tty_printf("[RAYCAST] flip (took %d cycles)\n", giet_proctime() - start);
-}
-
+}  // end dispRenderSlice()
+
+
Index: soft/giet_vm/applications/raycast/disp.h
===================================================================
--- soft/giet_vm/applications/raycast/disp.h	(revision 707)
+++ soft/giet_vm/applications/raycast/disp.h	(revision 708)
@@ -4,7 +4,6 @@
 #include "game.h"
 
-void dispInit();
-int dispRenderSlice(Game *game);
-void dispRender(Game *game);
+unsigned char* dispLoadTexture( char* path );
+unsigned int   dispRenderSlice( unsigned int* slice );
 
 #endif // __DISP_H
Index: soft/giet_vm/applications/raycast/game.c
===================================================================
--- soft/giet_vm/applications/raycast/game.c	(revision 707)
+++ soft/giet_vm/applications/raycast/game.c	(revision 708)
@@ -3,6 +3,9 @@
 #include "ctrl.h"
 #include <math.h>
-
-// Globals
+#include <stdio.h>
+
+//////////////////////////////
+// Game Maps
+//////////////////////////////
 
 static Map map[] =
@@ -70,58 +73,67 @@
 };
 
-static Game game =
-{
-    .mapId = 0
-};
-
-static bool g_exit;
-
+////////////////////////////
+// extern variables
+////////////////////////////
+
+extern Game         game;
+
+//////////////////////
 // Local functions
-
+//////////////////////
+
+////////////////////////////////////
 static void gameOnBlockHit(int type)
 {
-    g_exit = true;
-}
-
+    giet_tty_printf("\n[RAYCAST] Fatal collision, killed...\n");
+    game.exit = 1;
+}
+
+///////////////////////////////////////////////
 static void gameCollision(float opx, float opy)
 {
-    static bool collided_x = false;
-    static bool collided_y = false;
-
-    float px = game.player.x;
-    float py = game.player.y;
-    int fpx = floor(px);
-    int fpy = floor(py);
-    bool colliding_x = false;
-    bool colliding_y = false;
-    int collide_type_x = 0;
-    int collide_type_y = 0;
-    int type;
+    static unsigned int collided_x = 0;
+    static unsigned int collided_y = 0;
+
+    float               px = game.player.x;
+    float               py = game.player.y;
+    int                 fpx = floor(px);
+    int                 fpy = floor(py);
+    unsigned int        colliding_x = 0;
+    unsigned int        colliding_y = 0;
+    int                 collide_type_x = 0;
+    int                 collide_type_y = 0;
+    int                 type;
 
     // Check for x axis collisions
-    if      ((type = gameLocate(floor(px + COLLIDE_GAP), fpy))) {
-        colliding_x = true, collide_type_x = type;
+    if      ((type = gameLocate(floor(px + COLLIDE_GAP), fpy))) 
+    {
+        colliding_x = 1, collide_type_x = type;
         game.player.x = fpx - COLLIDE_GAP + 1;
     }
-    else if ((type = gameLocate(floor(px - COLLIDE_GAP), fpy))) {
-        colliding_x = true, collide_type_x = type;
+    else if ((type = gameLocate(floor(px - COLLIDE_GAP), fpy))) 
+    {
+        colliding_x = 1, collide_type_x = type;
         game.player.x = fpx + COLLIDE_GAP;
     }
 
     // Check for y axis collisions
-    if      ((type = gameLocate(fpx, floor(py + COLLIDE_GAP)))) {
-        colliding_y = true, collide_type_y = type;
+    if      ((type = gameLocate(fpx, floor(py + COLLIDE_GAP)))) 
+    {
+        colliding_y = 1, collide_type_y = type;
         game.player.y = fpy - COLLIDE_GAP + 1;
     }
-    else if ((type = gameLocate(fpx, floor(py - COLLIDE_GAP)))) {
-        colliding_y = true, collide_type_y = type;
+    else if ((type = gameLocate(fpx, floor(py - COLLIDE_GAP)))) 
+    {
+        colliding_y = 1, collide_type_y = type;
         game.player.y = fpy + COLLIDE_GAP;
     }
 
     // Check if we're inside a wall
-    if ((type = gameLocate(fpx, fpy))) {
-        colliding_x = true, collide_type_x = type;
-        colliding_y = true, collide_type_y = type;
-        game.player.x = opx, game.player.y = opy;
+    if ((type = gameLocate(fpx, fpy))) 
+    {
+        colliding_x   = 1   , collide_type_x = type;
+        colliding_y   = 1   , collide_type_y = type;
+        game.player.x = opx , game.player.y = opy;
     }
 
@@ -136,28 +148,122 @@
 }
 
-static void gameLogic()
-{
-    float opx = game.player.x;
-    float opy = game.player.y;
-
-    ctrlLogic(&game);
-    gameCollision(opx, opy);
-}
-
-static void gameInitMap()
-{
-    game.map = &map[game.mapId];
-    game.player.x = game.map->startX;
-    game.player.y = game.map->startY;
+/////////////////////////
+// Exported functions
+/////////////////////////
+
+///////////////
+void gameInit()
+{
+    game.mapId      = 0;
+    game.map        = &map[0];
+    game.player.x   = game.map->startX;
+    game.player.y   = game.map->startY;
     game.player.dir = game.map->startDir;
-    game.timeLeft = TIME_TOTAL;
-}
-
-// Exported functions
-
+    game.timeLeft   = TIME_TOTAL;
+    game.exit       = 0;
+}
+
+//////////////////
+void gameControl()
+{
+    // Only six commands are accepted:
+    // - key Q         : quit game
+    // - key UP        : forward move
+    // - key DOWN      : backward move
+    // - key RIGHT     : right move
+    // - key LEFT      : left move
+    // - any other key : continue
+    // several moves can be made before continue
+
+    char c;
+    unsigned int state = 0;
+    unsigned int done  = 0;
+
+    // display prompt
+    giet_tty_printf("\n[RAYCAST] > " );
+
+    // get one command
+    while ( done == 0 )
+    {
+        // get one character        
+        giet_tty_getc( &c );
+
+        if (state == 0) // first character
+        {
+            if      (c == 0x1B)         // ESCAPE : possible player move  
+            {
+                state = 1;
+            }
+            else if (c == 0x71)         // quit game
+            {
+                game.exit = 1;
+                done = 1;
+                giet_tty_printf("QUIT\n");
+            }
+            else                        // continue 
+            {
+                done = 1;
+                giet_tty_printf("\n");
+            }
+        } 
+        else if (state == 1) // previous character was ESCAPE
+        {
+            if      (c == 0x5B)        // BRAKET : possible player move
+            {
+                state = 2;
+            }
+            else                       // continue 
+            {
+                done = 1;
+                giet_tty_printf("\n");
+            }
+        }
+        else  // previous characters were ESCAPE,BRACKET
+        {
+            if      (c == 0x41)        // UP arrow <=> forward move
+            {
+                game.player.x += PLAYER_MOVE * cos(game.player.dir);
+                game.player.y += PLAYER_MOVE * sin(game.player.dir);
+                giet_tty_printf("GO ");
+                state = 0;
+            }
+            else if (c == 0x42)        // DOWN arrow <=> backward move
+            {
+                game.player.x -= PLAYER_MOVE * cos(game.player.dir);
+                game.player.y -= PLAYER_MOVE * sin(game.player.dir);
+                giet_tty_printf("BACK ");
+                state = 0;
+            }
+            else if (c == 0x43)        // RIGHT arrow <=> turn right     
+            {
+                game.player.dir += PLAYER_ROT;
+                giet_tty_printf("RIGHT ");
+                state = 0;
+            }
+            else if (c == 0x44)        // LEFT arrow <=> turn left      
+            {
+                game.player.dir -= PLAYER_ROT;
+                giet_tty_printf("LEFT ");
+                state = 0;
+            }
+            else                       // continue 
+            {
+                done = 1;
+                giet_tty_printf("\n");
+            }
+        }
+    }  // end while
+    
+    // check new position
+    int hit = gameLocate( game.player.x , game.player.y );
+    if ( hit )  gameOnBlockHit( hit );
+}
+
+////////////////////////////
 int gameLocate(int x, int y)
 {
     if ((x < 0 || x >= game.map->w) ||
-        (y < 0 || y >= game.map->h)) {
+        (y < 0 || y >= game.map->h)) 
+    {
         // Outside the map bounds
         return 1;
@@ -167,37 +273,13 @@
 }
 
-void gameRun()
-{
-    gameInitMap();
-
-    g_exit = false;
-
-    // Game loop
-    while (!g_exit) {
-        gameLogic();
-        dispRender(&game);
-    }
-
-    if (game.timeLeft == 0) {
-        // Time's up!
-        game.mapId = 0;
-    }
-    else {
-        // Go to next map
-        game.mapId++;
-    }
-}
-
+///////////////
 void gameTick()
 {
     game.timeLeft--;
 
-    if (game.timeLeft == 0) {
-        g_exit = true;
-    }
-}
-
-Game *gameInstance()
-{
-    return &game;
-}
+    if (game.timeLeft == 0) 
+    {
+        game.exit = 1;
+    }
+}
+
Index: soft/giet_vm/applications/raycast/game.h
===================================================================
--- soft/giet_vm/applications/raycast/game.h	(revision 707)
+++ soft/giet_vm/applications/raycast/game.h	(revision 708)
@@ -2,6 +2,4 @@
 #define __GAME_H
 
-#include <stdint.h>
-#include <stdbool.h>
 
 #define M_PI            (3.14159f)
@@ -11,5 +9,5 @@
 #define TIME_TOTAL      (30)
 
-typedef struct
+typedef struct 
 {
     float x;
@@ -20,24 +18,26 @@
 typedef struct
 {
-    uint8_t tile[10][10];
-    uint8_t w;
-    uint8_t h;
-    float startX;
-    float startY;
-    float startDir;
+    char    tile[10][10];
+    char    w;
+    char    h;
+    float   startX;
+    float   startY;
+    float   startDir;
 } Map;
 
 typedef struct
-{
-    Player player;
-    Map *map;
-    int mapId;
-    int timeLeft;
+{   
+    Player   player;
+    Map      *map;
+    int      mapId;
+    int      timeLeft;
+    int      exit;        // exit requested
 } Game;
 
-int gameLocate(int x, int y);
-void gameRun();
+
+void gameInit();
+int  gameLocate(int x, int y);
+void gameControl();
 void gameTick();
-Game *gameInstance();
 
 #endif // __GAME_H
Index: soft/giet_vm/applications/raycast/raycast.c
===================================================================
--- soft/giet_vm/applications/raycast/raycast.c	(revision 708)
+++ soft/giet_vm/applications/raycast/raycast.c	(revision 708)
@@ -0,0 +1,150 @@
+#include "game.h"
+#include "disp.h"
+#include <stdio.h>
+#include <malloc.h>
+#include <user_lock.h>
+
+///////////////////////
+// Global variables
+///////////////////////
+
+unsigned char*           buf[2];             // one image per buffer
+void *                   sts[2];             // for fbf_cma
+unsigned int             cur_buf;            // current buffer
+volatile unsigned int    slice_x;            // slice index (shared)
+volatile unsigned int    slice_count;        // slice count (shared)
+sqt_lock_t               slice_get_lock;     // slice index lock
+sqt_lock_t               slice_done_lock;    // slice index lock
+pthread_t                trdid[1024];        // thread identifiers array
+Game                     game;               // Game state
+
+// Textures
+unsigned char*  g_tex[] =
+{
+    NULL, // 0
+    NULL, // rock
+    NULL, // door
+    NULL, // handle
+    NULL, // wood
+};
+
+//////////////////////////
+// Exported functions
+//////////////////////////
+
+//////////////////////////////////////////
+__attribute__((constructor)) void render()
+{
+    unsigned int slice;
+
+    while ( game.exit == 0 ) 
+    {
+        dispRenderSlice( &slice );
+    }
+    giet_pthread_exit( NULL );
+}
+
+////////////////////////////////////////
+__attribute__((constructor)) void main()
+{
+    unsigned int w, h, p;   // platform parameters
+ 
+    unsigned int i, j, n;   // indexes for loops
+
+    giet_tty_alloc(0);
+
+    giet_tty_printf("\n[RAYCAST] enters main at cycle %d\n", 
+                    giet_proctime() );
+
+    // get and check platform parameters
+    giet_procs_number(&w, &h, &p);
+
+    giet_pthread_assert( (w<=16) , "[RAYCAST ERROR] check hardware config" );
+    giet_pthread_assert( (h<=16) , "[RAYCAST ERROR] check hardware config" );
+    giet_pthread_assert( (p<= 4) , "[RAYCAST ERROR] check hardware config" );
+
+    // compute total number of threads
+    unsigned int nthreads = w * h * p;
+
+    // Initialize heap for each cluster
+    for (i = 0; i < w; i++)
+        for (j = 0; j < h; j++)
+            heap_init(i, j);
+
+    // Initialize lock protecting slice allocator 
+    sqt_lock_init( &slice_get_lock,  w , h , p );
+    sqt_lock_init( &slice_done_lock, w , h , p );
+
+    // Allocate buffers
+    buf[0] = malloc(FBUF_X_SIZE * FBUF_Y_SIZE);
+    buf[1] = malloc(FBUF_X_SIZE * FBUF_Y_SIZE);
+    sts[0] = malloc(64);
+    sts[1] = malloc(64);
+
+    // Initialize frame buffer and start Chained buffer DMA
+    giet_fbf_cma_alloc();
+    giet_fbf_cma_init_buf(buf[0], buf[1], sts[0], sts[1]);
+    giet_fbf_cma_start(FBUF_X_SIZE * FBUF_Y_SIZE);
+    cur_buf = 0;
+
+    // Load textures
+    g_tex[1] = dispLoadTexture("misc/rock_32.raw");
+    g_tex[2] = dispLoadTexture("misc/door_32.raw");
+    g_tex[3] = dispLoadTexture("misc/handle_32.raw");
+    g_tex[4] = dispLoadTexture("misc/wood_32.raw");
+
+    // Initialise game
+    gameInit();
+
+    giet_tty_printf("\n[RAYCAST] initialisation completed at cycle %d\n",
+                    giet_proctime() );
+
+    // launch other threads to speed render
+    for ( n = 1 ; n < nthreads ; n++ )
+    {
+        if ( giet_pthread_create( &trdid[n],
+                                  NULL,                  // no attribute
+                                  &render,
+                                  NULL ) )               // no argument
+        {
+            giet_tty_printf("\n[RAYCAST ERROR] creating thread %d\n", n );
+            giet_pthread_exit( NULL );
+        }
+    }
+
+
+
+    // Game main loop : display one frame 
+    // and get one player move per iteration
+    while ( game.exit == 0 ) 
+    {
+        // initialise synchronisation variables
+        // this actually allows the render threads to make useful work
+        slice_count = 0;
+        slice_x     = 0;
+
+        // contribute to build current buffer
+        unsigned int slice;
+        while ( dispRenderSlice( &slice ) );
+/*
+        unsigned int again;
+        do
+        {
+            again = dispRenderSlice( &slice );
+        }
+        while ( again );
+*/
+        // Wait last slice completion
+        while (slice_count < FBUF_X_SIZE)  giet_tty_printf(" ");
+
+        // Flip buffer
+        giet_fbf_cma_display(cur_buf);
+        cur_buf = 1 - cur_buf;
+
+        // get new player position [x,y,dir]
+        gameControl();
+    }
+
+    giet_pthread_exit( NULL );
+
+}  // end main()
Index: soft/giet_vm/applications/raycast/raycast.py
===================================================================
--- soft/giet_vm/applications/raycast/raycast.py	(revision 707)
+++ soft/giet_vm/applications/raycast/raycast.py	(revision 708)
@@ -10,7 +10,7 @@
 #  This file describes the mapping of the multi-threaded "raycast" application 
 #  on a multi-clusters, multi-processors architecture.
-#  The mapping of tasks on processors is the following:
-#    - one "main" task on (0,0,0)
-#    - one "render" task per processor but (0,0,0)
+#  The mapping of threads on processors is the following:
+#    - one "main" thread on (0,0,0)
+#    - one "render" thread per processor but (0,0,0)
 #  The mapping of virtual segments is the following:
 #    - There is one shared data vseg in cluster[0][0]
@@ -89,7 +89,7 @@
                 mapping.addVseg( vspace, 'raycast_heap_%d_%d' %(x,y), base , size, 
                                  'C_WU', vtype = 'HEAP', x = x, y = y, pseg = 'RAM', 
-                                 local = False )
+                                 local = False, big = True )
 
-    # distributed tasks / one task per processor
+    # distributed threads / one thread per processor
     for x in xrange (x_size):
         for y in xrange (y_size):
@@ -98,15 +98,18 @@
                 for p in xrange( nprocs ):
                     trdid = (((x * y_size) + y) * nprocs) + p
-                    if  ( x == 0 and y == 0 and p == 0 ):       # main task
-                        task_index = 1
-                        task_name  = 'main_%d_%d_%d' %(x,y,p)
-                    else:                                       # render task
-                        task_index = 0
-                        task_name  = 'render_%d_%d_%d' % (x,y,p)
+                    if  ( x == 0 and y == 0 and p == 0 ):       # main thread
+                        start_id = 1
+                        is_main  = True
+                    else:                                       # render thread
+                        start_id = 0
+                        is_main  = False
 
-                    mapping.addTask( vspace, task_name, trdid, x, y, p,
-                                     'raycast_stack_%d_%d_%d' % (x,y,p), 
-                                     'raycast_heap_%d_%d' % (x,y),
-                                     task_index )
+                    mapping.addThread( vspace, 
+                                       'raycast_%d_%d_%d' % (x,y,p),
+                                       is_main,
+                                       x, y, p,
+                                       'raycast_stack_%d_%d_%d' % (x,y,p), 
+                                       'raycast_heap_%d_%d' % (x,y),
+                                       start_id )
 
     # extend mapping name
Index: soft/giet_vm/applications/shell/Makefile
===================================================================
--- soft/giet_vm/applications/shell/Makefile	(revision 707)
+++ soft/giet_vm/applications/shell/Makefile	(revision 708)
@@ -2,5 +2,5 @@
 APP_NAME = shell
 
-OBJS = main.o
+OBJS = shell.o
 
 LIBS = -L../../build/libs -luser
Index: soft/giet_vm/applications/shell/main.c
===================================================================
--- soft/giet_vm/applications/shell/main.c	(revision 707)
+++ 	(revision )
@@ -1,407 +1,0 @@
-///////////////////////////////////////////////////////////////////////////////////////
-// File   : main.c   (for shell application)
-// Date   : july 2015
-// author : ClÃ©ment GuÃ©rin
-///////////////////////////////////////////////////////////////////////////////////////
-// Simple shell for GIET_VM.
-///////////////////////////////////////////////////////////////////////////////////////
-
-#include "stdio.h"
-#include "stdlib.h"
-#include "malloc.h"
-
-#define BUF_SIZE    (256)
-#define MAX_ARGS    (32)
-
-struct command_t
-{
-    char *name;
-    void (*fn)(int, char**);
-};
-
-////////////////////////////////////////////////////////////////////////////////
-//  Shell  Commands
-////////////////////////////////////////////////////////////////////////////////
-
-struct command_t cmd[];
-
-///////////////////////////////////////////
-static void cmd_help(int argc, char** argv)
-{
-    int i;
-
-    giet_tty_printf("available commands:\n");
-
-    for (i = 0; cmd[i].name; i++)
-    {
-        giet_tty_printf("\t%s\n", cmd[i].name);
-    }
-}
-
-///////////////////////////////////////////////
-static void cmd_proctime(int argc, char** argv)
-{
-    giet_tty_printf("%u\n", giet_proctime());
-}
-
-/////////////////////////////////////////
-static void cmd_ls(int argc, char** argv)
-{
-    int fd;
-    fat_dirent_t entry;
-
-    if (argc < 2)
-        fd = giet_fat_opendir("/");
-    else
-        fd = giet_fat_opendir(argv[1]);
-
-    if (fd < 0)
-    {
-        giet_tty_printf("can't list directory (err=%d)\n", fd);
-        return;
-    }
-
-    while (giet_fat_readdir(fd, &entry) == 0)
-    {
-        if (entry.is_dir)
-            giet_tty_printf("dir ");
-        else
-            giet_tty_printf("file");
-
-        giet_tty_printf(" | size = %d \t| cluster = %X \t| %s\n",
-                        entry.size, entry.cluster, entry.name );
-    }
-
-    giet_fat_closedir(fd);
-}
-
-////////////////////////////////////////////
-static void cmd_mkdir(int argc, char** argv)
-{
-    if (argc < 2)
-    {
-        giet_tty_printf("%s <path>\n", argv[0]);
-        return;
-    }
-
-    int ret = giet_fat_mkdir(argv[1]);
-    if (ret < 0)
-    {
-        giet_tty_printf("can't create directory (err=%d)\n", ret);
-    }
-}
-
-/////////////////////////////////////////
-static void cmd_cp(int argc, char** argv)
-{
-    if (argc < 3)
-    {
-        giet_tty_printf("%s <src> <dst>\n", argv[0]);
-        return;
-    }
-
-    char buf[1024];
-    int src_fd = -1;
-    int dst_fd = -1;
-    fat_file_info_t info;
-    int size;
-    int i;
-
-    src_fd = giet_fat_open( argv[1] , O_RDONLY );
-    if (src_fd < 0)
-    {
-        giet_tty_printf("can't open %s (err=%d)\n", argv[1], src_fd);
-        goto exit;
-    }
-
-    giet_fat_file_info(src_fd, &info);
-    if (info.is_dir)
-    {
-        giet_tty_printf("can't copy a directory\n");
-        goto exit;
-    }
-    size = info.size;
-
-    dst_fd = giet_fat_open( argv[2] , O_CREATE | O_TRUNC );
-    if (dst_fd < 0)
-    {
-        giet_tty_printf("can't open %s (err=%d)\n", argv[2], dst_fd);
-        goto exit;
-    }
-
-    giet_fat_file_info(dst_fd, &info);
-    if (info.is_dir)
-    {
-        giet_tty_printf("can't copy to a directory\n"); // TODO
-        goto exit;
-    }
-
-    i = 0;
-    while (i < size)
-    {
-        int len = (size - i < 1024 ? size - i : 1024);
-        int wlen;
-
-        giet_tty_printf("\rwrite %d/%d (%d%%)", i, size, 100*i/size);
-
-        len = giet_fat_read(src_fd, &buf, len);
-        wlen = giet_fat_write(dst_fd, &buf, len);
-        if (wlen != len)
-        {
-            giet_tty_printf("\nwrite error\n");
-            goto exit;
-        }
-        i += len;
-    }
-    giet_tty_printf("\n");
-
-exit:
-    if (src_fd >= 0)
-        giet_fat_close(src_fd);
-    if (dst_fd >= 0)
-        giet_fat_close(dst_fd);
-}
-
-/////////////////////////////////////////
-static void cmd_rm(int argc, char **argv)
-{
-    if (argc < 2)
-    {
-        giet_tty_printf("%s <file>\n", argv[0]);
-        return;
-    }
-
-    int ret = giet_fat_remove(argv[1], 0);
-    if (ret < 0)
-    {
-        giet_tty_printf("can't remove %s (err=%d)\n", argv[1], ret);
-    }
-}
-
-////////////////////////////////////////////
-static void cmd_rmdir(int argc, char **argv)
-{
-    if (argc < 2)
-    {
-        giet_tty_printf("%s <path>\n", argv[0]);
-        return;
-    }
-
-    int ret = giet_fat_remove(argv[1], 1);
-    if (ret < 0)
-    {
-        giet_tty_printf("can't remove %s (err=%d)\n", argv[1], ret);
-    }
-}
-
-/////////////////////////////////////////
-static void cmd_mv(int argc, char **argv)
-{
-    if (argc < 3)
-    {
-        giet_tty_printf("%s <src> <dst>\n", argv[0]);
-        return;
-    }
-
-    int ret = giet_fat_rename(argv[1], argv[2]);
-    if (ret < 0)
-    {
-        giet_tty_printf("can't move %s to %s (err=%d)\n", argv[1], argv[2], ret);
-    }
-}
-
-///////////////////////////////////////////
-static void cmd_exec(int argc, char **argv)
-{
-    if (argc < 2)
-    {
-        giet_tty_printf("%s <pathname>\n", argv[0]);
-        return;
-    }
-
-    int ret = giet_exec_application(argv[1]);
-    if ( ret == -1 )
-    {
-        giet_tty_printf("\n  error : %s not found\n", argv[1] );
-    }
-}
-
-///////////////////////////////////////////
-static void cmd_kill(int argc, char **argv)
-{
-    if (argc < 2)
-    {
-        giet_tty_printf("%s <pathname>\n", argv[0]);
-        return;
-    }
-
-    int ret = giet_kill_application(argv[1]);
-    if ( ret == -1 )
-    {
-        giet_tty_printf("\n  error : %s not found\n", argv[1] );
-    }
-    if ( ret == -2 )
-    {
-        giet_tty_printf("\n  error : %s cannot be killed\n", argv[1] );
-    }
-}
-
-///////////////////////////////////////////////
-static void cmd_ps(int argc, char** argv)
-{
-    giet_tasks_status();
-}
-
-///////////////////////////////////////////////
-static void cmd_sleep(int argc, char** argv)
-{
-    int start = giet_proctime();
-
-    if (argc < 2)
-    {
-        giet_tty_printf("%s <cycles>\n", argv[0]);
-        return;
-    }
-
-    while (giet_proctime() < start + atoi(argv[1]));
-}
-
-////////////////////////////////////////////////////////////////////
-struct command_t cmd[] =
-{
-    { "help",       cmd_help },
-    { "proctime",   cmd_proctime },
-    { "ls",         cmd_ls },
-    { "mkdir",      cmd_mkdir },
-    { "cp",         cmd_cp },
-    { "rm",         cmd_rm },
-    { "rmdir",      cmd_rmdir },
-    { "mv",         cmd_mv },
-    { "exec",       cmd_exec },
-    { "kill",       cmd_kill },
-    { "ps",         cmd_ps },
-    { "sleep",      cmd_sleep },
-    { NULL,         NULL }
-};
-
-// shell
-
-///////////////////////////////////////
-static void parse(char *buf)
-{
-    int argc = 0;
-    char* argv[MAX_ARGS];
-    int i;
-    int len = strlen(buf);
-
-    // build argc/argv
-    for (i = 0; i < len; i++)
-    {
-        if (buf[i] == ' ')
-        {
-            buf[i] = '\0';
-        }
-        else if (i == 0 || buf[i - 1] == '\0')
-        {
-            if (argc < MAX_ARGS)
-            {
-                argv[argc] = &buf[i];
-                argc++;
-            }
-        }
-    }
-
-    if (argc > 0)
-    {
-        int found = 0;
-
-        // try to match typed command with built-ins
-        for (i = 0; cmd[i].name; i++)
-        {
-            if (strcmp(argv[0], cmd[i].name) == 0)
-            {
-                // invoke
-                cmd[i].fn(argc, argv);
-                found = 1;
-                break;
-            }
-        }
-
-        if (!found)
-        {
-            giet_tty_printf("undefined command %s\n", argv[0]);
-        }
-    }
-}
-
-////////////////////
-static void prompt()
-{
-    giet_tty_printf("# ");
-}
-
-//////////////////////////////////////////
-__attribute__ ((constructor)) void main()
-//////////////////////////////////////////
-{
-    char c;
-    char buf[BUF_SIZE];
-    int count = 0;
-
-    // get a private TTY
-    giet_tty_alloc( 0 );
-    giet_tty_printf( "~~~ shell ~~~\n" );
-
-    // display first prompt
-    prompt();
-
-    while (1)
-    {
-        giet_tty_getc(&c);
-
-        switch (c)
-        {
-        case '\b':      // backspace
-            if (count > 0)
-            {
-                giet_tty_printf("\b \b");
-                count--;
-            }
-            break;
-        case '\n':      // new line
-            giet_tty_printf("\n");
-            if (count > 0)
-            {
-                buf[count] = '\0';
-                parse((char*)&buf);
-            }
-            prompt();
-            count = 0;
-            break;
-        case '\t':      // tabulation
-            // do nothing
-            break;
-        case '\03':     // ^C
-            giet_tty_printf("^C\n");
-            prompt();
-            count = 0;
-            break;
-        default:        // regular character
-            if (count < sizeof(buf) - 1)
-            {
-                giet_tty_printf("%c", c);
-                buf[count] = c;
-                count++;
-            }
-        }
-    }
-} // end main()
-
-// Local Variables:
-// tab-width: 4
-// c-basic-offset: 4
-// c-file-offsets:((innamespace . 0)(inline-open . 0))
-// indent-tabs-mode: nil
-// End:
-// vim: filetype=c:expandtab:shiftwidth=4:tabstop=4:softtabstop=4
-
Index: soft/giet_vm/applications/shell/shell.c
===================================================================
--- soft/giet_vm/applications/shell/shell.c	(revision 708)
+++ soft/giet_vm/applications/shell/shell.c	(revision 708)
@@ -0,0 +1,469 @@
+///////////////////////////////////////////////////////////////////////////////////////
+// File   : shell.c   
+// Date   : july 2015
+// author : ClÃ©ment GuÃ©rin
+///////////////////////////////////////////////////////////////////////////////////////
+// Simple shell for GIET_VM.
+///////////////////////////////////////////////////////////////////////////////////////
+
+#include "stdio.h"
+#include "stdlib.h"
+#include "malloc.h"
+
+#define BUF_SIZE    (256)
+#define MAX_ARGS    (32)
+
+struct command_t
+{
+    char *name;
+    char *desc;
+    void (*fn)(int, char**);
+};
+
+////////////////////////////////////////////////////////////////////////////////
+//  Shell  Commands
+////////////////////////////////////////////////////////////////////////////////
+
+struct command_t cmd[];
+
+///////////////////////////////////////////
+static void cmd_help(int argc, char** argv)
+{
+    int i;
+
+    giet_tty_printf("available commands:\n");
+
+    for (i = 0; cmd[i].name; i++)
+    {
+        giet_tty_printf("\t%s\t : %s\n", cmd[i].name , cmd[i].desc );
+    }
+}
+
+///////////////////////////////////////////
+static void cmd_time(int argc, char** argv)
+{
+    giet_tty_printf("%d\n", giet_proctime());
+}
+
+/////////////////////////////////////////
+static void cmd_ls(int argc, char** argv)
+{
+    int fd;
+    fat_dirent_t entry;
+
+    if (argc < 2)
+    {
+        giet_tty_printf("  usage : %s <pathname>\n", argv[0]);
+        return;
+    }
+
+    fd = giet_fat_opendir(argv[1]);
+
+    if (fd < 0)
+    {
+        giet_tty_printf("  error : cannot open %s / err = %d)\n", argv[1], fd);
+        return;
+    }
+
+    while (giet_fat_readdir(fd, &entry) == 0)
+    {
+        if (entry.is_dir)
+            giet_tty_printf("dir ");
+        else
+            giet_tty_printf("file");
+
+        giet_tty_printf(" | size = %d \t| cluster = %X \t| %s\n",
+                        entry.size, entry.cluster, entry.name );
+    }
+
+    giet_fat_closedir(fd);
+}
+
+////////////////////////////////////////////
+static void cmd_mkdir(int argc, char** argv)
+{
+    if (argc < 2)
+    {
+        giet_tty_printf("  usage : %s <path>\n", argv[0]);
+        return;
+    }
+
+    int ret = giet_fat_mkdir(argv[1]);
+
+    if (ret < 0)
+    {
+        giet_tty_printf("  error : cannot create directory %s / err = %d\n", argv[1], ret);
+    }
+}
+
+/////////////////////////////////////////
+static void cmd_cp(int argc, char** argv)
+{
+    if (argc < 3)
+    {
+        giet_tty_printf("  usage : %s <src> <dst>\n", argv[0]);
+        return;
+    }
+
+    char buf[1024];
+    int src_fd = -1;
+    int dst_fd = -1;
+    fat_file_info_t info;
+    int size;
+    int i;
+
+    src_fd = giet_fat_open( argv[1] , O_RDONLY );
+    if (src_fd < 0)
+    {
+        giet_tty_printf("  error : cannot open %s / err = %d\n", argv[1], src_fd);
+        goto exit;
+    }
+
+    giet_fat_file_info(src_fd, &info);
+
+    if (info.is_dir)
+    {
+        giet_tty_printf("  error : %s is a directory\n", argv[1] );
+        goto exit;
+    }
+
+    size = info.size;
+
+    dst_fd = giet_fat_open( argv[2] , O_CREATE | O_TRUNC );
+
+    if (dst_fd < 0)
+    {
+        giet_tty_printf("  error : cannot open %s / err = %d\n", argv[2], dst_fd);
+        goto exit;
+    }
+
+    giet_fat_file_info(dst_fd, &info);
+
+    if (info.is_dir)
+    {
+        giet_tty_printf("error : %s is a directory\n", argv[2] );  // TODO
+        goto exit;
+    }
+
+    i = 0;
+    while (i < size)
+    {
+        int len = (size - i < 1024 ? size - i : 1024);
+        int wlen;
+
+        giet_tty_printf("\rwrite %d/%d (%d%%)", i, size, 100*i/size);
+
+        len = giet_fat_read(src_fd, &buf, len);
+        wlen = giet_fat_write(dst_fd, &buf, len);
+        if (wlen != len)
+        {
+            giet_tty_printf("  error : cannot write on device\n");
+            goto exit;
+        }
+        i += len;
+    }
+    giet_tty_printf("\n");
+
+exit:
+    if (src_fd >= 0)
+        giet_fat_close(src_fd);
+    if (dst_fd >= 0)
+        giet_fat_close(dst_fd);
+}
+
+/////////////////////////////////////////
+static void cmd_rm(int argc, char **argv)
+{
+    if (argc < 2)
+    {
+        giet_tty_printf("  usage : %s <file>\n", argv[0]);
+        return;
+    }
+
+    int ret = giet_fat_remove(argv[1], 0);
+
+    if (ret < 0)
+    {
+        giet_tty_printf("  error : cannot remove %s / err = %d\n", argv[1], ret );
+    }
+}
+
+////////////////////////////////////////////
+static void cmd_rmdir(int argc, char **argv)
+{
+    if (argc < 2)
+    {
+        giet_tty_printf("  usage : %s <pathname>\n", argv[0]);
+        return;
+    }
+
+    int ret = giet_fat_remove(argv[1], 1);
+    if (ret < 0)
+    {
+        giet_tty_printf("  error : cannot remove %s / err = %d\n", argv[1], ret );
+    }
+}
+
+/////////////////////////////////////////
+static void cmd_mv(int argc, char **argv)
+{
+    if (argc < 3)
+    {
+        giet_tty_printf("  usage : %s <src> <dst>\n", argv[0]);
+        return;
+    }
+
+    int ret = giet_fat_rename(argv[1], argv[2]);
+    if (ret < 0)
+    {
+        giet_tty_printf("error : cannot move %s to %s / err = %d\n", argv[1], argv[2], ret );
+    }
+}
+
+///////////////////////////////////////////
+static void cmd_exec(int argc, char **argv)
+{
+    if (argc < 2)
+    {
+        giet_tty_printf("  usage : %s <vspace_name>\n", argv[0]);
+        return;
+    }
+
+    int ret = giet_exec_application(argv[1]);
+    if ( ret == -1 )
+    {
+        giet_tty_printf("  error : %s not found\n", argv[1] );
+    }
+}
+
+///////////////////////////////////////////
+static void cmd_kill(int argc, char **argv)
+{
+    if (argc < 2)
+    {
+        giet_tty_printf("  usage : %s <vspace_name>\n", argv[0]);
+        return;
+    }
+
+    int ret = giet_kill_application(argv[1]);
+    if ( ret == -1 )
+    {
+        giet_tty_printf("  error : %s not found\n", argv[1] );
+    }
+    if ( ret == -2 )
+    {
+        giet_tty_printf("  error : %s cannot be killed\n", argv[1] );
+    }
+}
+
+/////////////////////////////////////////
+static void cmd_ps(int argc, char** argv)
+{
+    giet_applications_status();
+}
+
+////////////////////////////////////////////
+static void cmd_pause(int argc, char** argv)
+{
+    if (argc < 3)
+    {
+        giet_tty_printf("  usage : %s <vspace_name> <thread_name>\n", argv[0] );
+        return;
+    }
+
+    int ret = giet_pthread_pause( argv[1] , argv[2] );
+
+    if ( ret == -1 )
+    {
+        giet_tty_printf("  error : vspace %s not found\n", argv[1] );
+    }
+    if ( ret == -2 )
+    {
+        giet_tty_printf("  error : thread %s not found\n", argv[2] );
+    }
+}
+
+/////////////////////////////////////////////
+static void cmd_resume(int argc, char** argv)
+{
+    if (argc < 3)
+    {
+        giet_tty_printf("  usage : %s <vspace_name> <thread_name>\n", argv[0] );
+        return;
+    }
+
+    int ret = giet_pthread_resume( argv[1] , argv[2] );
+
+    if ( ret == -1 )
+    {
+        giet_tty_printf("  error : vspace %s not found\n", argv[1] );
+    }
+    if ( ret == -2 )
+    {
+        giet_tty_printf("  error : thread %s not found\n", argv[2] );
+    }
+}
+
+/////////////////////////////////////////////
+static void cmd_context(int argc, char** argv)
+{
+    if (argc < 3)
+    {
+        giet_tty_printf("  usage : %s <vspace_name> <thread_name>\n", argv[0] );
+        return;
+    }
+
+    int ret = giet_pthread_context( argv[1] , argv[2] );
+
+    if ( ret == -1 )
+    {
+        giet_tty_printf("  error : vspace %s not found\n", argv[1] );
+    }
+    if ( ret == -2 )
+    {
+        giet_tty_printf("  error : thread %s not found\n", argv[2] );
+    }
+}
+
+
+////////////////////////////////////////////////////////////////////
+struct command_t cmd[] =
+{
+    { "help",       "list available commands",              cmd_help },
+    { "time",       "return current date",                  cmd_time },
+    { "ls",         "list content of a directory",          cmd_ls },
+    { "mkdir",      "create a new directory",               cmd_mkdir },
+    { "cp",         "replicate a file in file system",      cmd_cp },
+    { "rm",         "remove a file from file system",       cmd_rm },
+    { "rmdir",      "remove a directory from file system",  cmd_rmdir },
+    { "mv",         "move a file in file system",           cmd_mv },
+    { "exec",       "start an application",                 cmd_exec },
+    { "kill",       "kill an application (all threads)",    cmd_kill },
+    { "ps",         "list all mapped applications status",  cmd_ps },
+    { "pause",      "pause a thread",                       cmd_pause },
+    { "resume",     "resume a thread",                      cmd_resume },
+    { "context",    "display a thread context",             cmd_context },
+    { NULL,         NULL,                                   NULL }
+};
+
+// shell
+
+////////////////////////////
+static void parse(char *buf)
+{
+    int argc = 0;
+    char* argv[MAX_ARGS];
+    int i;
+    int len = strlen(buf);
+
+    // build argc/argv
+    for (i = 0; i < len; i++)
+    {
+        if (buf[i] == ' ')
+        {
+            buf[i] = '\0';
+        }
+        else if (i == 0 || buf[i - 1] == '\0')
+        {
+            if (argc < MAX_ARGS)
+            {
+                argv[argc] = &buf[i];
+                argc++;
+            }
+        }
+    }
+
+    if (argc > 0)
+    {
+        int found = 0;
+
+        // try to match typed command with built-ins
+        for (i = 0; cmd[i].name; i++)
+        {
+            if (strcmp(argv[0], cmd[i].name) == 0)
+            {
+                // invoke
+                cmd[i].fn(argc, argv);
+                found = 1;
+                break;
+            }
+        }
+
+        if (!found)
+        {
+            giet_tty_printf("undefined command %s\n", argv[0]);
+        }
+    }
+}
+
+////////////////////
+static void prompt()
+{
+    giet_tty_printf("# ");
+}
+
+//////////////////////////////////////////
+__attribute__ ((constructor)) void main()
+//////////////////////////////////////////
+{
+    char c;
+    char buf[BUF_SIZE];
+    int count = 0;
+
+    // get a private TTY
+    giet_tty_alloc( 0 );
+    giet_tty_printf( "~~~ shell ~~~\n\n" );
+
+    // display first prompt
+    prompt();
+
+    while (1)
+    {
+        giet_tty_getc(&c);
+
+        switch (c)
+        {
+        case '\b':      // backspace
+            if (count > 0)
+            {
+                giet_tty_printf("\b \b");
+                count--;
+            }
+            break;
+        case '\n':      // new line
+            giet_tty_printf("\n");
+            if (count > 0)
+            {
+                buf[count] = '\0';
+                parse((char*)&buf);
+            }
+            prompt();
+            count = 0;
+            break;
+        case '\t':      // tabulation
+            // do nothing
+            break;
+        case '\03':     // ^C
+            giet_tty_printf("^C\n");
+            prompt();
+            count = 0;
+            break;
+        default:        // regular character
+            if (count < sizeof(buf) - 1)
+            {
+                giet_tty_printf("%c", c);
+                buf[count] = c;
+                count++;
+            }
+        }
+    }
+} // end main()
+
+// Local Variables:
+// tab-width: 4
+// c-basic-offset: 4
+// c-file-offsets:((innamespace . 0)(inline-open . 0))
+// indent-tabs-mode: nil
+// End:
+// vim: filetype=c:expandtab:shiftwidth=4:tabstop=4:softtabstop=4
+
Index: soft/giet_vm/applications/shell/shell.py
===================================================================
--- soft/giet_vm/applications/shell/shell.py	(revision 707)
+++ soft/giet_vm/applications/shell/shell.py	(revision 708)
@@ -58,5 +58,5 @@
 
     # task 
-    mapping.addTask( vspace, 'shell', 0, 0, 0, 0, 'shell_stack', 'shell_heap', 0 )
+    mapping.addThread( vspace, 'shell', True, 0, 0, 0, 'shell_stack', 'shell_heap', 0 )
 
     # extend mapping name
Index: soft/giet_vm/applications/transpose/Makefile
===================================================================
--- soft/giet_vm/applications/transpose/Makefile	(revision 707)
+++ soft/giet_vm/applications/transpose/Makefile	(revision 708)
@@ -1,6 +1,12 @@
+
+CC = mipsel-unknown-elf-gcc
+AS = mipsel-unknown-elf-as
+LD = mipsel-unknown-elf-ld
+DU = mipsel-unknown-elf-objdump
+AR = mipsel-unknown-elf-ar
 
 APP_NAME = transpose
 
-OBJS= main.o 
+OBJS= transpose.o 
 
 LIBS= -L../../build/libs -luser
Index: soft/giet_vm/applications/transpose/main.c
===================================================================
--- soft/giet_vm/applications/transpose/main.c	(revision 707)
+++ 	(revision )
@@ -1,527 +1,0 @@
-///////////////////////////////////////////////////////////////////////////////////////
-// File   : main.c   (for transpose application)
-// Date   : february 2014
-// author : Alain Greiner
-///////////////////////////////////////////////////////////////////////////////////////
-// This multi-threaded application makes a transpose for a NN*NN pixels image.
-// It can run on a multi-processors, multi-clusters architecture, with one thread
-// per processor. 
-//
-// The image is read from a file (one byte per pixel), transposed and
-// saved in a second file. Then the transposed image is read from the second file,
-// transposed again and saved in a third file.
-//
-// The input and output buffers containing the image are distributed in all clusters.
-//
-// - The image size NN must fit the frame buffer size.
-// - The block size in block device must be 512 bytes.
-// - The number of clusters  must be a power of 2 no larger than 64.
-// - The number of processors per cluster must be a power of 2 no larger than 4.
-//
-// For each image the application makes a self test (checksum for each line).
-// The actual display on the frame buffer depends on frame buffer availability.
-///////////////////////////////////////////////////////////////////////////////////////
-
-#include "stdio.h"
-#include "user_barrier.h"
-#include "malloc.h"
-
-#define BLOCK_SIZE            512                         // block size on disk
-#define X_MAX                 8                           // max number of clusters in row
-#define Y_MAX                 8                           // max number of clusters in column
-#define PROCS_MAX             4                           // max number of procs per cluster
-#define CLUSTER_MAX           (X_MAX * Y_MAX)             // max number of clusters
-#define NN                    256                         // image size : nlines = npixels
-#define INITIAL_FILE_PATH     "/misc/lena_256.raw"        // pathname on virtual disk
-#define TRANSPOSED_FILE_PATH  "/home/lena_transposed.raw" // pathname on virtual disk
-#define RESTORED_FILE_PATH    "/home/lena_restored.raw"   // pathname on virtual disk
-#define INSTRUMENTATION_OK    1                           // display statistics on TTY 
-
-// macro to use a shared TTY
-#define printf(...)     lock_acquire( &tty_lock ); \
-                        giet_tty_printf(__VA_ARGS__);  \
-                        lock_release( &tty_lock )
-
-///////////////////////////////////////////////////////
-// global variables stored in seg_data in cluster(0,0)
-///////////////////////////////////////////////////////
-
-// instrumentation counters for each processor in each cluster 
-unsigned int LOAD_START[X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
-unsigned int LOAD_END  [X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
-unsigned int TRSP_START[X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
-unsigned int TRSP_END  [X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
-unsigned int DISP_START[X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
-unsigned int DISP_END  [X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
-unsigned int STOR_START[X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
-unsigned int STOR_END  [X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
-
-// arrays of pointers on distributed buffers
-// one input buffer & one output buffer per cluster
-unsigned char*  buf_in [CLUSTER_MAX];
-unsigned char*  buf_out[CLUSTER_MAX];
-
-// checksum variables 
-unsigned check_line_before[NN];
-unsigned check_line_after[NN];
-
-// lock protecting shared TTY
-user_lock_t  tty_lock;
-
-// global & local synchronisation variables
-giet_sqt_barrier_t barrier;
-
-volatile unsigned int global_init_ok = 0;
-volatile unsigned int local_init_ok[X_MAX][Y_MAX] = {{ 0 }};
-
-//////////////////////////////////////////
-__attribute__ ((constructor)) void main()
-//////////////////////////////////////////
-{
-    unsigned int l;                  // line index for loops
-    unsigned int p;                  // pixel index for loops
-
-    // processor identifiers
-    unsigned int x;                  // x cluster coordinate
-    unsigned int y;                  // y cluster coordinate
-    unsigned int lpid;               // local processor index
-
-    // plat-form parameters
-    unsigned int x_size;             // number of clusters in a row
-    unsigned int y_size;             // number of clusters in a column
-    unsigned int nprocs;             // number of processors per cluster
-    
-    giet_proc_xyp( &x, &y, &lpid);             
-
-    giet_procs_number( &x_size , &y_size , &nprocs );
-
-    unsigned int nclusters     = x_size * y_size;               // number of clusters
-    unsigned int ntasks        = x_size * y_size * nprocs;      // number of tasks
-    unsigned int npixels       = NN * NN;                       // pixels per image
-    unsigned int iteration     = 0;                             // iiteration iter
-    int          fd_initial    = 0;                             // initial file descriptor
-    int          fd_transposed = 0;                             // transposed file descriptor
-    int          fd_restored   = 0;                             // restored file descriptor
-    unsigned int cluster_id    = (x * y_size) + y;              // "continuous" index   
-    unsigned int task_id       = (cluster_id * nprocs) + lpid;  // "continuous" task index
-
-    // checking parameters
-    giet_assert( ((nprocs == 1) || (nprocs == 2) || (nprocs == 4)),
-                 "[TRANSPOSE ERROR] number of procs per cluster must be 1, 2 or 4");
-
-    giet_assert( ((x_size == 1) || (x_size == 2) || (x_size == 4) || 
-                  (x_size == 8) || (x_size == 16)),
-                 "[TRANSPOSE ERROR] x_size must be 1,2,4,8,16");
-
-    giet_assert( ((y_size == 1) || (y_size == 2) || (y_size == 4) || 
-                  (y_size == 8) || (y_size == 16)),
-                 "[TRANSPOSE ERROR] y_size must be 1,2,4,8,16");
-
-    giet_assert( (ntasks <= NN ),
-                 "[TRANSPOSE ERROR] number of tasks larger than number of lines");
-
-    ///////////////////////////////////////////////////////////////////////
-    // Processor [0,0,0] makes global initialisation
-    // It includes parameters checking, heap and barrier initialization.
-    // Others processors wait initialisation completion
-    ///////////////////////////////////////////////////////////////////////
-
-    if ( (x==0) && (y==0) && (lpid==0) )
-    {
-        // shared TTY allocation
-        giet_tty_alloc( 1 );
-
-        // TTY lock initialisation
-        lock_init( &tty_lock);
-      
-        // distributed heap initialisation
-        unsigned int cx , cy;
-        for ( cx = 0 ; cx < x_size ; cx++ ) 
-        {
-            for ( cy = 0 ; cy < y_size ; cy++ ) 
-            {
-                heap_init( cx , cy );
-            }
-        }
-
-        // barrier initialisation
-        sqt_barrier_init( &barrier, x_size , y_size , nprocs );
-
-        printf("\n[TRANSPOSE] Proc [0,0,0] completes initialisation at cycle %d\n",
-               giet_proctime() );
-
-        global_init_ok = 1;
-    }
-    else  
-    {
-        while ( global_init_ok == 0 );
-    }
-    
-    ///////////////////////////////////////////////////////////////////////
-    // In each cluster, only task running on processor[x,y,0] allocates 
-    // the local buffers containing the images in the distributed heap
-    // (one buf_in and one buf_out per cluster).
-    // Other processors in cluster wait completion. 
-    ///////////////////////////////////////////////////////////////////////
-
-    if ( lpid == 0 ) 
-    {
-        buf_in[cluster_id]  = remote_malloc( npixels/nclusters, x, y );
-        buf_out[cluster_id] = remote_malloc( npixels/nclusters, x, y );
-
-        if ( (x==0) && (y==0) )
-        printf("\n[TRANSPOSE] Proc [%d,%d,%d] completes buffer allocation"
-               " for cluster[%d,%d] at cycle %d\n"
-               " - buf_in  = %x\n"
-               " - buf_out = %x\n",
-               x, y, lpid, x, y, giet_proctime(), 
-               (unsigned int)buf_in[cluster_id], (unsigned int)buf_out[cluster_id] );
-
-        ///////////////////////////////////////////////////////////////////////
-        // In each cluster, only task running on procesor[x,y,0] open the
-        // three private file descriptors for the three files
-        ///////////////////////////////////////////////////////////////////////
-
-        // open initial file
-        fd_initial = giet_fat_open( INITIAL_FILE_PATH , O_RDONLY );  // read_only
-        if ( fd_initial < 0 ) 
-        { 
-            printf("\n[TRANSPOSE ERROR] Proc [%d,%d,%d] cannot open file %s\n",
-                   x , y , lpid , INITIAL_FILE_PATH );
-            giet_exit(" open() failure");
-        }
-        else if ( (x==0) && (y==0) && (lpid==0) )
-        {
-            printf("\n[TRANSPOSE] Proc [0,0,0] open file %s / fd = %d\n",
-                   INITIAL_FILE_PATH , fd_initial );
-        }
-
-        // open transposed file
-        fd_transposed = giet_fat_open( TRANSPOSED_FILE_PATH , O_CREATE );   // create if required
-        if ( fd_transposed < 0 ) 
-        { 
-            printf("\n[TRANSPOSE ERROR] Proc [%d,%d,%d] cannot open file %s\n",
-                            x , y , lpid , TRANSPOSED_FILE_PATH );
-            giet_exit(" open() failure");
-        }
-        else if ( (x==0) && (y==0) && (lpid==0) )
-        {
-            printf("\n[TRANSPOSE] Proc [0,0,0] open file %s / fd = %d\n",
-                   TRANSPOSED_FILE_PATH , fd_transposed );
-        }
-
-        // open restored file
-        fd_restored = giet_fat_open( RESTORED_FILE_PATH , O_CREATE );   // create if required
-        if ( fd_restored < 0 ) 
-        { 
-            printf("\n[TRANSPOSE ERROR] Proc [%d,%d,%d] cannot open file %s\n",
-                   x , y , lpid , RESTORED_FILE_PATH );
-            giet_exit(" open() failure");
-        }
-        else if ( (x==0) && (y==0) && (lpid==0) )
-        {
-            printf("\n[TRANSPOSE] Proc [0,0,0] open file %s / fd = %d\n",
-                   RESTORED_FILE_PATH , fd_restored );
-        }
-
-        local_init_ok[x][y] = 1;
-    }
-    else
-    {
-        while( local_init_ok[x][y] == 0 );
-    }
-
-    ///////////////////////////////////////////////////////////////////////
-    // Main loop / two iterations:
-    // - first makes  initial    => transposed
-    // - second makes transposed => restored 
-    // All processors execute this main loop.
-    ///////////////////////////////////////////////////////////////////////
-
-    unsigned int fd_in  = fd_initial;
-    unsigned int fd_out = fd_transposed;
-
-    while (iteration < 2)
-    {
-        ///////////////////////////////////////////////////////////////////////
-        // pseudo parallel load from disk to buf_in buffers: npixels/nclusters 
-        // only task running on processor(x,y,0) does it
-        ///////////////////////////////////////////////////////////////////////
-
-        LOAD_START[x][y][lpid] = giet_proctime();
-
-        if (lpid == 0)
-        {
-            unsigned int offset = ((npixels*cluster_id)/nclusters);
-            if ( giet_fat_lseek( fd_in,
-                                 offset,
-                                 SEEK_SET ) != offset )
-            {
-                printf("\n[TRANSPOSE ERROR] Proc [%d,%d,%d] cannot seek fd = %d\n",
-                       x , y , lpid , fd_in );
-                giet_exit(" seek() failure");
-            }
-
-            unsigned int pixels = npixels / nclusters;
-            if ( giet_fat_read( fd_in,
-                                buf_in[cluster_id],
-                                pixels ) != pixels )
-            {
-                printf("\n[TRANSPOSE ERROR] Proc [%d,%d,%d] cannot read fd = %d\n",
-                       x , y , lpid , fd_in );
-                giet_exit(" read() failure");
-            }
-
-            if ( (x==0) && (y==0) )
-            printf("\n[TRANSPOSE] Proc [%d,%d,%d] completes load"
-                   "  for iteration %d at cycle %d\n",
-                   x, y, lpid, iteration, giet_proctime() );
-        }
-
-        LOAD_END[x][y][lpid] = giet_proctime();
-
-        /////////////////////////////
-        sqt_barrier_wait( &barrier );
-
-        ///////////////////////////////////////////////////////////////////////
-        // parallel transpose from buf_in to buf_out
-        // each task makes the transposition for nlt lines (nlt = NN/ntasks)
-        // from line [task_id*nlt] to line [(task_id + 1)*nlt - 1]
-        // (p,l) are the absolute pixel coordinates in the source image
-        ///////////////////////////////////////////////////////////////////////
-
-        TRSP_START[x][y][lpid] = giet_proctime();
-
-        unsigned int nlt   = NN / ntasks;      // number of lines per task
-        unsigned int nlc   = NN / nclusters;   // number of lines per cluster
-
-        unsigned int src_cluster;
-        unsigned int src_index;
-        unsigned int dst_cluster;
-        unsigned int dst_index;
-
-        unsigned char byte;
-
-        unsigned int first = task_id * nlt;    // first line index for a given task
-        unsigned int last  = first + nlt;      // last line index for a given task
-
-        for ( l = first ; l < last ; l++ )
-        {
-            check_line_before[l] = 0;
-         
-            // in each iteration we transfer one byte
-            for ( p = 0 ; p < NN ; p++ )
-            {
-                // read one byte from local buf_in
-                src_cluster = l / nlc;
-                src_index   = (l % nlc)*NN + p;
-                byte        = buf_in[src_cluster][src_index];
-
-                // compute checksum
-                check_line_before[l] = check_line_before[l] + byte;
-
-                // write one byte to remote buf_out
-                dst_cluster = p / nlc; 
-                dst_index   = (p % nlc)*NN + l;
-                buf_out[dst_cluster][dst_index] = byte;
-            }
-        }
-
-//        if ( lpid == 0 )
-        {
-//            if ( (x==0) && (y==0) )
-            printf("\n[TRANSPOSE] proc [%d,%d,0] completes transpose"
-                   " for iteration %d at cycle %d\n", 
-                   x, y, iteration, giet_proctime() );
-
-        }
-        TRSP_END[x][y][lpid] = giet_proctime();
-
-        /////////////////////////////
-        sqt_barrier_wait( &barrier );
-
-        ///////////////////////////////////////////////////////////////////////
-        // parallel display from local buf_out to frame buffer
-        // all tasks contribute to display using memcpy...
-        ///////////////////////////////////////////////////////////////////////
-
-        DISP_START[x][y][lpid] = giet_proctime();
-
-        unsigned int  npt   = npixels / ntasks;   // number of pixels per task
-
-        giet_fbf_sync_write( npt * task_id, 
-                             &buf_out[cluster_id][lpid*npt], 
-                             npt );
-
-//        if ( (x==0) && (y==0) && (lpid==0) )
-        printf("\n[TRANSPOSE] Proc [%d,%d,%d] completes display"
-               " for iteration %d at cycle %d\n",
-               x, y, lpid, iteration, giet_proctime() );
-
-        DISP_END[x][y][lpid] = giet_proctime();
-
-        /////////////////////////////
-        sqt_barrier_wait( &barrier );
-
-        ///////////////////////////////////////////////////////////////////////
-        // pseudo parallel store : buf_out buffers to disk : npixels/nclusters 
-        // only task running on processor(x,y,0) does it
-        ///////////////////////////////////////////////////////////////////////
-
-        STOR_START[x][y][lpid] = giet_proctime();
-
-        if ( lpid == 0 )
-        {
-            unsigned int offset = ((npixels*cluster_id)/nclusters);
-            if ( giet_fat_lseek( fd_out,
-                                 offset,
-                                 SEEK_SET ) != offset )
-            {
-                printf("\n[TRANSPOSE ERROR] Proc [%d,%d,%d] cannot seek fr = %d\n",
-                       x , y , lpid , fd_out );
-                giet_exit(" seek() failure");
-            }
-
-            unsigned int pixels = npixels / nclusters;
-            if ( giet_fat_write( fd_out,
-                                 buf_out[cluster_id],
-                                 pixels ) != pixels )
-            {
-                printf("\n[TRANSPOSE ERROR] Proc [%d,%d,%d] cannot write fd = %d\n",
-                       x , y , lpid , fd_out );
-                giet_exit(" write() failure");
-            }
-
-            if ( (x==0) && (y==0) )
-            printf("\n[TRANSPOSE] Proc [%d,%d,%d] completes store"
-                   "  for iteration %d at cycle %d\n",
-                   x, y, lpid, iteration, giet_proctime() );
-        }
-
-        STOR_END[x][y][lpid] = giet_proctime();
-
-        /////////////////////////////
-        sqt_barrier_wait( &barrier );
-
-        // instrumentation done by processor [0,0,0] 
-        if ( (x==0) && (y==0) && (lpid==0) && INSTRUMENTATION_OK )
-        {
-            int cx , cy , pp ;
-            unsigned int min_load_start = 0xFFFFFFFF;
-            unsigned int max_load_start = 0;
-            unsigned int min_load_ended = 0xFFFFFFFF;
-            unsigned int max_load_ended = 0;
-            unsigned int min_trsp_start = 0xFFFFFFFF;
-            unsigned int max_trsp_start = 0;
-            unsigned int min_trsp_ended = 0xFFFFFFFF;
-            unsigned int max_trsp_ended = 0;
-            unsigned int min_disp_start = 0xFFFFFFFF;
-            unsigned int max_disp_start = 0;
-            unsigned int min_disp_ended = 0xFFFFFFFF;
-            unsigned int max_disp_ended = 0;
-            unsigned int min_stor_start = 0xFFFFFFFF;
-            unsigned int max_stor_start = 0;
-            unsigned int min_stor_ended = 0xFFFFFFFF;
-            unsigned int max_stor_ended = 0;
-
-            for (cx = 0; cx < x_size; cx++)
-            {
-            for (cy = 0; cy < y_size; cy++)
-            {
-            for (pp = 0; pp < NB_PROCS_MAX; pp++)
-            {
-                if (LOAD_START[cx][cy][pp] < min_load_start)  min_load_start = LOAD_START[cx][cy][pp];
-                if (LOAD_START[cx][cy][pp] > max_load_start)  max_load_start = LOAD_START[cx][cy][pp];
-                if (LOAD_END[cx][cy][pp]   < min_load_ended)  min_load_ended = LOAD_END[cx][cy][pp]; 
-                if (LOAD_END[cx][cy][pp]   > max_load_ended)  max_load_ended = LOAD_END[cx][cy][pp];
-                if (TRSP_START[cx][cy][pp] < min_trsp_start)  min_trsp_start = TRSP_START[cx][cy][pp];
-                if (TRSP_START[cx][cy][pp] > max_trsp_start)  max_trsp_start = TRSP_START[cx][cy][pp];
-                if (TRSP_END[cx][cy][pp]   < min_trsp_ended)  min_trsp_ended = TRSP_END[cx][cy][pp];
-                if (TRSP_END[cx][cy][pp]   > max_trsp_ended)  max_trsp_ended = TRSP_END[cx][cy][pp];
-                if (DISP_START[cx][cy][pp] < min_disp_start)  min_disp_start = DISP_START[cx][cy][pp];
-                if (DISP_START[cx][cy][pp] > max_disp_start)  max_disp_start = DISP_START[cx][cy][pp];
-                if (DISP_END[cx][cy][pp]   < min_disp_ended)  min_disp_ended = DISP_END[cx][cy][pp];
-                if (DISP_END[cx][cy][pp]   > max_disp_ended)  max_disp_ended = DISP_END[cx][cy][pp];
-                if (STOR_START[cx][cy][pp] < min_stor_start)  min_stor_start = STOR_START[cx][cy][pp];
-                if (STOR_START[cx][cy][pp] > max_stor_start)  max_stor_start = STOR_START[cx][cy][pp];
-                if (STOR_END[cx][cy][pp]   < min_stor_ended)  min_stor_ended = STOR_END[cx][cy][pp];
-                if (STOR_END[cx][cy][pp]   > max_stor_ended)  max_stor_ended = STOR_END[cx][cy][pp];
-            }
-            }
-            }
-
-            printf("\n   ---------------- Instrumentation Results ---------------------\n");
-
-            printf(" - LOAD_START : min = %d / max = %d / med = %d / delta = %d\n",
-                   min_load_start, max_load_start, (min_load_start+max_load_start)/2, 
-                   max_load_start-min_load_start); 
-
-            printf(" - LOAD_END   : min = %d / max = %d / med = %d / delta = %d\n",
-                   min_load_ended, max_load_ended, (min_load_ended+max_load_ended)/2, 
-                   max_load_ended-min_load_ended); 
-
-            printf(" - TRSP_START : min = %d / max = %d / med = %d / delta = %d\n",
-                   min_trsp_start, max_trsp_start, (min_trsp_start+max_trsp_start)/2, 
-                   max_trsp_start-min_trsp_start); 
-
-            printf(" - TRSP_END   : min = %d / max = %d / med = %d / delta = %d\n",
-                   min_trsp_ended, max_trsp_ended, (min_trsp_ended+max_trsp_ended)/2, 
-                   max_trsp_ended-min_trsp_ended); 
-
-            printf(" - DISP_START : min = %d / max = %d / med = %d / delta = %d\n",
-                   min_disp_start, max_disp_start, (min_disp_start+max_disp_start)/2, 
-                   max_disp_start-min_disp_start); 
-
-            printf(" - DISP_END   : min = %d / max = %d / med = %d / delta = %d\n",
-                   min_disp_ended, max_disp_ended, (min_disp_ended+max_disp_ended)/2, 
-                   max_disp_ended-min_disp_ended); 
-
-            printf(" - STOR_START : min = %d / max = %d / med = %d / delta = %d\n",
-                   min_stor_start, max_stor_start, (min_stor_start+max_stor_start)/2, 
-                   max_stor_start-min_stor_start); 
-
-            printf(" - STOR_END   : min = %d / max = %d / med = %d / delta = %d\n",
-                   min_stor_ended, max_stor_ended, (min_stor_ended+max_stor_ended)/2, 
-                   max_stor_ended-min_stor_ended); 
-        }
-
-        /////////////////////////////
-        sqt_barrier_wait( &barrier );
-
-        // update iteration variables
-        fd_in  = fd_transposed;
-        fd_out = fd_restored;
-        iteration++;
-
-    } // end while      
-
-    ///////////////////////////////////////////////////////////////////////
-    // In each cluster, only task running on Processor[x,y,0] releases
-    // the distributed buffers and close the file descriptors.
-    ///////////////////////////////////////////////////////////////////////
-
-    if ( lpid==0 )
-    {
-        free( buf_in[cluster_id] );
-        free( buf_out[cluster_id] );
-
-        giet_fat_close( fd_initial );
-        giet_fat_close( fd_transposed );
-        giet_fat_close( fd_restored );
-    }
-
-    giet_exit("Completed");
-
-} // end main()
-
-// Local Variables:
-// tab-width: 3
-// c-basic-offset: 
-// c-file-offsets:((innamespace . 0)(inline-open . 0))
-// indent-tabs-mode: nil
-// End:
-
-// vim: filetype=cpp:expandtab:shiftwidth=3:tabstop=3:softtabstop=3
-
-
-
Index: soft/giet_vm/applications/transpose/transpose.c
===================================================================
--- soft/giet_vm/applications/transpose/transpose.c	(revision 708)
+++ soft/giet_vm/applications/transpose/transpose.c	(revision 708)
@@ -0,0 +1,504 @@
+///////////////////////////////////////////////////////////////////////////////////////
+// File   : transpose.c   
+// Date   : september 2015
+// author : Alain Greiner
+///////////////////////////////////////////////////////////////////////////////////////
+// This multi-threaded aplication transposes a raw image (one pbyte per pixel).
+// It can run on a multi-processors, multi-clusters architecture, with one thread
+// per processor, and uses the POSIX threads API. 
+//
+// The main() function can be launched on any processor P[x,y,l].
+// It makes the initialisations, launch (N-1) threads to run the execute() function
+// on the (N-1) other processors than P[x,y,l], call himself the execute() function, 
+// and finally call the instrument() function to display instrumentation results 
+// when the parallel execution is completed.
+//
+// The input and output buffers containing the image are distributed in clusters.
+//
+// The execute() function read a set of lines from an input file on disk,
+// to the local buffer buf_in[x][y], transpose it, write the result to a remote buffer 
+// buf_out[x'][y'], display the content of the local buffer buf_out[x][y] to the
+// frame buffer, and store it on disk to another output file. 
+//
+// - The image size must fit the frame buffer size.
+// - The block size in block device must be 512 bytes.
+// - The number of clusters  must be a power of 2 no larger than 256.
+// - The number of processors per cluster must be a power of 2 no larger than 4.
+///////////////////////////////////////////////////////////////////////////////////////
+
+#include "stdio.h"
+#include "user_barrier.h"
+#include "malloc.h"
+
+#define BLOCK_SIZE            512                         // block size on disk
+#define X_MAX                 16                          // max number of clusters in row
+#define Y_MAX                 16                          // max number of clusters in column
+#define PROCS_MAX             4                           // max number of procs per cluster
+#define CLUSTER_MAX           (X_MAX * Y_MAX)             // max number of clusters
+#define IMAGE_SIZE            256                         // image size : nlines = npixels
+#define INPUT_FILE_PATH       "/misc/lena_256.raw"        // pathname on virtual disk
+#define OUTPUT_FILE_PATH      "/home/lena_transposed.raw" // pathname on virtual disk
+
+// macro to use a shared TTY
+#define printf(...);    { lock_acquire( &tty_lock ); \
+                          giet_tty_printf(__VA_ARGS__);  \
+                          lock_release( &tty_lock ); }
+
+///////////////////////////////////////////////////////
+// global variables stored in seg_data in cluster(0,0)
+///////////////////////////////////////////////////////
+
+// instrumentation counters for each processor in each cluster 
+unsigned int LOAD_START[X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
+unsigned int LOAD_END  [X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
+unsigned int TRSP_START[X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
+unsigned int TRSP_END  [X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
+unsigned int DISP_START[X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
+unsigned int DISP_END  [X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
+unsigned int STOR_START[X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
+unsigned int STOR_END  [X_MAX][Y_MAX][PROCS_MAX] = {{{ 0 }}};
+
+// arrays of pointers on distributed buffers
+// one input buffer & one output buffer per cluster
+unsigned char*  buf_in [CLUSTER_MAX];
+unsigned char*  buf_out[CLUSTER_MAX];
+
+// checksum variables 
+unsigned check_line_before[IMAGE_SIZE];
+unsigned check_line_after[IMAGE_SIZE];
+
+// lock protecting shared TTY
+user_lock_t  tty_lock;
+
+// synchronisation barrier (all threads)
+giet_sqt_barrier_t barrier;
+
+////////////////////////////////////////////
+__attribute__ ((constructor)) void execute()
+////////////////////////////////////////////
+{
+    unsigned int l;                            // line index for loops
+    unsigned int p;                            // pixel index for loops
+
+    // get processor identifiers 
+    unsigned int x_id;                         // x cluster coordinate
+    unsigned int y_id;                         // y cluster coordinate
+    unsigned int p_id;                         // local processor index
+
+    giet_proc_xyp( &x_id, &y_id, &p_id);             
+
+    // get & check plat-form parameters
+    unsigned int x_size;                       // number of clusters in a row
+    unsigned int y_size;                       // number of clusters in a column
+    unsigned int nprocs;                       // number of processors per cluster
+    
+    giet_procs_number( &x_size , &y_size , &nprocs );
+
+    unsigned int nclusters     = x_size * y_size;               // number of clusters
+    unsigned int nthreads      = x_size * y_size * nprocs;      // number of threads
+    unsigned int npixels       = IMAGE_SIZE * IMAGE_SIZE;       // pixels per image
+    int          fd_in         = 0;                             // initial file descriptor
+    int          fd_out        = 0;                             // output file descriptor
+    unsigned int cluster_id    = (x_id * y_size) + y_id;        // "continuous" index   
+    unsigned int thread_id     = (cluster_id * nprocs) + p_id;  // "continuous" thread index
+
+    // parallel load of image:
+    // allocate buf_in and buf_out distributed buffers (one buf_in & one buf_out per cluster). 
+    // open input and output files, and load the relevant lines in local buf_in.
+    // only thread running on processor[x,y,0] does it.
+
+    LOAD_START[x_id][y_id][p_id] = giet_proctime();
+
+    if ( p_id == 0 ) 
+    {
+        buf_in[cluster_id]  = remote_malloc( npixels/nclusters, x_id, y_id );
+        buf_out[cluster_id] = remote_malloc( npixels/nclusters, x_id, y_id );
+
+        if ( (x_id==0) && (y_id==0) )
+        {
+            printf("\n[TRANSPOSE] Proc [%d,%d,%d] completes buffer allocation at cycle %d\n",
+                   x_id, y_id, p_id, giet_proctime() );
+        }
+
+        // open input file
+        fd_in = giet_fat_open( INPUT_FILE_PATH , O_RDONLY );  // read_only
+        if ( fd_in < 0 ) 
+        { 
+            printf("\n[TRANSPOSE ERROR] Proc [%d,%d,%d] cannot open file %s\n",
+                   x_id , y_id , p_id , INPUT_FILE_PATH );
+            giet_pthread_exit(" open() failure");
+        }
+        else if ( (x_id==0) && (y_id==0) )
+        {
+            printf("\n[TRANSPOSE] Proc [0,0,0] open file %s / fd = %d\n",
+                   INPUT_FILE_PATH , fd_in );
+        }
+
+        // open output file
+        fd_out = giet_fat_open( OUTPUT_FILE_PATH , O_CREATE );   // create if required
+        if ( fd_out < 0 ) 
+        { 
+            printf("\n[TRANSPOSE ERROR] Proc [%d,%d,%d] cannot open file %s\n",
+                            x_id , y_id , p_id , OUTPUT_FILE_PATH );
+            giet_pthread_exit(" open() failure");
+        }
+        else if ( (x_id==0) && (y_id==0) )
+        {
+            printf("\n[TRANSPOSE] Proc [0,0,0] open file %s / fd = %d\n",
+                   OUTPUT_FILE_PATH , fd_out );
+        }
+
+
+        unsigned int offset = ((npixels*cluster_id)/nclusters);
+        if ( giet_fat_lseek( fd_in,
+                             offset,
+                             SEEK_SET ) != offset )
+        {
+            printf("\n[TRANSPOSE ERROR] Proc [%d,%d,%d] cannot seek fd = %d\n",
+                   x_id , y_id , p_id , fd_in );
+            giet_pthread_exit(" seek() failure");
+        }
+
+        unsigned int pixels = npixels / nclusters;
+        if ( giet_fat_read( fd_in,
+                            buf_in[cluster_id],
+                            pixels ) != pixels )
+        {
+            printf("\n[TRANSPOSE ERROR] Proc [%d,%d,%d] cannot read fd = %d\n",
+                   x_id , y_id , p_id , fd_in );
+            giet_pthread_exit(" read() failure");
+        }
+
+        if ( (x_id==0) && (y_id==0) )
+        {
+            printf("\n[TRANSPOSE] Proc [%d,%d,%d] completes load at cycle %d\n",
+                   x_id, y_id, p_id, giet_proctime() );
+        }
+    }
+
+    LOAD_END[x_id][y_id][p_id] = giet_proctime();
+
+    /////////////////////////////
+    sqt_barrier_wait( &barrier );
+    /////////////////////////////
+
+    // parallel transpose from buf_in to buf_out
+    // each thread makes the transposition for nlt lines (nlt = IMAGE_SIZE/nthreads)
+    // from line [thread_id*nlt] to line [(thread_id + 1)*nlt - 1]
+    // (p,l) are the absolute pixel coordinates in the source image
+
+    TRSP_START[x_id][y_id][p_id] = giet_proctime();
+
+    unsigned int nlt   = IMAGE_SIZE / nthreads;    // number of lines per thread
+    unsigned int nlc   = IMAGE_SIZE / nclusters;   // number of lines per cluster
+
+    unsigned int src_cluster;
+    unsigned int src_index;
+    unsigned int dst_cluster;
+    unsigned int dst_index;
+
+    unsigned char byte;
+
+    unsigned int first = thread_id * nlt;  // first line index for a given thread
+    unsigned int last  = first + nlt;      // last line index for a given thread
+
+    for ( l = first ; l < last ; l++ )
+    {
+        check_line_before[l] = 0;
+     
+        // in each iteration we transfer one byte
+        for ( p = 0 ; p < IMAGE_SIZE ; p++ )
+        {
+            // read one byte from local buf_in
+            src_cluster = l / nlc;
+            src_index   = (l % nlc)*IMAGE_SIZE + p;
+            byte        = buf_in[src_cluster][src_index];
+
+            // compute checksum
+            check_line_before[l] = check_line_before[l] + byte;
+
+            // write one byte to remote buf_out
+            dst_cluster = p / nlc; 
+            dst_index   = (p % nlc)*IMAGE_SIZE + l;
+            buf_out[dst_cluster][dst_index] = byte;
+        }
+    }
+
+    if ( (p_id == 0) && (x_id==0) && (y_id==0) )
+    {
+        printf("\n[TRANSPOSE] proc [%d,%d,%d] completes transpose at cycle %d\n", 
+        x_id, y_id, p_id, giet_proctime() );
+    }
+
+    TRSP_END[x_id][y_id][p_id] = giet_proctime();
+
+    /////////////////////////////
+    sqt_barrier_wait( &barrier );
+    /////////////////////////////
+
+    // parallel display from local buf_out to frame buffer
+    // all threads contribute to display using memcpy...
+
+    DISP_START[x_id][y_id][p_id] = giet_proctime();
+
+    unsigned int  npt   = npixels / nthreads;   // number of pixels per thread
+
+    giet_fbf_sync_write( npt * thread_id, 
+                         &buf_out[cluster_id][p_id*npt], 
+                         npt );
+
+    if ( (x_id==0) && (y_id==0) && (p_id==0) )
+    {
+        printf("\n[TRANSPOSE] Proc [%d,%d,%d] completes display at cycle %d\n",
+               x_id, y_id, p_id, giet_proctime() );
+    }
+
+    DISP_END[x_id][y_id][p_id] = giet_proctime();
+
+    /////////////////////////////
+    sqt_barrier_wait( &barrier );
+    /////////////////////////////
+
+    // parallel store : buf_out buffers to disk
+    // only thread running on processor(x,y,0) does it
+
+    STOR_START[x_id][y_id][p_id] = giet_proctime();
+
+    if ( p_id == 0 )
+    {
+        unsigned int offset = ((npixels*cluster_id)/nclusters);
+        if ( giet_fat_lseek( fd_out,
+                             offset,
+                             SEEK_SET ) != offset )
+        {
+            printf("\n[TRANSPOSE ERROR] Proc [%d,%d,%d] cannot seek fr = %d\n",
+                   x_id , y_id , p_id , fd_out );
+            giet_pthread_exit(" seek() failure");
+        }
+
+        unsigned int pixels = npixels / nclusters;
+        if ( giet_fat_write( fd_out,
+                             buf_out[cluster_id],
+                             pixels ) != pixels )
+        {
+            printf("\n[TRANSPOSE ERROR] Proc [%d,%d,%d] cannot write fd = %d\n",
+                   x_id , y_id , p_id , fd_out );
+            giet_pthread_exit(" write() failure");
+        }
+
+        if ( (x_id==0) && (y_id==0) )
+        {
+            printf("\n[TRANSPOSE] Proc [%d,%d,%d] completes store at cycle %d\n",
+                   x_id, y_id, p_id, giet_proctime() );
+        }
+    }
+
+    STOR_END[x_id][y_id][p_id] = giet_proctime();
+
+    // In each cluster, only thread running on Processor[x,y,0] releases
+    // the distributed buffers and close the file descriptors.
+
+    if ( p_id==0 )
+    {
+        free( buf_in[cluster_id] );
+        free( buf_out[cluster_id] );
+
+        giet_fat_close( fd_in );
+        giet_fat_close( fd_out );
+    }
+
+    if ( (x_id != 0) || (y_id != 0) || (p_id != 0) ) 
+    giet_pthread_exit( "completed" );
+
+} // end execute()
+
+
+
+//////////////////////////////////////
+void instrument( unsigned int x_size,
+                 unsigned int y_size,
+                 unsigned int nprocs )
+//////////////////////////////////////
+{
+    unsigned int x, y, l;
+
+    unsigned int min_load_start = 0xFFFFFFFF;
+    unsigned int max_load_start = 0;
+    unsigned int min_load_ended = 0xFFFFFFFF;
+    unsigned int max_load_ended = 0;
+    unsigned int min_trsp_start = 0xFFFFFFFF;
+    unsigned int max_trsp_start = 0;
+    unsigned int min_trsp_ended = 0xFFFFFFFF;
+    unsigned int max_trsp_ended = 0;
+    unsigned int min_disp_start = 0xFFFFFFFF;
+    unsigned int max_disp_start = 0;
+    unsigned int min_disp_ended = 0xFFFFFFFF;
+    unsigned int max_disp_ended = 0;
+    unsigned int min_stor_start = 0xFFFFFFFF;
+    unsigned int max_stor_start = 0;
+    unsigned int min_stor_ended = 0xFFFFFFFF;
+    unsigned int max_stor_ended = 0;
+
+    for (x = 0; x < x_size; x++)
+    {
+        for (y = 0; y < y_size; y++)
+        {
+            for ( l = 0 ; l < nprocs ; l++ )
+            {
+                if (LOAD_START[x][y][l] < min_load_start)  min_load_start = LOAD_START[x][y][l];
+                if (LOAD_START[x][y][l] > max_load_start)  max_load_start = LOAD_START[x][y][l];
+                if (LOAD_END[x][y][l]   < min_load_ended)  min_load_ended = LOAD_END[x][y][l]; 
+                if (LOAD_END[x][y][l]   > max_load_ended)  max_load_ended = LOAD_END[x][y][l];
+                if (TRSP_START[x][y][l] < min_trsp_start)  min_trsp_start = TRSP_START[x][y][l];
+                if (TRSP_START[x][y][l] > max_trsp_start)  max_trsp_start = TRSP_START[x][y][l];
+                if (TRSP_END[x][y][l]   < min_trsp_ended)  min_trsp_ended = TRSP_END[x][y][l];
+                if (TRSP_END[x][y][l]   > max_trsp_ended)  max_trsp_ended = TRSP_END[x][y][l];
+                if (DISP_START[x][y][l] < min_disp_start)  min_disp_start = DISP_START[x][y][l];
+                if (DISP_START[x][y][l] > max_disp_start)  max_disp_start = DISP_START[x][y][l];
+                if (DISP_END[x][y][l]   < min_disp_ended)  min_disp_ended = DISP_END[x][y][l];
+                if (DISP_END[x][y][l]   > max_disp_ended)  max_disp_ended = DISP_END[x][y][l];
+                if (STOR_START[x][y][l] < min_stor_start)  min_stor_start = STOR_START[x][y][l];
+                if (STOR_START[x][y][l] > max_stor_start)  max_stor_start = STOR_START[x][y][l];
+                if (STOR_END[x][y][l]   < min_stor_ended)  min_stor_ended = STOR_END[x][y][l];
+                if (STOR_END[x][y][l]   > max_stor_ended)  max_stor_ended = STOR_END[x][y][l];
+            }
+        }
+    }
+
+    printf("\n   ---------------- Instrumentation Results ---------------------\n");
+
+    printf(" - LOAD_START : min = %d / max = %d / med = %d / delta = %d\n",
+           min_load_start, max_load_start, (min_load_start+max_load_start)/2, 
+           max_load_start-min_load_start); 
+
+    printf(" - LOAD_END   : min = %d / max = %d / med = %d / delta = %d\n",
+           min_load_ended, max_load_ended, (min_load_ended+max_load_ended)/2, 
+           max_load_ended-min_load_ended); 
+
+    printf(" - TRSP_START : min = %d / max = %d / med = %d / delta = %d\n",
+           min_trsp_start, max_trsp_start, (min_trsp_start+max_trsp_start)/2, 
+           max_trsp_start-min_trsp_start); 
+
+    printf(" - TRSP_END   : min = %d / max = %d / med = %d / delta = %d\n",
+           min_trsp_ended, max_trsp_ended, (min_trsp_ended+max_trsp_ended)/2, 
+           max_trsp_ended-min_trsp_ended); 
+
+    printf(" - DISP_START : min = %d / max = %d / med = %d / delta = %d\n",
+           min_disp_start, max_disp_start, (min_disp_start+max_disp_start)/2, 
+           max_disp_start-min_disp_start); 
+
+    printf(" - DISP_END   : min = %d / max = %d / med = %d / delta = %d\n",
+           min_disp_ended, max_disp_ended, (min_disp_ended+max_disp_ended)/2, 
+           max_disp_ended-min_disp_ended); 
+
+    printf(" - STOR_START : min = %d / max = %d / med = %d / delta = %d\n",
+           min_stor_start, max_stor_start, (min_stor_start+max_stor_start)/2, 
+           max_stor_start-min_stor_start); 
+
+    printf(" - STOR_END   : min = %d / max = %d / med = %d / delta = %d\n",
+           min_stor_ended, max_stor_ended, (min_stor_ended+max_stor_ended)/2, 
+           max_stor_ended-min_stor_ended); 
+
+}  // end instrument()
+
+
+
+//////////////////////////////////////////
+__attribute__ ((constructor)) void main()
+//////////////////////////////////////////
+{
+    // indexes for loops
+    unsigned int x , y , n;
+
+    // get identifiers for proc executing main
+    unsigned int x_id;                          // x cluster coordinate
+    unsigned int y_id;                          // y cluster coordinate
+    unsigned int p_id;                          // local processor index
+
+    giet_proc_xyp( &x_id , &y_id , &p_id );
+
+    // get & check plat-form parameters
+    unsigned int x_size;                       // number of clusters in a row
+    unsigned int y_size;                       // number of clusters in a column
+    unsigned int nprocs;                       // number of processors per cluster
+
+    giet_procs_number( &x_size , &y_size , &nprocs );
+
+    giet_pthread_assert( ((nprocs == 1) || (nprocs == 2) || (nprocs == 4)),
+                         "[TRANSPOSE ERROR] number of procs per cluster must be 1, 2 or 4");
+
+    giet_pthread_assert( ((x_size == 1) || (x_size == 2) || (x_size == 4) || 
+                  (x_size == 8) || (x_size == 16)),
+                         "[TRANSPOSE ERROR] x_size must be 1,2,4,8,16");
+
+    giet_pthread_assert( ((y_size == 1) || (y_size == 2) || (y_size == 4) || 
+                  (y_size == 8) || (y_size == 16)),
+                         "[TRANSPOSE ERROR] y_size must be 1,2,4,8,16");
+
+    giet_pthread_assert( (nprocs * x_size * y_size <= IMAGE_SIZE ),
+                         "[TRANSPOSE ERROR] number of threads larger than number of lines");
+
+    unsigned int nthreads = x_size * y_size * nprocs;
+
+    // shared TTY allocation
+    giet_tty_alloc( 1 );     
+    lock_init( &tty_lock);
+
+    printf("\n[TRANSPOSE] start at cycle %d on %d cores\n", giet_proctime(), nthreads );
+
+    // distributed heap initialisation
+    for ( x = 0 ; x < x_size ; x++ ) 
+    {
+        for ( y = 0 ; y < y_size ; y++ ) 
+        {
+            heap_init( x , y );
+        }
+    }
+
+    // allocate thread[] array
+    pthread_t* thread = malloc( nthreads * sizeof(pthread_t) );
+
+    // barrier initialisation
+    sqt_barrier_init( &barrier, x_size , y_size , nprocs );
+
+    // Initialisation completed
+    printf("\n[TRANSPOSE] initialisation completed at cycle %d\n", giet_proctime() );
+    
+    // launch other threads to run execute() function
+    for ( n = 1 ; n < nthreads ; n++ )
+    {
+        if ( giet_pthread_create( &thread[n],
+                                  NULL,                  // no attribute
+                                  &execute,
+                                  NULL ) )               // no argument
+        {
+            printf("\n[TRANSPOSE ERROR] creating thread %x\n", thread[n] );
+            giet_pthread_exit( NULL );
+        }
+    }
+
+    // run the execute() function
+    execute();
+
+    // wait other threads completion
+    for ( n = 1 ; n < nthreads ; n++ )
+    {
+        if ( giet_pthread_join( thread[n], NULL ) )
+        {
+            printf("\n[TRANSPOSE ERROR] joining thread %x\n", thread[n] );
+            giet_pthread_exit( NULL );
+        }
+        else
+        {
+            printf("\n[TRANSPOSE] thread %x joined at cycle %d\n",
+                   thread[n] , giet_proctime() );
+        }
+    }
+
+    // call the instrument() function
+    instrument( x_size , y_size , nprocs );
+
+    giet_pthread_exit( "completed" );
+    
+} // end main()
+
Index: soft/giet_vm/applications/transpose/transpose.py
===================================================================
--- soft/giet_vm/applications/transpose/transpose.py	(revision 707)
+++ soft/giet_vm/applications/transpose/transpose.py	(revision 708)
@@ -10,7 +10,5 @@
 #  This file describes the mapping of the multi-threaded "transpose" 
 #  application on a multi-clusters, multi-processors architecture.
-#  This include both the mapping of virtual segments on the clusters,
-#  and the mapping of tasks on processors.
-#  There is one task per processor.
+#  There is one thread per processor.
 #  The mapping of virtual segments is the following:
 #    - There is one shared data vseg in cluster[0][0]
@@ -37,5 +35,5 @@
     # define vsegs base & size
     code_base  = 0x10000000
-    code_size  = 0x00010000     # 64 Kbytes (replicated in each cluster)
+    code_size  = 0x00010000     # 64 Kbytes (per cluster)
     
     data_base  = 0x20000000
@@ -43,5 +41,5 @@
 
     stack_base = 0x40000000 
-    stack_size = 0x00200000     # 2 Mbytes (per cluster)
+    stack_size = 0x00010000     # 64 Kbytes (per thread)
 
     heap_base  = 0x60000000 
@@ -76,9 +74,8 @@
                 for p in xrange( nprocs ):
                     proc_id = (((x * y_size) + y) * nprocs) + p
-                    size    = (stack_size / nprocs) & 0xFFFFF000
-                    base    = stack_base + (proc_id * size)
+                    base    = stack_base + (proc_id * stack_size)
 
                     mapping.addVseg( vspace, 'trsp_stack_%d_%d_%d' % (x,y,p), 
-                                     base, size, 'C_WU', vtype = 'BUFFER', 
+                                     base, stack_size, 'C_WU', vtype = 'BUFFER', 
                                      x = x , y = y , pseg = 'RAM',
                                      local = True, big = True )
@@ -96,5 +93,5 @@
                                  local = False, big = True )
 
-    # distributed tasks / one task per processor
+    # distribute one thread per processor / main on P[0,0,0]
     for x in xrange (x_size):
         for y in xrange (y_size):
@@ -102,10 +99,18 @@
             if ( mapping.clusters[cluster_id].procs ):
                 for p in xrange( nprocs ):
-                    trdid = (((x * y_size) + y) * nprocs) + p
+                    if (x == 0) and (y == 0) and (p == 0) :   # main thread
+                        startid = 1
+                        is_main = True
+                    else :                                    # other threads
+                        startid = 0
+                        is_main = False
 
-                    mapping.addTask( vspace, 'trsp_%d_%d_%d' % (x,y,p),
-                                     trdid, x, y, p,
-                                     'trsp_stack_%d_%d_%d' % (x,y,p),
-                                     'trsp_heap_%d_%d' % (x,y), 0 )
+                    mapping.addThread( vspace,
+                                       'trsp_%d_%d_%d' % (x,y,p),
+                                       is_main,
+                                       x, y, p,
+                                       'trsp_stack_%d_%d_%d' % (x,y,p),
+                                       'trsp_heap_%d_%d' % (x,y),
+                                       startid )
 
     # extend mapping name
